This runbook covers daily operations, common tasks, and procedures for Mantissa Log.
# 1. Check system health
bash scripts/smoke-test.sh terraform-outputs.json
# 2. Review overnight alerts
aws logs filter-log-events \
--log-group-name /aws/lambda/mantissa-log-detection-engine \
--start-time $(($(date -d 'yesterday' +%s) * 1000)) \
--filter-pattern "[time, request_id, level=ERROR, ...]"
# 3. Check alert delivery
aws logs tail /aws/lambda/mantissa-log-alert-router --since 24h \
| grep "Alert sent"
# 4. Monitor costs
aws ce get-cost-and-usage \
--time-period Start=$(date -d '7 days ago' +%Y-%m-%d),End=$(date +%Y-%m-%d) \
--granularity DAILY \
--metrics BlendedCost- Review new detection rules
- Analyze alert trends
- Check for false positives
- Update rule thresholds
- Rotate API keys (if applicable)
- Review CloudWatch dashboards
- Check S3 storage costs
- Review and tune detection rules
- Audit user access (Cognito)
- Review compliance logs
- Update documentation
- Plan capacity changes
- Review cost optimization opportunities
- Test disaster recovery procedures
Create Cognito user for API access:
USER_POOL_ID=$(cat terraform-outputs.json | jq -r '.user_pool_id.value')
USER_EMAIL="newuser@company.com"
TEMP_PASSWORD="TempPass123!"
# Create user
aws cognito-idp admin-create-user \
--user-pool-id $USER_POOL_ID \
--username $USER_EMAIL \
--user-attributes Name=email,Value=$USER_EMAIL Name=email_verified,Value=true \
--temporary-password $TEMP_PASSWORD \
--message-action SUPPRESS
# Set permanent password
aws cognito-idp admin-set-user-password \
--user-pool-id $USER_POOL_ID \
--username $USER_EMAIL \
--password "SecurePass123!" \
--permanentAnalyze rule performance:
# Check how often each rule triggers
aws logs filter-log-events \
--log-group-name /aws/lambda/mantissa-log-detection-engine \
--start-time $(($(date -d '7 days ago' +%s) * 1000)) \
--filter-pattern '"rule_name"' \
| jq '.events[].message' | jq -r '.rule_name' | sort | uniq -c | sort -nrAdjust threshold:
# Edit rule file
vim rules/aws/cloudtrail/failed-logins.yaml
# Change threshold
threshold:
count: 10 # Increased from 5
window: "5m"
# Upload updated rule
RULES_BUCKET=$(cat terraform-outputs.json | jq -r '.rules_bucket.value')
aws s3 cp rules/aws/cloudtrail/failed-logins.yaml \
s3://$RULES_BUCKET/rules/aws/cloudtrail/View recent alerts:
# Get alerts from DynamoDB
STATE_TABLE=$(cat terraform-outputs.json | jq -r '.state_table_name.value')
aws dynamodb scan \
--table-name $STATE_TABLE \
--filter-expression "begins_with(alert_id, :prefix)" \
--expression-attribute-values '{":prefix":{"S":"alert-"}}' \
--limit 10Deep dive into specific alert:
ALERT_ID="alert-20240115-001"
# Get alert details
aws dynamodb get-item \
--table-name $STATE_TABLE \
--key "{\"alert_id\":{\"S\":\"$ALERT_ID\"}}"
# Find related logs
aws logs filter-log-events \
--log-group-name /aws/lambda/mantissa-log-detection-engine \
--filter-pattern "$ALERT_ID"Query evidence:
# Use natural language query
curl -X POST "$API_ENDPOINT/query" \
-H "Authorization: Bearer $TOKEN" \
-H "Content-Type: application/json" \
-d '{
"question": "Show me all events from IP 203.0.113.42 in the last hour",
"execute": true
}'Update Lambda functions:
# Package new code
bash scripts/package-lambdas.sh
# Update functions
DETECTION_ENGINE=$(cat terraform-outputs.json | jq -r '.detection_engine_function_name.value')
aws lambda update-function-code \
--function-name $DETECTION_ENGINE \
--zip-file fileb://build/lambda/detection-engine.zipAdd partitions to tables:
DATABASE=$(cat terraform-outputs.json | jq -r '.database_name.value')
WORKGROUP=$(cat terraform-outputs.json | jq -r '.athena_workgroup_name.value')
# Add today's partition
YEAR=$(date +%Y)
MONTH=$(date +%m)
DAY=$(date +%d)
aws athena start-query-execution \
--query-string "MSCK REPAIR TABLE cloudtrail" \
--query-execution-context Database=$DATABASE \
--work-group $WORKGROUPClean old data:
# Remove old Athena query results
ATHENA_BUCKET=$(cat terraform-outputs.json | jq -r '.athena_results_bucket.value')
aws s3 rm s3://$ATHENA_BUCKET/ --recursive \
--exclude "*" \
--include "*/$(date -d '30 days ago' +%Y/%m/%d)/*"Detection Engine:
- Execution success rate
- Query execution time
- Rules processed per cycle
- Alerts generated
Alert Router:
- Delivery success rate
- Destination failures
- Routing latency
LLM Query Handler:
- Query generation success rate
- LLM API latency
- SQL validation failures
Create monitoring dashboard:
cat > dashboard.json << 'EOF'
{
"widgets": [
{
"type": "metric",
"properties": {
"metrics": [
["AWS/Lambda", "Invocations", {"stat": "Sum"}],
[".", "Errors", {"stat": "Sum"}],
[".", "Duration", {"stat": "Average"}]
],
"period": 300,
"region": "us-east-1",
"title": "Lambda Metrics"
}
}
]
}
EOF
aws cloudwatch put-dashboard \
--dashboard-name mantissa-log-ops \
--dashboard-body file://dashboard.jsonDetection engine failures:
aws cloudwatch put-metric-alarm \
--alarm-name mantissa-log-detection-errors \
--alarm-description "Alert on detection engine errors" \
--metric-name Errors \
--namespace AWS/Lambda \
--statistic Sum \
--period 300 \
--evaluation-periods 2 \
--threshold 5 \
--comparison-operator GreaterThanThreshold \
--dimensions Name=FunctionName,Value=mantissa-log-detection-engine \
--alarm-actions arn:aws:sns:us-east-1:123456789012:ops-alertsHigh query costs:
aws cloudwatch put-metric-alarm \
--alarm-name mantissa-log-high-athena-cost \
--alarm-description "Alert on high Athena costs" \
--metric-name DataScannedInBytes \
--namespace AWS/Athena \
--statistic Sum \
--period 86400 \
--evaluation-periods 1 \
--threshold 1000000000000 \
--comparison-operator GreaterThanThresholdMantissa Log continuously monitors the health of all configured log sources. Two scheduled functions run automatically:
- Health Check (every 5 minutes): Evaluates each enabled source for latency, silence, volume anomalies, and data gaps. Generates and routes alerts on status transitions.
- Baseline Computation (daily): Computes hourly volume baselines from the data lake for z-score anomaly detection.
| Status | Meaning | Operator Action |
|---|---|---|
HEALTHY |
Data flowing within expected latency | No action needed |
DELAYED |
Last event is older than max latency threshold | Investigate collector logs; check upstream API status |
SILENT |
No data received within silence threshold | Immediate investigation required; check credentials, API access, collector function |
VOLUME_ANOMALY |
Volume significantly higher or lower than baseline | Review if change is expected (maintenance, new deployment, incident) |
UNKNOWN |
Insufficient baseline data | Wait for baseline computation (first 7 days) |
Web UI: Navigate to Source Health in the sidebar or visit /health. The dashboard shows:
- Summary cards with counts by status
- Sortable table of all sources with status badges
- Per-source detail view with volume charts and gap timelines
API: Query the health summary endpoint:
curl -H "Authorization: Bearer $TOKEN" "$API_ENDPOINT/health/summary"AWS CLI: Check the health check Lambda logs:
aws logs tail /aws/lambda/mantissa-log-prod-health-check \
--follow --format shortDELAYED alert:
- Check collector Lambda logs for errors
- Verify upstream API credentials are valid
- Check if the source API is experiencing an outage
- If expected (maintenance window), acknowledge the alert via the UI or API
SILENT alert (high/critical):
- Immediately check collector Lambda invocation history
- Verify API credentials in Secrets Manager / Secret Manager / Key Vault
- Test manual collector invocation
- Check CloudWatch/Cloud Monitoring/Azure Monitor for collector errors
- If the source is intentionally offline, acknowledge the alert
VOLUME_ANOMALY alert:
- Check if volume drop corresponds to a known event (weekend, holiday, maintenance)
- For spikes, investigate if a log flood is occurring (misconfigured application, attack)
- Adjust thresholds if the volume change is the new normal
Suppress further alerts for a source while investigating:
curl -X POST "$API_ENDPOINT/health/sources/okta/acknowledge" \
-H "Authorization: Bearer $TOKEN" \
-H "Content-Type: application/json" \
-d '{
"suppression_duration_seconds": 7200,
"notes": "Investigating Okta API outage - ticket INC-1234"
}'View data gaps for a specific source:
curl -H "Authorization: Bearer $TOKEN" \
"$API_ENDPOINT/health/sources/okta/history?granularity=hour"The response includes gap_windows with start/end timestamps for detected gaps.
Update thresholds for a specific source via the API:
curl -X PUT "$API_ENDPOINT/health/sources/okta/config" \
-H "Authorization: Bearer $TOKEN" \
-H "Content-Type: application/json" \
-d '{
"silence_threshold_seconds": 7200,
"volume_anomaly_stddev_threshold": 2.5
}'Or update the custom config file in S3/GCS/Blob Storage. See Log Sources Configuration for details.
Run a health check for a specific source immediately:
curl -X POST "$API_ENDPOINT/health/sources/okta/check" \
-H "Authorization: Bearer $TOKEN"| Metric | Source | Alarm Threshold |
|---|---|---|
| Health check Lambda errors | CloudWatch | > 0 errors in 5 min |
| Health check duration | CloudWatch | > 30s average |
| Sources in SILENT state | Health API summary | > 0 |
| Sources in DELAYED state | Health API summary | > 3 |
| Baseline computation errors | CloudWatch | > 0 errors |
Find all activity from suspicious IP:
SELECT
eventtime,
eventname,
useridentity.principalid,
requestparameters
FROM cloudtrail
WHERE sourceipaddress = '203.0.113.42'
AND year = '2024'
AND month = '01'
AND day = '15'
ORDER BY eventtime DESCTrace user activity:
SELECT
eventtime,
eventname,
sourceipaddress,
resources
FROM cloudtrail
WHERE useridentity.principalid = 'AIDAI23EXAMPLE'
AND year = '2024'
AND month = '01'
ORDER BY eventtime DESCFind privilege escalations:
SELECT
eventtime,
useridentity.principalid,
eventname,
requestparameters
FROM cloudtrail
WHERE eventsource = 'iam.amazonaws.com'
AND eventname IN (
'AttachUserPolicy',
'AttachRolePolicy',
'PutUserPolicy',
'PutRolePolicy'
)
AND year = '2024'
AND month = '01'
ORDER BY eventtime DESCQuery execution times:
SELECT
query_id,
query,
data_scanned_in_bytes / 1024 / 1024 / 1024 as gb_scanned,
execution_time_millis / 1000 as execution_seconds
FROM athena_query_history
WHERE submission_date >= CURRENT_DATE - INTERVAL '7' DAY
ORDER BY data_scanned_in_bytes DESC
LIMIT 20Most expensive queries:
# Check CloudWatch Logs Insights
aws logs start-query \
--log-group-name /aws/lambda/mantissa-log-detection-engine \
--start-time $(($(date -d '7 days ago' +%s) * 1000)) \
--end-time $(($(date +%s) * 1000)) \
--query-string '
fields @timestamp, rule_name, data_scanned_bytes
| filter data_scanned_bytes > 1000000000
| sort data_scanned_bytes desc
| limit 20
'Priority 1: Critical Alerts
- Root account usage
- IAM policy changes granting admin access
- S3 bucket made public
- Security group opened to internet
- GuardDuty critical findings
Actions:
- Verify alert is legitimate
- Assess scope of impact
- Begin containment
- Notify security team
- Start incident log
Priority 2: High Alerts
- Multiple failed logins
- Unusual API activity
- Large data transfers
- Configuration changes
Actions:
- Review evidence
- Check for related alerts
- Investigate user/resource
- Determine if escalation needed
# 1. Get alert details
ALERT_ID="alert-20240115-001"
aws dynamodb get-item \
--table-name $STATE_TABLE \
--key "{\"alert_id\":{\"S\":\"$ALERT_ID\"}}" \
> alert-details.json
# 2. Extract key fields
jq -r '.Item.evidence.S' alert-details.json
# 3. Query related activity
# Use fields from evidence to build query
# 4. Check for related alerts
aws dynamodb scan \
--table-name $STATE_TABLE \
--filter-expression "sourceip = :ip" \
--expression-attribute-values "{\":ip\":{\"S\":\"203.0.113.42\"}}"
# 5. Document findings
echo "Investigation: $ALERT_ID" > investigation-log.md
echo "Date: $(date)" >> investigation-log.md
echo "Analyst: $USER" >> investigation-log.mdDisable compromised user:
USER_NAME="compromised-user"
aws iam delete-access-key \
--user-name $USER_NAME \
--access-key-id AKIAIOSFODNN7EXAMPLE
aws iam attach-user-policy \
--user-name $USER_NAME \
--policy-arn arn:aws:iam::aws:policy/AWSDenyAllBlock IP address:
# Add to NACL
VPC_ID="vpc-12345678"
NACL_ID=$(aws ec2 describe-network-acls \
--filters "Name=vpc-id,Values=$VPC_ID" \
--query 'NetworkAcls[0].NetworkAclId' --output text)
aws ec2 create-network-acl-entry \
--network-acl-id $NACL_ID \
--rule-number 100 \
--protocol -1 \
--rule-action deny \
--cidr-block 203.0.113.42/32 \
--ingressRevoke active sessions:
# Force user to re-authenticate
aws cognito-idp admin-user-global-sign-out \
--user-pool-id $USER_POOL_ID \
--username $USER_NAME# 1. Check EventBridge rule
RULE_NAME=$(cat terraform-outputs.json | jq -r '.detection_schedule_rule_name.value')
aws events describe-rule --name $RULE_NAME
# 2. Check Lambda executions
DETECTION_ENGINE=$(cat terraform-outputs.json | jq -r '.detection_engine_function_name.value')
aws cloudwatch get-metric-statistics \
--namespace AWS/Lambda \
--metric-name Invocations \
--dimensions Name=FunctionName,Value=$DETECTION_ENGINE \
--start-time $(date -u -d '1 hour ago' +%Y-%m-%dT%H:%M:%S) \
--end-time $(date -u +%Y-%m-%dT%H:%M:%S) \
--period 300 \
--statistics Sum
# 3. Check for errors
aws logs tail /aws/lambda/$DETECTION_ENGINE --since 1h
# 4. Verify rules exist
RULES_BUCKET=$(cat terraform-outputs.json | jq -r '.rules_bucket.value')
aws s3 ls s3://$RULES_BUCKET/rules/ --recursive
# 5. Test rule manually
aws lambda invoke \
--function-name $DETECTION_ENGINE \
--log-type Tail \
response.json# 1. Check query execution
DATABASE=$(cat terraform-outputs.json | jq -r '.database_name.value')
WORKGROUP=$(cat terraform-outputs.json | jq -r '.athena_workgroup_name.value')
# Get recent query executions
aws athena list-query-executions \
--work-group $WORKGROUP \
--max-results 10
# Get specific query details
QUERY_ID="abc-def-123"
aws athena get-query-execution --query-execution-id $QUERY_ID
# 2. Check data scanned
# Look for queries scanning > 1GB
# 3. Verify partitions exist
aws athena start-query-execution \
--query-string "SHOW PARTITIONS cloudtrail" \
--query-execution-context Database=$DATABASE \
--work-group $WORKGROUP
# 4. Optimize rule
# Add partition filters, reduce time window# 1. Check Athena costs
aws ce get-cost-and-usage \
--time-period Start=2024-01-01,End=2024-01-31 \
--granularity DAILY \
--metrics BlendedCost \
--filter '{"Dimensions":{"Key":"SERVICE","Values":["Amazon Athena"]}}'
# 2. Find expensive queries
# Use CloudWatch Logs Insights on detection engine logs
# 3. Optimize
# - Add partition filters to all rules
# - Convert to Parquet format
# - Reduce detection frequency
# - Disable low-value rules# 1. Export detection rules
aws s3 sync s3://$RULES_BUCKET/rules/ rules-backup/
# 2. Backup DynamoDB state
aws dynamodb create-backup \
--table-name $STATE_TABLE \
--backup-name mantissa-log-state-$(date +%Y%m%d)
# 3. Export Terraform state
cd infrastructure/aws/terraform
terraform state pull > terraform-state-backup-$(date +%Y%m%d).json
# 4. Backup secrets
aws secretsmanager list-secrets \
--query 'SecretList[?starts_with(Name, `mantissa-log`)].Name' \
--output text | while read SECRET; do
aws secretsmanager get-secret-value --secret-id $SECRET > "secrets-backup/$SECRET.json"
doneComplete environment loss:
# 1. Restore from backups
git clone <repository-url>
cd mantissa-log-dev
# 2. Restore Terraform state
cd infrastructure/aws/terraform
terraform init
# Restore state from backup
# 3. Re-run deployment
bash scripts/deploy.sh
# 4. Restore rules
aws s3 sync rules-backup/ s3://$RULES_BUCKET/rules/
# 5. Restore secrets
for SECRET_FILE in secrets-backup/*.json; do
SECRET_NAME=$(basename $SECRET_FILE .json)
aws secretsmanager create-secret \
--name $SECRET_NAME \
--secret-string "$(jq -r '.SecretString' $SECRET_FILE)"
done-- Bad: No partition filters
SELECT * FROM cloudtrail
WHERE eventtime > '2024-01-15T00:00:00Z'
-- Good: With partition filters
SELECT * FROM cloudtrail
WHERE year = '2024'
AND month = '01'
AND day = '15'
AND eventtime > '2024-01-15T00:00:00Z'- Use partition filters in all queries
- Convert logs to Parquet format
- Set S3 lifecycle policies
- Tune detection frequency
- Disable low-value rules
- Use query result reuse in Athena
See Scaling Guide for detailed scaling strategies.