# Foxhunt Production Alert Rules # Agent H5: Production-Grade Alerting Configuration # Created: 2025-10-18 groups: # ============================================================================ # CRITICAL ALERTS - P99 Latency and Service Availability # ============================================================================ - name: production-latency-critical interval: 15s rules: # P99 Latency > 100ms - CRITICAL - alert: CriticalP99LatencyAPI expr: | histogram_quantile(0.99, rate(grpc_server_handling_seconds_bucket{job="api"}[1m]) ) > 0.1 for: 1m labels: severity: critical component: latency service: api annotations: summary: "CRITICAL: API P99 latency exceeds 100ms" description: | API P99 latency is {{ $value | humanizeDuration }} Target: < 100ms Current: {{ $value | humanizeDuration }} This impacts all client requests. runbook_url: "https://wiki.foxhunt.local/runbooks/high-latency" - alert: CriticalP99LatencyTradingService expr: | histogram_quantile(0.99, rate(grpc_server_handling_seconds_bucket{job="trading_service"}[1m]) ) > 0.1 for: 1m labels: severity: critical component: latency service: trading_service annotations: summary: "CRITICAL: Trading Service P99 latency exceeds 100ms" description: | Trading Service P99 latency is {{ $value | humanizeDuration }} Target: < 100ms Current: {{ $value | humanizeDuration }} This impacts order execution speed. runbook_url: "https://wiki.foxhunt.local/runbooks/high-latency" # Order Processing Latency (Direct metric if available) - alert: CriticalOrderProcessingLatency expr: | histogram_quantile(0.99, rate(foxhunt_order_processing_duration_seconds_bucket[1m]) ) > 0.1 for: 30s labels: severity: critical component: trading service: trading_service annotations: summary: "CRITICAL: Order processing P99 latency > 100ms" description: | Order processing P99 latency: {{ $value | humanizeDuration }} Target: < 100ms This directly impacts trade execution quality. runbook_url: "https://wiki.foxhunt.local/runbooks/order-latency" # Service Down - CRITICAL - alert: CriticalServiceDown expr: up{job=~"api|trading_service|trading_agent_service"} == 0 for: 30s labels: severity: critical component: availability annotations: summary: "CRITICAL: {{ $labels.job }} is DOWN" description: | Service: {{ $labels.job }} Instance: {{ $labels.instance }} Down for: > 30 seconds This is a critical production outage. runbook_url: "https://wiki.foxhunt.local/runbooks/service-down" # ============================================================================ # WARNING ALERTS - Error Rates # ============================================================================ - name: production-error-rates interval: 15s rules: # Error Rate > 1% - WARNING - alert: HighErrorRateAPI expr: | sum(rate(grpc_server_handled_total{job="api",grpc_code!="OK"}[5m])) / sum(rate(grpc_server_handled_total{job="api"}[5m])) > 0.01 for: 3m labels: severity: warning component: errors service: api annotations: summary: "WARNING: API error rate exceeds 1%" description: | Error rate: {{ $value | humanizePercentage }} Target: < 1% Total errors in last 5m: {{ with query "sum(increase(grpc_server_handled_total{job='api',grpc_code!='OK'}[5m]))" }}{{ . | first | value | humanize }}{{ end }} Review error logs immediately. runbook_url: "https://wiki.foxhunt.local/runbooks/high-error-rate" - alert: HighErrorRateTradingService expr: | sum(rate(grpc_server_handled_total{job="trading_service",grpc_code!="OK"}[5m])) / sum(rate(grpc_server_handled_total{job="trading_service"}[5m])) > 0.01 for: 3m labels: severity: warning component: errors service: trading_service annotations: summary: "WARNING: Trading Service error rate exceeds 1%" description: | Error rate: {{ $value | humanizePercentage }} Target: < 1% Total errors in last 5m: {{ with query "sum(increase(grpc_server_handled_total{job='trading_service',grpc_code!='OK'}[5m]))" }}{{ . | first | value | humanize }}{{ end }} Check order rejection reasons. runbook_url: "https://wiki.foxhunt.local/runbooks/high-error-rate" # Order Rejection Rate - alert: HighOrderRejectionRate expr: | rate(foxhunt_orders_rejected_total[5m]) / rate(foxhunt_orders_total[5m]) > 0.01 for: 3m labels: severity: warning component: trading service: trading_service annotations: summary: "WARNING: Order rejection rate exceeds 1%" description: | Rejection rate: {{ $value | humanizePercentage }} Target: < 1% Check risk limits and margin requirements. runbook_url: "https://wiki.foxhunt.local/runbooks/order-rejections" # ============================================================================ # CRITICAL ALERTS - Memory Growth and Resource Exhaustion # ============================================================================ - name: production-memory-critical interval: 30s rules: # Memory Growth > 10% per hour - CRITICAL - alert: CriticalMemoryGrowth expr: | ( process_resident_memory_bytes - (process_resident_memory_bytes offset 1h) ) / (process_resident_memory_bytes offset 1h) > 0.10 for: 5m labels: severity: critical component: memory annotations: summary: "CRITICAL: Memory growth exceeds 10% per hour" description: | Service: {{ $labels.job }} Growth rate: {{ $value | humanizePercentage }} Current memory: {{ with query (printf "process_resident_memory_bytes{job='%s',instance='%s'}" .Labels.job .Labels.instance) }}{{ . | first | value | humanize1024 }}B{{ end }} Memory 1h ago: {{ with query (printf "process_resident_memory_bytes{job='%s',instance='%s'} offset 1h" .Labels.job .Labels.instance) }}{{ . | first | value | humanize1024 }}B{{ end }} Potential memory leak detected. runbook_url: "https://wiki.foxhunt.local/runbooks/memory-leak" # Absolute Memory Threshold - alert: CriticalMemoryUsageAbsolute expr: | process_resident_memory_bytes > 8 * 1024 * 1024 * 1024 for: 2m labels: severity: critical component: memory annotations: summary: "CRITICAL: Process memory exceeds 8GB" description: | Service: {{ $labels.job }} Current memory: {{ $value | humanize1024 }}B This may lead to OOM kills. runbook_url: "https://wiki.foxhunt.local/runbooks/high-memory" # System Memory Pressure - alert: CriticalSystemMemoryPressure expr: | (1 - (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes)) > 0.90 for: 2m labels: severity: critical component: system annotations: summary: "CRITICAL: System memory usage exceeds 90%" description: | Memory usage: {{ $value | humanizePercentage }} Available: {{ with query "node_memory_MemAvailable_bytes" }}{{ . | first | value | humanize1024 }}B{{ end }} Total: {{ with query "node_memory_MemTotal_bytes" }}{{ . | first | value | humanize1024 }}B{{ end }} Risk of OOM condition. runbook_url: "https://wiki.foxhunt.local/runbooks/system-memory" # ============================================================================ # WARNING ALERTS - Resource Monitoring # ============================================================================ - name: production-resources interval: 30s rules: # CPU Usage - alert: HighCPUUsage expr: | 100 - (avg by(instance) (rate(node_cpu_seconds_total{mode="idle"}[2m])) * 100) > 80 for: 5m labels: severity: warning component: cpu annotations: summary: "WARNING: CPU usage exceeds 80%" description: | CPU usage: {{ $value | humanize }}% Instance: {{ $labels.instance }} Check for runaway processes. runbook_url: "https://wiki.foxhunt.local/runbooks/high-cpu" # Disk Space - alert: DiskSpaceLow expr: | (node_filesystem_avail_bytes{mountpoint="/"} / node_filesystem_size_bytes{mountpoint="/"}) * 100 < 15 for: 5m labels: severity: warning component: disk annotations: summary: "WARNING: Disk space below 15%" description: | Available: {{ $value | humanize }}% Instance: {{ $labels.instance }} Free space: {{ with query (printf "node_filesystem_avail_bytes{instance='%s',mountpoint='/'}" .Labels.instance) }}{{ . | first | value | humanize1024 }}B{{ end }} runbook_url: "https://wiki.foxhunt.local/runbooks/disk-space" - alert: DiskSpaceCritical expr: | (node_filesystem_avail_bytes{mountpoint="/"} / node_filesystem_size_bytes{mountpoint="/"}) * 100 < 10 for: 2m labels: severity: critical component: disk annotations: summary: "CRITICAL: Disk space below 10%" description: | Available: {{ $value | humanize }}% Instance: {{ $labels.instance }} Free space: {{ with query (printf "node_filesystem_avail_bytes{instance='%s',mountpoint='/'}" .Labels.instance) }}{{ . | first | value | humanize1024 }}B{{ end }} Immediate action required. runbook_url: "https://wiki.foxhunt.local/runbooks/disk-space" # ============================================================================ # DATABASE ALERTS # ============================================================================ - name: production-database interval: 30s rules: # PostgreSQL Down - alert: CriticalPostgreSQLDown expr: up{job="postgres_exporter"} == 0 for: 30s labels: severity: critical component: database annotations: summary: "CRITICAL: PostgreSQL is DOWN" description: | PostgreSQL has been unreachable for > 30 seconds. All database operations are failing. runbook_url: "https://wiki.foxhunt.local/runbooks/postgres-down" # Connection Pool Exhaustion - alert: PostgreSQLConnectionPoolExhaustion expr: | pg_stat_database_numbackends / 200 > 0.90 for: 2m labels: severity: critical component: database annotations: summary: "CRITICAL: PostgreSQL connection pool near exhaustion" description: | Active connections: {{ $value | humanize }} Max connections: 200 Usage: {{ $value | humanizePercentage }} runbook_url: "https://wiki.foxhunt.local/runbooks/postgres-connections" # Slow Queries - alert: SlowDatabaseQueries expr: | pg_stat_statements_mean_exec_time_seconds > 0.1 for: 3m labels: severity: warning component: database annotations: summary: "WARNING: Slow database queries detected" description: | Average query time: {{ $value | humanizeDuration }} Target: < 100ms Review query performance. runbook_url: "https://wiki.foxhunt.local/runbooks/slow-queries" # ============================================================================ # TRADING HEALTH ALERTS # ============================================================================ - name: production-trading-health interval: 15s rules: # Position Limit Breach - alert: CriticalPositionLimitBreach expr: | foxhunt_position_size_total > foxhunt_position_limit_total for: 0s labels: severity: critical component: risk annotations: summary: "CRITICAL: Position limit breached" description: | Current position: {{ with query "foxhunt_position_size_total" }}{{ . | first | value | humanize }}{{ end }} Position limit: {{ with query "foxhunt_position_limit_total" }}{{ . | first | value | humanize }}{{ end }} Immediate position reduction required. runbook_url: "https://wiki.foxhunt.local/runbooks/position-limit" # Drawdown Alert - alert: HighDrawdown expr: foxhunt_portfolio_drawdown_percent > 5 for: 0s labels: severity: critical component: risk annotations: summary: "CRITICAL: Portfolio drawdown exceeds 5%" description: | Drawdown: {{ $value | humanize }}% Target: < 5% Review trading strategy and risk parameters. runbook_url: "https://wiki.foxhunt.local/runbooks/drawdown" # Market Data Stale - alert: CriticalMarketDataStale expr: | time() - foxhunt_last_market_data_timestamp_seconds > 5 for: 0s labels: severity: critical component: market_data annotations: summary: "CRITICAL: Market data is stale" description: | Last update: {{ $value | humanizeDuration }} ago Target: < 5 seconds Trading decisions may be based on outdated information. runbook_url: "https://wiki.foxhunt.local/runbooks/stale-data" # Risk Check Failures - alert: RiskCheckFailures expr: | increase(foxhunt_risk_check_failures_total[5m]) > 5 for: 2m labels: severity: critical component: risk annotations: summary: "CRITICAL: Multiple risk check failures" description: | Failed checks in last 5m: {{ $value | humanize }} Risk management may be compromised. runbook_url: "https://wiki.foxhunt.local/runbooks/risk-failures" # ============================================================================ # ML MODEL HEALTH # ============================================================================ - name: production-ml-health interval: 30s rules: # Model Prediction Latency - alert: HighMLPredictionLatency expr: | histogram_quantile(0.99, rate(foxhunt_ml_prediction_duration_seconds_bucket[5m]) ) > 0.050 for: 5m labels: severity: warning component: ml annotations: summary: "WARNING: ML prediction P99 latency > 50ms" description: | P99 latency: {{ $value | humanizeDuration }} Target: < 50ms Model: {{ $labels.model }} runbook_url: "https://wiki.foxhunt.local/runbooks/ml-latency" # Model Prediction Errors - alert: MLPredictionErrors expr: | rate(foxhunt_ml_prediction_errors_total[5m]) > 0.01 for: 3m labels: severity: warning component: ml annotations: summary: "WARNING: ML prediction error rate > 1%" description: | Error rate: {{ $value | humanize }} errors/sec Model: {{ $labels.model }} Check model health. runbook_url: "https://wiki.foxhunt.local/runbooks/ml-errors" # ============================================================================ # AGGREGATED HEALTH CHECKS # ============================================================================ - name: production-aggregate-health interval: 1m rules: # System-wide Health Score - alert: DegradedSystemHealth expr: | ( sum(up{job=~"api|trading_service|backtesting_service|ml_training_service"}) / count(up{job=~"api|trading_service|backtesting_service|ml_training_service"}) ) < 0.75 for: 2m labels: severity: critical component: system annotations: summary: "CRITICAL: System health degraded" description: | System health: {{ $value | humanizePercentage }} Multiple services are down or unhealthy. This is a systemic issue. runbook_url: "https://wiki.foxhunt.local/runbooks/system-health" # Alert Storm Detection - alert: AlertStorm expr: | sum(ALERTS{alertstate="firing"}) > 10 for: 5m labels: severity: warning component: monitoring annotations: summary: "WARNING: Alert storm detected" description: | Active alerts: {{ $value | humanize }} Multiple alerts firing simultaneously. Investigate root cause. runbook_url: "https://wiki.foxhunt.local/runbooks/alert-storm"