# HFT alert rules — discovered automatically by Prometheus Operator # Apply: kubectl apply -f prometheus-rules-hft.yaml apiVersion: monitoring.coreos.com/v1 kind: PrometheusRule metadata: name: foxhunt-hft-alerts namespace: foxhunt labels: app.kubernetes.io/part-of: foxhunt role: alert-rules spec: groups: # ═══════════════════════════════ # CRITICAL HFT ALERTS # ═══════════════════════════════ - name: hft_critical rules: - alert: TradingServiceDown expr: up{job="foxhunt-services", service="trading-service"} == 0 for: 5s labels: severity: critical component: trading_engine impact: trading_halt annotations: summary: "Trading Service is DOWN" description: "Trading Service has been down for >5s. All trading operations halted." action: "Immediate intervention required" - alert: DataAcquisitionDown expr: up{job="foxhunt-services", service="data-acquisition-service"} == 0 for: 10s labels: severity: critical component: market_data impact: blind_trading annotations: summary: "Data Acquisition Service is DOWN" description: "Market data feed down >10s. Trading without data is dangerous." action: "Stop all trading immediately" - alert: BrokerGatewayDown expr: up{job="foxhunt-services", service="broker-gateway-service"} == 0 for: 5s labels: severity: critical component: risk_management impact: uncontrolled_risk annotations: summary: "Broker Gateway (risk controls) is DOWN" description: "Broker gateway down >5s. Order routing and risk controls offline." action: "Emergency trading halt" - alert: TradingLatencyHigh expr: histogram_quantile(0.95, foxhunt_trading_latency_seconds_bucket{operation="order_placement"}) > 0.001 for: 30s labels: severity: critical component: trading_engine impact: competitive_disadvantage annotations: summary: "Trading latency P95 > 1ms" description: "95th percentile order placement latency is {{ $value }}s. HFT advantage compromised." action: "Investigate performance bottlenecks" - alert: PositionLimitBreach expr: foxhunt_position_size_usd / foxhunt_position_limit_usd > 0.95 for: 0s labels: severity: critical component: risk_management impact: regulatory_breach annotations: summary: "Position limit nearly breached ({{ $value | humanizePercentage }})" description: "Current position at {{ $value | humanizePercentage }} of limit." action: "Reduce positions immediately" - alert: LossLimitBreach expr: foxhunt_daily_pnl_usd < foxhunt_loss_limit_usd for: 0s labels: severity: critical component: risk_management impact: financial_loss annotations: summary: "Daily loss limit breached" description: "Daily P&L ${{ $value }} breaches loss limit. Trading must halt." action: "Emergency trading halt" # ═══════════════════════════════ # HIGH PRIORITY ALERTS # ═══════════════════════════════ - name: hft_high rules: - alert: DatabaseConnectionHigh expr: foxhunt_database_connections_active / foxhunt_database_connections_max > 0.85 for: 1m labels: severity: high component: database impact: performance_degradation annotations: summary: "Database connection pool at {{ $value | humanizePercentage }}" - alert: MemoryUsageHigh expr: (1 - (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes)) > 0.90 for: 2m labels: severity: high component: system impact: performance_degradation annotations: summary: "Memory usage {{ $value | humanizePercentage }} on {{ $labels.instance }}" - alert: CPUUsageHigh expr: 100 - (avg by (instance) (irate(node_cpu_seconds_total{mode="idle"}[5m])) * 100) > 90 for: 5m labels: severity: high component: system impact: performance_degradation annotations: summary: "CPU usage {{ $value | printf \"%.1f\" }}% on {{ $labels.instance }}" - alert: DiskUsageHigh expr: (1 - (node_filesystem_avail_bytes{fstype!="tmpfs"} / node_filesystem_size_bytes{fstype!="tmpfs"})) > 0.85 for: 5m labels: severity: high component: system impact: storage_failure annotations: summary: "Disk usage {{ $value | humanizePercentage }} on {{ $labels.instance }} {{ $labels.mountpoint }}" - alert: OrderRejectionRateHigh expr: rate(foxhunt_orders_rejected_total[5m]) / rate(foxhunt_orders_total[5m]) > 0.05 for: 2m labels: severity: high component: trading_engine impact: trading_inefficiency annotations: summary: "Order rejection rate {{ $value | humanizePercentage }}" - alert: MarketDataLag expr: foxhunt_market_data_lag_seconds > 0.1 for: 1m labels: severity: high component: market_data impact: stale_data annotations: summary: "Market data lag {{ $value }}s — trading on stale data" - alert: DrawdownHigh expr: foxhunt_drawdown_percent > 0.05 for: 15m labels: severity: high component: risk_management impact: financial_risk annotations: summary: "Portfolio drawdown {{ $value | humanizePercentage }}" # ═══════════════════════════════ # MEDIUM / WARNING ALERTS # ═══════════════════════════════ - name: hft_medium rules: - alert: ServiceRestarted expr: increase(process_start_time_seconds[10m]) > 0 for: 0s labels: severity: medium component: service impact: disruption annotations: summary: "Service {{ $labels.job }} restarted on {{ $labels.instance }}" - alert: NetworkLatencyHigh expr: foxhunt_network_latency_seconds > 0.01 for: 5m labels: severity: medium component: network impact: performance_impact annotations: summary: "Network latency {{ $value }}s" - alert: BackupFailed expr: time() - foxhunt_last_backup_timestamp > 86400 for: 1h labels: severity: medium component: backup impact: data_risk annotations: summary: "Last backup {{ $value | humanizeDuration }} ago" - alert: SSLCertificateExpiringSoon expr: (ssl_certificate_expiry_timestamp - time()) / 86400 < 30 for: 1h labels: severity: medium component: security impact: service_disruption annotations: summary: "SSL cert expires in {{ $value }} days for {{ $labels.instance }}" # ═══════════════════════════════ # BUSINESS LOGIC ALERTS # ═══════════════════════════════ - name: hft_business rules: - alert: TradingVolumeAnomalyHigh expr: rate(foxhunt_trades_total[5m]) > (avg_over_time(rate(foxhunt_trades_total[5m])[1h:5m]) * 3) for: 2m labels: severity: medium component: trading_engine impact: business_anomaly annotations: summary: "Trading volume 3x hourly average ({{ $value }} trades/sec)" - alert: TradingVolumeAnomalyLow expr: rate(foxhunt_trades_total[5m]) < (avg_over_time(rate(foxhunt_trades_total[5m])[1h:5m]) * 0.1) for: 10m labels: severity: medium component: trading_engine impact: business_anomaly annotations: summary: "Trading volume 10% of hourly average ({{ $value }} trades/sec)" - alert: PnLVolatilityHigh expr: stddev_over_time(foxhunt_pnl_usd[1h]) > 10000 for: 30m labels: severity: medium component: risk_management impact: financial_risk annotations: summary: "P&L std dev ${{ $value }} over 1h" # ═══════════════════════════════ # INFRASTRUCTURE ALERTS # ═══════════════════════════════ - name: hft_infrastructure rules: - alert: RedisConnectionFailed expr: redis_connected_clients{job="redis"} == 0 for: 30s labels: severity: critical component: redis impact: cache_failure annotations: summary: "Redis has 0 connected clients" - alert: PostgreSQLDown expr: pg_up{job="postgres"} == 0 for: 30s labels: severity: critical component: postgresql impact: data_unavailable annotations: summary: "PostgreSQL is not responding" # ═══════════════════════════════ # SECURITY ALERTS # ═══════════════════════════════ - name: hft_security rules: - alert: AuthenticationFailuresHigh expr: rate(foxhunt_auth_failures_total[5m]) > 5 for: 2m labels: severity: high component: security impact: security_breach annotations: summary: "{{ $value }} auth failures/sec" - alert: UnauthorizedAccessAttempts expr: rate(foxhunt_unauthorized_requests_total[5m]) > 1 for: 1m labels: severity: critical component: security impact: security_breach annotations: summary: "{{ $value }} unauthorized requests/sec from {{ $labels.source_ip }}" - alert: APIRateLimitExceeded expr: rate(foxhunt_api_rate_limit_exceeded_total[5m]) > 0.1 for: 5m labels: severity: medium component: api impact: service_degradation annotations: summary: "API rate limits exceeded {{ $value }}/sec" # ═══════════════════════════════ # MONITORING HEALTH # ═══════════════════════════════ - name: hft_monitoring rules: - alert: PrometheusTargetDown expr: up == 0 for: 2m labels: severity: medium component: monitoring impact: observability_loss annotations: summary: "{{ $labels.job }} target {{ $labels.instance }} down >2m" - alert: HighAlertRate expr: rate(prometheus_notifications_total[5m]) > 10 for: 10m labels: severity: medium component: monitoring impact: alert_fatigue annotations: summary: "{{ $value }} alerts/sec — check for alert storm"