# ML Training Service Notification Configuration # # Integrated with main AlertManager configuration for ML-specific routing # ML Training Service Alert Routes (add to alertmanager.yml) ml_training_routes: # Critical ML alerts - PagerDuty + Slack - match: severity: critical component: ml receiver: 'ml-critical-alerts' group_wait: 0s repeat_interval: 30m continue: false # High severity ML alerts - Slack + Email - match: severity: high component: ml receiver: 'ml-high-alerts' group_wait: 15s repeat_interval: 1h continue: false # Warning ML alerts - Slack only - match: severity: warning component: ml receiver: 'ml-warning-alerts' group_wait: 30s repeat_interval: 4h continue: false # Info ML alerts - Slack #ml-info channel - match: severity: info component: ml receiver: 'ml-info-alerts' group_wait: 5m repeat_interval: 24h continue: false # ML Training Service Alert Receivers ml_receivers: # Critical ML alerts - PagerDuty + Slack - name: 'ml-critical-alerts' slack_configs: - channel: '#foxhunt-ml-critical' api_url: '${SLACK_WEBHOOK_URL}' title: '🚨 ML TRAINING CRITICAL: {{ .GroupLabels.alertname }}' text: | *Alert:* {{ .GroupLabels.alertname }} *Model Type:* {{ .CommonLabels.model_type }} *Job ID:* {{ .CommonLabels.job_id }} {{ range .Alerts }} *Summary:* {{ .Annotations.summary }} *Description:* {{ .Annotations.description }} *Impact:* {{ .Annotations.impact }} *Action Required:* {{ .Annotations.action }} *Runbook:* {{ .Annotations.runbook_url }} {{ end }} send_resolved: true color: '{{ if eq .Status "firing" }}danger{{ else }}good{{ end }}' pagerduty_configs: - routing_key: '${PAGERDUTY_ML_INTEGRATION_KEY}' severity: 'critical' description: '{{ .GroupLabels.alertname }}: {{ .CommonAnnotations.summary }}' details: alert_type: '{{ .CommonLabels.alert_type }}' model_type: '{{ .CommonLabels.model_type }}' job_id: '{{ .CommonLabels.job_id }}' impact: '{{ .CommonAnnotations.impact }}' action: '{{ .CommonAnnotations.action }}' runbook_url: '{{ .CommonAnnotations.runbook_url }}' client: 'Foxhunt ML Training Service' client_url: 'http://localhost:3000/d/ml-training-monitoring' # High severity ML alerts - name: 'ml-high-alerts' slack_configs: - channel: '#foxhunt-ml-high' api_url: '${SLACK_WEBHOOK_URL}' title: '⚠️ ML TRAINING HIGH: {{ .GroupLabels.alertname }}' text: | *Alert:* {{ .GroupLabels.alertname }} {{ range .Alerts }} *Summary:* {{ .Annotations.summary }} *Description:* {{ .Annotations.description }} *Impact:* {{ .Annotations.impact }} {{ if .Annotations.action }}*Action:* {{ .Annotations.action }}{{ end }} {{ end }} send_resolved: true color: 'danger' # Warning ML alerts - name: 'ml-warning-alerts' slack_configs: - channel: '#foxhunt-ml-warnings' api_url: '${SLACK_WEBHOOK_URL}' title: '⚠️ ML TRAINING WARNING: {{ .GroupLabels.alertname }}' text: | *Alert:* {{ .GroupLabels.alertname }} {{ range .Alerts }} *Summary:* {{ .Annotations.summary }} *Description:* {{ .Annotations.description }} {{ end }} send_resolved: true color: 'warning' # Info ML alerts - name: 'ml-info-alerts' slack_configs: - channel: '#foxhunt-ml-info' api_url: '${SLACK_WEBHOOK_URL}' title: 'ℹ️ ML TRAINING INFO: {{ .GroupLabels.alertname }}' text: | *Alert:* {{ .GroupLabels.alertname }} {{ range .Alerts }} *Summary:* {{ .Annotations.summary }} *Description:* {{ .Annotations.description }} {{ end }} send_resolved: true color: 'good' # Inhibition Rules for ML Training Service ml_inhibit_rules: # If GPU memory exhausted, suppress GPU memory high warning - source_match: alertname: 'GPUMemoryExhausted' target_match: alertname: 'GPUMemoryUsageHigh' equal: ['gpu_id'] # If training job failed, suppress progress/slowdown alerts - source_match: alertname: 'TrainingJobFailed' target_match_re: alertname: 'TrainingSlowdown|ModelConvergenceStalled' equal: ['job_id'] # If automated job stuck, suppress other job-related alerts - source_match: alertname: 'AutomatedTrainingJobStuck' target_match_re: alertname: 'TrainingSlowdown|TrainingIterationTimeSlow' equal: ['job_id'] # If data drift detected, suppress model accuracy degraded - source_match: alertname: 'ModelDriftDetected' target_match: alertname: 'MLModelAccuracyDegraded' equal: ['model'] # If S3 connection errors, suppress checkpoint save failures - source_match: alertname: 'S3ConnectionErrors' target_match: alertname: 'CheckpointSaveFailures' equal: ['job_id'] # Environment Variables (set in deployment environment) # export SLACK_WEBHOOK_URL=https://hooks.slack.com/services/YOUR/SLACK/WEBHOOK # export PAGERDUTY_ML_INTEGRATION_KEY=your_pagerduty_integration_key