From ac688cac24e4ae65fb242ef1cf2157d3bb4f0cfe Mon Sep 17 00:00:00 2001 From: jgrusewski Date: Wed, 25 Feb 2026 22:45:44 +0100 Subject: [PATCH] fix(ci): skip Rust builds on infra-only changes, remove stale files Add `rules:` with `changes:` filters to check, test, and .kaniko-base jobs so the Rust build pipeline only triggers when source code changes (crates/, bin/, services/, Cargo.*). Prevents H100 spin-up on infra-only pushes. Remove 31 stale/orphan files (-11,885 lines): - monitoring/ directory (duplicated by config/grafana + config/prometheus) - 10 broken/placeholder migration files (.broken, .skip) - 10 unreferenced config files (haproxy, nginx-lb, mutants, etc.) Add 3 historical design docs from previous restructure work. Co-Authored-By: Claude Opus 4.6 --- .gitlab-ci.yml | 53 + config/README.md | 89 -- config/foxhunt-validator.toml | 120 -- config/haproxy-lb.cfg | 365 ----- config/mcp-servers.md | 136 -- config/mutants.toml | 126 -- config/nginx-lb.conf | 331 ---- config/phase1_test_config.toml | 115 -- config/repomix.config.json | 41 - config/rust-toolchain-2024-security.toml | 39 - config/validate-production-config.sh | 180 --- .../plans/2026-02-25-legacy-cleanup-design.md | 46 + .../2026-02-25-repo-restructure-design.md | 163 ++ .../plans/2026-02-25-repo-restructure-plan.md | 640 ++++++++ migrations/002_risk_events.sql.broken | 890 ----------- migrations/003_audit_system.sql.broken | 1006 ------------ migrations/023_*.sql.skip | 376 ----- migrations/024_*.sql.skip | 70 - migrations/025_*.sql.skip | 293 ---- migrations/026_*.sql.skip | 51 - migrations/027_*.sql.skip | 41 - migrations/028_*.sql.skip | 54 - migrations/029_*.sql.skip | 43 - migrations/030_*.sql.skip | 80 - monitoring/alertmanager/alertmanager.yml | 257 ---- .../alertmanager/ml_notification_config.yml | 166 -- monitoring/docker-compose.yml | 126 -- monitoring/grafana/api_gateway_dashboard.json | 421 ----- .../grafana/ensemble_ml_production.json | 977 ------------ monitoring/grafana/ml_trading_dashboard.json | 1354 ----------------- monitoring/grafana/ml_training_dashboard.json | 613 -------- monitoring/latency_tracker.rs | 416 ----- monitoring/metrics.rs | 482 ------ monitoring/mod.rs | 32 - .../prometheus/alerts/api_gateway_alerts.yml | 162 -- .../prometheus/alerts/backtesting_alerts.yml | 311 ---- .../prometheus/alerts/ensemble_ml_alerts.yml | 601 -------- .../prometheus/alerts/ml_trading_alerts.yml | 393 ----- .../prometheus/alerts/ml_training_alerts.yml | 446 ------ .../prometheus/alerts/system_alerts.yml | 370 ----- .../alerts/trading_service_alerts.yml | 314 ---- monitoring/prometheus/prometheus.yml | 88 -- .../prometheus/trading_service_metrics.yml | 339 ----- monitoring/server.rs | 420 ----- 44 files changed, 902 insertions(+), 12734 deletions(-) delete mode 100644 config/README.md delete mode 100644 config/foxhunt-validator.toml delete mode 100644 config/haproxy-lb.cfg delete mode 100644 config/mcp-servers.md delete mode 100644 config/mutants.toml delete mode 100644 config/nginx-lb.conf delete mode 100644 config/phase1_test_config.toml delete mode 100644 config/repomix.config.json delete mode 100644 config/rust-toolchain-2024-security.toml delete mode 100755 config/validate-production-config.sh create mode 100644 docs/plans/2026-02-25-legacy-cleanup-design.md create mode 100644 docs/plans/2026-02-25-repo-restructure-design.md create mode 100644 docs/plans/2026-02-25-repo-restructure-plan.md delete mode 100644 migrations/002_risk_events.sql.broken delete mode 100644 migrations/003_audit_system.sql.broken delete mode 100644 migrations/023_*.sql.skip delete mode 100644 migrations/024_*.sql.skip delete mode 100644 migrations/025_*.sql.skip delete mode 100644 migrations/026_*.sql.skip delete mode 100644 migrations/027_*.sql.skip delete mode 100644 migrations/028_*.sql.skip delete mode 100644 migrations/029_*.sql.skip delete mode 100644 migrations/030_*.sql.skip delete mode 100644 monitoring/alertmanager/alertmanager.yml delete mode 100644 monitoring/alertmanager/ml_notification_config.yml delete mode 100644 monitoring/docker-compose.yml delete mode 100644 monitoring/grafana/api_gateway_dashboard.json delete mode 100644 monitoring/grafana/ensemble_ml_production.json delete mode 100644 monitoring/grafana/ml_trading_dashboard.json delete mode 100644 monitoring/grafana/ml_training_dashboard.json delete mode 100644 monitoring/latency_tracker.rs delete mode 100644 monitoring/metrics.rs delete mode 100644 monitoring/mod.rs delete mode 100644 monitoring/prometheus/alerts/api_gateway_alerts.yml delete mode 100644 monitoring/prometheus/alerts/backtesting_alerts.yml delete mode 100644 monitoring/prometheus/alerts/ensemble_ml_alerts.yml delete mode 100644 monitoring/prometheus/alerts/ml_trading_alerts.yml delete mode 100644 monitoring/prometheus/alerts/ml_training_alerts.yml delete mode 100644 monitoring/prometheus/alerts/system_alerts.yml delete mode 100644 monitoring/prometheus/alerts/trading_service_alerts.yml delete mode 100644 monitoring/prometheus/prometheus.yml delete mode 100644 monitoring/prometheus/trading_service_metrics.yml delete mode 100644 monitoring/server.rs diff --git a/.gitlab-ci.yml b/.gitlab-ci.yml index 3d0292f99..dc2b29a18 100644 --- a/.gitlab-ci.yml +++ b/.gitlab-ci.yml @@ -131,6 +131,27 @@ check: extends: .rust-base stage: check needs: [] + rules: + - if: $CI_PIPELINE_SOURCE == "push" + changes: + - Cargo.toml + - Cargo.lock + - crates/** + - bin/** + - services/** + - testing/** + - infra/docker/Dockerfile.service + - infra/docker/Dockerfile.web-gateway + - infra/docker/Dockerfile.training + - .gitlab-ci.yml + - if: $CI_PIPELINE_SOURCE == "merge_request_event" + changes: + - Cargo.toml + - Cargo.lock + - crates/** + - bin/** + - services/** + - testing/** script: - sccache --zero-stats || true - cargo check --workspace @@ -144,6 +165,27 @@ test: extends: .rust-base stage: test needs: [check] + rules: + - if: $CI_PIPELINE_SOURCE == "push" + changes: + - Cargo.toml + - Cargo.lock + - crates/** + - bin/** + - services/** + - testing/** + - infra/docker/Dockerfile.service + - infra/docker/Dockerfile.web-gateway + - infra/docker/Dockerfile.training + - .gitlab-ci.yml + - if: $CI_PIPELINE_SOURCE == "merge_request_event" + changes: + - Cargo.toml + - Cargo.lock + - crates/** + - bin/** + - services/** + - testing/** services: - name: redis:7-alpine alias: redis @@ -175,6 +217,17 @@ test: - docker rules: - if: $CI_COMMIT_BRANCH == "main" && $CI_PIPELINE_SOURCE == "push" + changes: + - Cargo.toml + - Cargo.lock + - crates/** + - bin/** + - services/** + - testing/** + - infra/docker/Dockerfile.service + - infra/docker/Dockerfile.web-gateway + - infra/docker/Dockerfile.training + - .gitlab-ci.yml before_script: - mkdir -p /kaniko/.docker - echo "{\"auths\":{\"rg.fr-par.scw.cloud\":{\"username\":\"nologin\",\"password\":\"${SCW_SECRET_KEY}\"}}}" > /kaniko/.docker/config.json diff --git a/config/README.md b/config/README.md deleted file mode 100644 index ba7ae6af6..000000000 --- a/config/README.md +++ /dev/null @@ -1,89 +0,0 @@ -# Config Crate - -## Overview - -The `config` crate provides a centralized, dynamic, and secure configuration management solution for Foxhunt HFT services. It enables hot-reloading of configurations and integrates with robust secret management systems, ensuring operational flexibility and security. - -## Features - -* **Centralized PostgreSQL Storage**: Stores all application configurations in a PostgreSQL database, providing a single source of truth. -* **Dynamic Hot-Reloading**: Leverages PostgreSQL's `NOTIFY/LISTEN` mechanism to push live configuration updates to running services without restarts. -* **Secure Secret Management**: Integrates with HashiCorp Vault for secure storage and retrieval of sensitive credentials and secrets. -* **Schema-Validated Configurations**: Enforces structured configuration schemas to prevent malformed or invalid configurations. -* **Model Configuration Management**: Manages configurations for various trading models, including their parameters and associated S3 asset paths. -* **Service-Specific Schemas**: Allows defining and validating distinct configuration schemas for each microservice or component. - -## Architecture - -The `config` crate's architecture comprises: - -* **Config Store**: A PostgreSQL database instance dedicated to storing configuration data. -* **Config Loader**: Component responsible for fetching configurations from PostgreSQL. -* **Vault Client**: Interface for securely interacting with HashiCorp Vault to retrieve secrets. -* **Notifier/Listener**: Utilizes PostgreSQL `NOTIFY/LISTEN` channels to signal and receive configuration changes for hot-reloading. -* **Schema Validator**: Ensures that loaded configurations adhere to predefined JSON or YAML schemas. -* **Configuration Models**: Rust structs that represent the structured configuration data, often deserialized from JSON/YAML stored in the database. - -## Usage - -To load a configuration and listen for live updates: - -```rust -use config::{ - ConfigManager, - schema::ServiceConfig, -}; -use serde::{Deserialize, Serialize}; - -#[derive(Debug, Clone, Serialize, Deserialize)] -struct MyServiceSpecificConfig { - api_key_name: String, - trade_threshold: f64, -} - -#[tokio::main] -async fn main() -> Result<(), Box> { - // Initialize ConfigManager with database connection and Vault client - let config_manager = ConfigManager::new( - "postgres://user:pass@localhost/foxhunt_config", - "http://localhost:8200", // Vault address - ).await?; - - // Load initial configuration for a specific service - let initial_config: MyServiceSpecificConfig = config_manager - .get_service_config("my_trading_service") - .await?; - println!("Initial config: {:?}", initial_config); - - // Subscribe to updates for this service's configuration - let mut config_stream = config_manager - .subscribe_to_service_config::("my_trading_service") - .await?; - - println!("Listening for config updates..."); - - tokio::spawn(async move { - while let Some(updated_config) = config_stream.recv().await { - println!("Configuration updated: {:?}", updated_config); - // Apply the new configuration to the running service - } - }); - - tokio::signal::ctrl_c().await?; - println!("Shutting down config listener."); - - Ok(()) -} -``` - -## Testing - -To run the tests for the `config` crate: - -```bash -cargo test --package config -``` - -## Documentation - -Comprehensive API documentation is available at [docs.rs/config](https://docs.rs/config). diff --git a/config/foxhunt-validator.toml b/config/foxhunt-validator.toml deleted file mode 100644 index 54c9b9413..000000000 --- a/config/foxhunt-validator.toml +++ /dev/null @@ -1,120 +0,0 @@ -# Foxhunt Validator Configuration -# This file configures the behavior of the E2E compilation validation suite - -[global] -# Default timeout in seconds for each compilation target -timeout_seconds = 300 - -# Maximum number of parallel jobs (null = auto-detect CPU cores) -max_jobs = null - -# Whether to continue validation even if some targets fail -continue_on_error = false - -# Skip Docker validation entirely -skip_docker = false - -# Enable verbose output by default -verbose = false - -[categories.libraries] -# Enable library validation -enabled = true - -# Override timeout for library compilation (null = use global default) -timeout_seconds = null - -# Additional cargo arguments for library compilation -cargo_args = ["--all-features"] - -# Environment variables for library compilation -[categories.libraries.env_vars] -# RUST_LOG = "debug" - -# Exclude specific library crates from validation -exclude = [ - # Example: exclude problematic or work-in-progress crates - # "crates/infrastructure/gpu-compute", -] - -# Include only specific crates (if specified, only these will be validated) -include_only = [] - -[categories.binaries] -enabled = true -timeout_seconds = 600 # Binaries may take longer to compile -cargo_args = [] - -[categories.binaries.env_vars] - -exclude = [] -include_only = [] - -[categories.tests] -enabled = true -timeout_seconds = 400 # Tests can be complex -cargo_args = ["--all-targets"] - -[categories.tests.env_vars] - -exclude = [] -include_only = [] - -[categories.examples] -enabled = true -timeout_seconds = 200 # Examples are usually simpler -cargo_args = [] - -[categories.examples.env_vars] - -exclude = [] -include_only = [] - -[categories.docker] -enabled = true -timeout_seconds = 1200 # Docker builds can be very slow -cargo_args = [] - -[categories.docker.env_vars] -# Docker-specific environment variables -DOCKER_BUILDKIT = "1" - -exclude = [ - # Example: exclude Docker files that require special setup - # "deploy/docker/performance-test/Dockerfile", -] -include_only = [] - -# Target-specific overrides -# Use the target name (crate name, service name, etc.) as the key - -[targets."security-service"] -# Enable or disable this specific target -enabled = true - -# Override timeout for this specific target -timeout_seconds = 450 - -# Additional cargo arguments for this target only -cargo_args = ["--features", "production"] - -# Environment variables for this target -[targets."security-service".env_vars] -FOXHUNT_SECURITY_MODE = "strict" - -[targets."trading-engine"] -enabled = true -timeout_seconds = 800 # Trading engine is complex -cargo_args = ["--release"] - -[targets."trading-engine".env_vars] -FOXHUNT_TRADING_MODE = "simulation" - -# Example of disabling a problematic target temporarily -[targets."gpu-compute"] -enabled = false # Disable until GPU infrastructure is stable - -# Example of custom command override (advanced usage) -# [targets."custom-target"] -# enabled = true -# custom_command = ["cargo", "build", "--custom-flag"] \ No newline at end of file diff --git a/config/haproxy-lb.cfg b/config/haproxy-lb.cfg deleted file mode 100644 index c01f1cbd8..000000000 --- a/config/haproxy-lb.cfg +++ /dev/null @@ -1,365 +0,0 @@ -# HAProxy Load Balancer Configuration for Foxhunt HFT Trading System -# Version: 1.0 -# Last Updated: 2025-10-07 -# -# This configuration provides: -# - Layer 7 (L7) gRPC load balancing -# - TLS termination at load balancer -# - Advanced health checks (gRPC Health Checking Protocol) -# - Rate limiting (via stick tables) -# - High availability with stats page -# - Connection limits - -# ============================================================================ -# GLOBAL SETTINGS -# ============================================================================ - -global - # Process management - daemon - maxconn 4096 # Maximum concurrent connections - - # Logging - log /dev/log local0 - log /dev/log local1 notice - - # User/group for HAProxy process - user haproxy - group haproxy - - # Security - chroot /var/lib/haproxy - pidfile /var/run/haproxy.pid - - # SSL/TLS settings - ssl-default-bind-ciphers ECDHE+AESGCM:ECDHE+CHACHA20:!aNULL:!MD5:!DSS - ssl-default-bind-options ssl-min-ver TLSv1.2 no-tls-tickets - ssl-default-server-ciphers ECDHE+AESGCM:ECDHE+CHACHA20:!aNULL:!MD5:!DSS - ssl-default-server-options ssl-min-ver TLSv1.2 no-tls-tickets - - # Performance tuning - tune.ssl.default-dh-param 2048 - tune.h2.initial-window-size 65536 - tune.h2.max-concurrent-streams 128 - -# ============================================================================ -# DEFAULTS -# ============================================================================ - -defaults - # Mode (tcp for raw TCP, http for HTTP/gRPC) - mode http - - # Logging - log global - option httplog - option dontlognull - - # Timeouts - timeout connect 10s # Time to establish connection to backend - timeout client 300s # Client inactivity timeout (5 minutes for long-running gRPC) - timeout server 300s # Server inactivity timeout - timeout http-request 10s - timeout http-keep-alive 10s - - # Health checks - timeout check 5s - - # Error handling - retries 3 - option redispatch - - # HTTP/2 support (required for gRPC) - option http-use-htx # Enable HTTP/2 - -# ============================================================================ -# STATS PAGE (Monitoring) -# ============================================================================ - -listen stats - bind *:8404 - mode http - stats enable - stats uri /stats - stats refresh 10s - stats show-legends - stats show-node - - # Authentication (change username/password in production) - stats auth admin:foxhunt123 - - # Additional stats options - stats admin if TRUE # Enable admin commands (drain, disable servers) - -# ============================================================================ -# FRONTEND (External Traffic) -# ============================================================================ - -# HTTP Frontend (redirect to HTTPS) -frontend http_frontend - bind *:80 - mode http - - # Redirect all HTTP to HTTPS - http-request redirect scheme https code 301 - -# HTTPS Frontend (TLS termination + gRPC load balancing) -frontend https_frontend - # Bind to port 443 with TLS and HTTP/2 (ALPN: h2) - bind *:443 ssl crt /etc/haproxy/certs/foxhunt.pem alpn h2,http/1.1 - - mode http - option httplog - - # Connection limits (DDoS protection) - maxconn 2000 - - # ======================================================================== - # ACCESS CONTROL LISTS (ACLs) - # ======================================================================== - - # Detect gRPC protocol (Content-Type: application/grpc) - acl is_grpc hdr(content-type) -m beg application/grpc - - # Route based on gRPC service path - acl is_api_gateway path_beg /foxhunt.api_gateway - acl is_trading_service path_beg /foxhunt.trading_service - acl is_backtesting_service path_beg /foxhunt.backtesting_service - acl is_ml_training_service path_beg /foxhunt.ml_training_service - - # Internal IP ranges (for backend services) - acl is_internal_ip src 10.0.0.0/8 172.16.0.0/12 192.168.0.0/16 - - # Health check endpoint - acl is_health_check path /health - - # ======================================================================== - # RATE LIMITING (Stick Tables) - # ======================================================================== - - # Track request rate per client IP - stick-table type ip size 100k expire 30s store http_req_rate(10s) - - # Track connection rate per client IP - http-request track-sc0 src - - # Rate limit: 1000 requests per 10 seconds per IP - http-request deny deny_status 429 if { sc_http_req_rate(0) gt 1000 } - - # Connection limit: 100 concurrent connections per IP - http-request deny deny_status 429 if { src_conn_cur ge 100 } - - # ======================================================================== - # REQUEST HEADERS - # ======================================================================== - - # Add X-Forwarded-* headers (for backend logging) - http-request set-header X-Real-IP %[src] - http-request set-header X-Forwarded-For %[src] - http-request set-header X-Forwarded-Proto https - - # ======================================================================== - # ROUTING - # ======================================================================== - - # Health check (return 200 OK) - use_backend health_backend if is_health_check - - # Route gRPC traffic to appropriate backend - use_backend api_gateway_backend if is_grpc is_api_gateway - use_backend trading_service_backend if is_grpc is_trading_service is_internal_ip - use_backend backtesting_service_backend if is_grpc is_backtesting_service is_internal_ip - use_backend ml_training_service_backend if is_grpc is_ml_training_service is_internal_ip - - # Default: Return 404 for unknown routes - default_backend not_found_backend - -# ============================================================================ -# BACKENDS (Service Clusters) -# ============================================================================ - -# API Gateway Backend (primary entry point) -backend api_gateway_backend - mode http - balance leastconn # Route to backend with fewest connections (best for gRPC) - - # HTTP/2 support - option http-use-htx - - # Health check using gRPC Health Checking Protocol - option httpchk - http-check send meth GET uri /grpc.health.v1.Health/Check ver HTTP/2 - http-check expect status 200 - - # Backend servers - # Format: server : [options] - server api-gateway-1 api-gateway-1:50051 check inter 5s rise 2 fall 3 maxconn 1000 - server api-gateway-2 api-gateway-2:50051 check inter 5s rise 2 fall 3 maxconn 1000 - server api-gateway-3 api-gateway-3:50051 check inter 5s rise 2 fall 3 maxconn 1000 - - # Connection reuse (important for gRPC performance) - http-reuse always - -# Trading Service Backend -backend trading_service_backend - mode http - balance leastconn - - option http-use-htx - - # Health check - option httpchk - http-check send meth GET uri /grpc.health.v1.Health/Check ver HTTP/2 - http-check expect status 200 - - # Backend servers - server trading-service-1 trading-service-1:50052 check inter 5s rise 2 fall 3 maxconn 1000 - server trading-service-2 trading-service-2:50052 check inter 5s rise 2 fall 3 maxconn 1000 - server trading-service-3 trading-service-3:50052 check inter 5s rise 2 fall 3 maxconn 1000 - - http-reuse always - -# Backtesting Service Backend (worker pool pattern) -backend backtesting_service_backend - mode http - balance leastconn - - option http-use-htx - - # Health check - option httpchk - http-check send meth GET uri /grpc.health.v1.Health/Check ver HTTP/2 - http-check expect status 200 - - # Backend servers (fewer instances, CPU-intensive workload) - server backtesting-service-1 backtesting-service-1:50053 check inter 5s rise 2 fall 3 maxconn 500 - server backtesting-service-2 backtesting-service-2:50053 check inter 5s rise 2 fall 3 maxconn 500 - - http-reuse always - -# ML Training Service Backend (GPU instances) -backend ml_training_service_backend - mode http - balance leastconn - - option http-use-htx - - # Health check - option httpchk - http-check send meth GET uri /grpc.health.v1.Health/Check ver HTTP/2 - http-check expect status 200 - - # Backend servers (GPU instances, fewer replicas) - server ml-training-service-1 ml-training-service-1:50054 check inter 5s rise 2 fall 3 maxconn 200 - server ml-training-service-2 ml-training-service-2:50054 check inter 5s rise 2 fall 3 maxconn 200 - - http-reuse always - -# Health Check Backend (returns 200 OK) -backend health_backend - mode http - http-request return status 200 content-type "text/plain" string "healthy\n" - -# Not Found Backend (returns 404) -backend not_found_backend - mode http - http-request return status 404 content-type "text/plain" string "not found\n" - -# ============================================================================ -# MONITORING & OBSERVABILITY -# ============================================================================ - -# HAProxy Stats (available at http://localhost:8404/stats) -# Metrics include: -# - Request rate, error rate, latency -# - Active connections per backend -# - Health check status -# - Session rate - -# Prometheus Exporter: -# Use haproxy_exporter: https://github.com/prometheus/haproxy_exporter -# docker run -d -p 9101:9101 prom/haproxy-exporter:latest \ -# --haproxy.scrape-uri="http://admin:foxhunt123@localhost:8404/stats;csv" - -# Example Prometheus scrape config: -# scrape_configs: -# - job_name: 'haproxy' -# static_configs: -# - targets: ['haproxy:9101'] - -# ============================================================================ -# NOTES -# ============================================================================ - -# 1. SSL Certificate: -# - Combine certificate and key into single PEM file: -# cat foxhunt.crt foxhunt.key > /etc/haproxy/certs/foxhunt.pem -# - Ensure correct permissions: chmod 600 /etc/haproxy/certs/foxhunt.pem -# -# 2. Health Checks: -# - Uses gRPC Health Checking Protocol (standard) -# - Check interval: 5 seconds -# - Rise: 2 successful checks = healthy -# - Fall: 3 failed checks = unhealthy -# - Backends must implement grpc.health.v1.Health service -# -# 3. Rate Limiting: -# - Stick tables track request rate per IP -# - Adjust limits based on traffic patterns -# - Monitor 429 error rate to avoid over-throttling -# -# 4. Load Balancing Algorithm: -# - leastconn: Best for gRPC (long-lived connections) -# - Alternative: roundrobin (simpler, less accurate) -# - Alternative: source (session affinity by IP) -# -# 5. Connection Reuse: -# - http-reuse always: Reuse connections to backends (critical for gRPC) -# - Reduces latency and connection overhead -# -# 6. Logging: -# - Logs sent to syslog (/dev/log) -# - Configure rsyslog to route HAProxy logs to file -# - Example: /var/log/haproxy.log - -# ============================================================================ -# PRODUCTION CHECKLIST -# ============================================================================ - -# [ ] Replace SSL certificate path (/etc/haproxy/certs/foxhunt.pem) -# [ ] Change stats page credentials (admin:foxhunt123) -# [ ] Configure DNS for foxhunt.trading -# [ ] Set up log rotation (logrotate) -# [ ] Enable Prometheus exporter (haproxy_exporter) -# [ ] Configure alerts (high error rate, backend down) -# [ ] Test rate limiting with load testing tools -# [ ] Configure firewall rules (allow 80, 443, 8404; deny others) -# [ ] Document backend server IP addresses -# [ ] Test failover (kill backend, verify traffic reroutes) -# [ ] Set up automated certificate renewal (certbot cron) - -# ============================================================================ -# HAPROXY MANAGEMENT COMMANDS -# ============================================================================ - -# Start HAProxy: -# haproxy -f /etc/haproxy/haproxy.cfg -# -# Check configuration: -# haproxy -c -f /etc/haproxy/haproxy.cfg -# -# Reload configuration (zero-downtime): -# haproxy -f /etc/haproxy/haproxy.cfg -sf $(cat /var/run/haproxy.pid) -# -# View stats: -# http://localhost:8404/stats -# -# Admin commands (via stats page): -# - Drain server: Set server to "drain" mode (no new connections) -# - Disable server: Take server out of rotation -# - Enable server: Put server back into rotation -# -# Socket commands (requires stats socket): -# echo "show stat" | socat stdio /var/run/haproxy.sock -# echo "show servers state" | socat stdio /var/run/haproxy.sock -# echo "disable server api_gateway_backend/api-gateway-1" | socat stdio /var/run/haproxy.sock diff --git a/config/mcp-servers.md b/config/mcp-servers.md deleted file mode 100644 index 7b2bfd65c..000000000 --- a/config/mcp-servers.md +++ /dev/null @@ -1,136 +0,0 @@ -# Recommended MCP Servers for Foxhunt - -## Currently Installed - -### Essential (Already Available) -| Server | Purpose | Key Tools | -|--------|---------|-----------| -| `corrode-mcp` | Rust development | `check_code`, `read_file`, `patch_file`, `lookup_crate_docs` | -| `zen` | Analysis & debugging | `codereview`, `debug`, `analyze`, `thinkdeep` | -| `context7` | Library documentation | `get-library-docs`, `resolve-library-id` | -| `mcp-omnisearch` | Web research | `tavily_search`, `perplexity_search`, `jina_reader` | -| `claude-flow` | Swarm coordination | `swarm_init`, `memory_usage`, `task_orchestrate` | -| `ruv-swarm` | Agent management | `agent_spawn`, `neural_train`, `daa_*` | -| `flow-nexus` | Cloud features | `sandbox_*`, `neural_*`, `workflow_*` | -| `codebase-mcp` | Codebase overview | `getCodebase`, `saveCodebase` | - -## Recommended Additions - -### High Priority - -#### 1. rust-mcp (19 tools) -**Purpose**: Enhanced Rust tooling beyond corrode-mcp -```bash -# Installation -npm install -g rust-mcp -# or -cargo install rust-mcp -``` -**Key features**: -- Cargo workspace analysis -- Dependency tree visualization -- Compile error explanation -- Rustdoc integration - -#### 2. cratedocs-mcp -**Purpose**: Direct crate documentation access -```bash -npm install -g cratedocs-mcp -``` -**Why**: Faster than context7 for Rust-specific docs, understands Candle/Tokio better - -#### 3. postgres-mcp-pro -**Purpose**: PostgreSQL optimization for trading data -```bash -npm install -g postgres-mcp-pro -``` -**Key features**: -- Query optimization suggestions -- Index recommendations -- EXPLAIN ANALYZE integration -- Slow query detection - -### Medium Priority - -#### 4. databento-mcp (Trading Data) -**Purpose**: ES futures market data integration -**Status**: Check availability at https://databento.com/docs -**Key features**: -- DBN file parsing assistance -- Market data schema validation -- CME Globex symbol resolution - -### Low Priority (Nice to Have) - -#### 5. redis-mcp -**Purpose**: Redis cache optimization -```bash -npm install -g redis-mcp -``` - -#### 6. docker-mcp -**Purpose**: Container management for services -```bash -npm install -g docker-mcp -``` - -## Configuration - -### Claude Desktop / Claude Code -Add to `~/.claude/mcp_servers.json`: -```json -{ - "servers": { - "rust-mcp": { - "command": "rust-mcp", - "args": ["start"] - }, - "cratedocs-mcp": { - "command": "cratedocs-mcp", - "args": ["--port", "3001"] - }, - "postgres-mcp-pro": { - "command": "postgres-mcp-pro", - "args": ["--connection", "postgresql://localhost/foxhunt"] - } - } -} -``` - -### Via CLI -```bash -claude mcp add rust-mcp rust-mcp start -claude mcp add cratedocs-mcp cratedocs-mcp start -claude mcp add postgres-mcp postgres-mcp-pro start -``` - -## Tool Overlap Analysis - -### Avoid Duplication -| Capability | Primary Tool | Avoid Using | -|------------|--------------|-------------| -| File reading | `corrode-mcp__read_file` | `codebase-mcp` for single files | -| Code check | `corrode-mcp__check_code` | Manual `cargo check` | -| Web search | `omnisearch__tavily_search` | Multiple search tools | -| Crate docs | `cratedocs-mcp` (if installed) | `context7` for Rust | -| Memory | `claude-flow__memory_usage` | `ruv-swarm` memory | - -### Complementary Usage -- Use `corrode-mcp` for code changes, `zen` for analysis -- Use `claude-flow` for coordination, `ruv-swarm` for neural features -- Use `omnisearch` for external research, `context7` for library docs - -## Memory Namespaces - -Store findings using claude-flow memory: -``` -project/ - Codebase structure, architecture decisions -knowledge/ - Best practices, patterns, documentation -cache/ - Recent searches, temporary data -swarm/ - Agent coordination state -agent/ - Individual agent context -``` - ---- - -*Last updated: 2025-11-28* diff --git a/config/mutants.toml b/config/mutants.toml deleted file mode 100644 index 7066c3767..000000000 --- a/config/mutants.toml +++ /dev/null @@ -1,126 +0,0 @@ -# Cargo Mutants Configuration for Foxhunt HFT System -# This file configures mutation testing for critical system components - -# Test timeout (seconds) - abort tests that run longer than this -timeout = 60 - -# Minimum test score (percentage of tests that must pass with mutants) -minimum_test_score = 80 - -# Packages to test (focus on critical components) -# We prioritize ML, trading engine, and risk management -exclude_packages = [ - "fxt", # Pure client - less critical - "data_acquisition_service", # Lower priority - "monitoring_service", # Lower priority -] - -# Files to exclude from mutation testing -exclude_files = [ - # Generated code - "*/proto/*.rs", - "*/generated/*.rs", - - # Test files - "**/tests/**", - "**/*test*.rs", - "**/*_tests.rs", - - # Benches - "**/benches/**", - "**/*bench*.rs", - - # Examples - "**/examples/**", - - # Build scripts - "**/build.rs", -] - -# Exclude specific mutants by pattern -exclude_mutants = [ - # Don't mutate logging statements - "log::", - "tracing::", - "info!", - "warn!", - "error!", - "debug!", - - # Don't mutate error messages - "anyhow::bail!", - "panic!", - - # Don't mutate test assertions - "assert_eq!", - "assert_ne!", - "assert!", -] - -# Critical modules that MUST have high mutation score -[[critical_modules]] -path = "ml/src/ensemble" -minimum_score = 90 - -[[critical_modules]] -path = "trading_engine/src/engine" -minimum_score = 90 - -[[critical_modules]] -path = "risk/src/var" -minimum_score = 85 - -[[critical_modules]] -path = "ml/src/data_loaders" -minimum_score = 85 - -# Test execution options -[test_options] -# Run tests with optimizations -release = false - -# Run tests in parallel -jobs = 4 - -# Fail fast on first error -fail_fast = false - -# Show output from tests -nocapture = true - -# Mutation strategies -[strategies] -# Replace arithmetic operators (+, -, *, /) -arithmetic = true - -# Replace comparison operators (<, >, <=, >=, ==, !=) -comparison = true - -# Replace logical operators (&&, ||, !) -logical = true - -# Replace return values -return_values = true - -# Negate conditions -negate_conditions = true - -# Remove function calls -remove_calls = false # Too aggressive for production code - -# Replace constants -constants = true - -# Output configuration -[output] -# Output format: text, json, or both -format = "json" - -# Output directory -directory = "mutation_results" - -# Generate HTML report -html = true - -# Report verbosity -verbose = true diff --git a/config/nginx-lb.conf b/config/nginx-lb.conf deleted file mode 100644 index 957bafea9..000000000 --- a/config/nginx-lb.conf +++ /dev/null @@ -1,331 +0,0 @@ -# nginx Load Balancer Configuration for Foxhunt HFT Trading System -# Version: 1.0 -# Last Updated: 2025-10-07 -# -# This configuration provides: -# - Layer 7 (L7) gRPC load balancing -# - TLS termination at load balancer -# - Health checks using gRPC Health Checking Protocol -# - Rate limiting by user tier -# - DDoS protection -# - Connection limits - -# ============================================================================ -# UPSTREAM DEFINITIONS -# ============================================================================ - -# API Gateway cluster (primary entry point for all clients) -upstream api_gateway_backend { - # Load balancing algorithm - # - least_conn: Route to backend with fewest active connections (recommended for long-lived gRPC connections) - # - round_robin: Default, simple rotation - # - ip_hash: Session affinity based on client IP (use for stateful operations) - least_conn; - - # Backend servers - # Format: server : [parameters]; - server api-gateway-1:50051 max_fails=3 fail_timeout=30s; - server api-gateway-2:50051 max_fails=3 fail_timeout=30s; - server api-gateway-3:50051 max_fails=3 fail_timeout=30s; - - # Health check (requires nginx Plus or nginx-module-health-check) - # For open-source nginx, use passive health checks (max_fails) - # For nginx Plus: - # health_check interval=5s fails=2 passes=1 uri=/grpc.health.v1.Health/Check; - - # Keepalive connections to backend - # Reuse connections instead of creating new ones (critical for gRPC performance) - keepalive 100; - keepalive_timeout 60s; -} - -# Trading Service cluster (internal, routed via API Gateway) -upstream trading_service_backend { - least_conn; - server trading-service-1:50052 max_fails=3 fail_timeout=30s; - server trading-service-2:50052 max_fails=3 fail_timeout=30s; - server trading-service-3:50052 max_fails=3 fail_timeout=30s; - keepalive 100; -} - -# Backtesting Service cluster (worker pool pattern) -upstream backtesting_service_backend { - least_conn; - server backtesting-service-1:50053 max_fails=3 fail_timeout=30s; - server backtesting-service-2:50053 max_fails=3 fail_timeout=30s; - keepalive 50; -} - -# ML Training Service cluster (GPU instances, fewer replicas) -upstream ml_training_service_backend { - least_conn; - server ml-training-service-1:50054 max_fails=3 fail_timeout=30s; - server ml-training-service-2:50054 max_fails=3 fail_timeout=30s; - keepalive 20; -} - -# ============================================================================ -# RATE LIMITING ZONES -# ============================================================================ - -# Rate limiting by user tier (extracted from JWT) -# Zone size: 10m = 10 MB = ~160,000 IP addresses - -# Free tier: 100 requests/second -limit_req_zone $jwt_user_tier zone=tier_free:10m rate=100r/s; - -# Premium tier: 500 requests/second -limit_req_zone $jwt_user_tier zone=tier_premium:10m rate=500r/s; - -# Enterprise tier: 2000 requests/second -limit_req_zone $jwt_user_tier zone=tier_enterprise:10m rate=2000r/s; - -# Internal services: No rate limit -# (Identified by internal JWT or mTLS certificate) - -# Connection limits per IP (DDoS protection) -limit_conn_zone $binary_remote_addr zone=addr:10m; - -# ============================================================================ -# MAP DEFINITIONS -# ============================================================================ - -# Extract user tier from JWT (requires lua-nginx-module or custom logic) -# For simplicity, this example assumes tier is in a custom header -# In production, extract from JWT claims -map $http_x_user_tier $rate_limit_zone { - default tier_free; - "free" tier_free; - "premium" tier_premium; - "enterprise" tier_enterprise; - "internal" ""; # No rate limit for internal services -} - -# ============================================================================ -# SERVER BLOCKS -# ============================================================================ - -# HTTP server (redirect to HTTPS) -server { - listen 80; - listen [::]:80; - server_name foxhunt.trading; - - # Redirect all HTTP traffic to HTTPS - return 301 https://$server_name$request_uri; -} - -# HTTPS server (TLS termination + gRPC load balancing) -server { - # Listen on port 443 with HTTP/2 (required for gRPC) - listen 443 ssl http2; - listen [::]:443 ssl http2; - - server_name foxhunt.trading; - - # ======================================================================== - # TLS CONFIGURATION - # ======================================================================== - - # SSL certificate (replace with your actual certificate paths) - ssl_certificate /etc/ssl/certs/foxhunt.crt; - ssl_certificate_key /etc/ssl/private/foxhunt.key; - - # TLS protocols (TLS 1.2 and 1.3 only, no SSLv3/TLS 1.0/1.1) - ssl_protocols TLSv1.2 TLSv1.3; - - # Cipher suites (prefer modern, secure ciphers) - ssl_ciphers HIGH:!aNULL:!MD5:!3DES; - ssl_prefer_server_ciphers on; - - # SSL session cache (improves performance by reusing TLS sessions) - ssl_session_cache shared:SSL:10m; - ssl_session_timeout 10m; - - # OCSP stapling (improves TLS handshake performance) - ssl_stapling on; - ssl_stapling_verify on; - ssl_trusted_certificate /etc/ssl/certs/ca-bundle.crt; - - # HSTS (HTTP Strict Transport Security) - add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always; - - # ======================================================================== - # GRPC CONFIGURATION - # ======================================================================== - - # gRPC-specific settings - grpc_read_timeout 300s; # 5 minutes (for long-running backtests) - grpc_send_timeout 30s; - grpc_connect_timeout 10s; - - # Buffer sizes (tune based on payload size) - client_body_buffer_size 1M; - client_max_body_size 10M; - - # HTTP/2 settings - http2_max_concurrent_streams 128; - http2_recv_timeout 300s; - - # ======================================================================== - # RATE LIMITING & DDOS PROTECTION - # ======================================================================== - - # Connection limit per IP (max 10 concurrent connections) - limit_conn addr 10; - - # Rate limiting by user tier (applied per location) - # See location blocks below - - # ======================================================================== - # LOGGING - # ======================================================================== - - # Access log (includes gRPC status) - access_log /var/log/nginx/access.log combined; - - # Error log - error_log /var/log/nginx/error.log warn; - - # ======================================================================== - # LOCATIONS (ROUTING) - # ======================================================================== - - # API Gateway (primary entry point for clients) - location /foxhunt.api_gateway { - # Apply rate limiting based on user tier - limit_req zone=$rate_limit_zone burst=20 nodelay; - - # gRPC proxy to backend - grpc_pass grpc://api_gateway_backend; - - # Pass original client IP (for logging and rate limiting) - grpc_set_header X-Real-IP $remote_addr; - grpc_set_header X-Forwarded-For $proxy_add_x_forwarded_for; - grpc_set_header X-Forwarded-Proto $scheme; - - # Error handling - grpc_next_upstream error timeout invalid_header http_500 http_502 http_503 http_504; - grpc_next_upstream_tries 3; - grpc_next_upstream_timeout 10s; - } - - # Trading Service (internal, typically accessed via API Gateway) - # Exposed for direct access if needed (e.g., for monitoring) - location /foxhunt.trading_service { - # Restrict to internal IPs only - allow 10.0.0.0/8; # Internal network - allow 172.16.0.0/12; # Docker network - deny all; - - grpc_pass grpc://trading_service_backend; - grpc_set_header X-Real-IP $remote_addr; - grpc_set_header X-Forwarded-For $proxy_add_x_forwarded_for; - } - - # Backtesting Service (internal) - location /foxhunt.backtesting_service { - allow 10.0.0.0/8; - allow 172.16.0.0/12; - deny all; - - grpc_pass grpc://backtesting_service_backend; - grpc_set_header X-Real-IP $remote_addr; - grpc_set_header X-Forwarded-For $proxy_add_x_forwarded_for; - } - - # ML Training Service (internal) - location /foxhunt.ml_training_service { - allow 10.0.0.0/8; - allow 172.16.0.0/12; - deny all; - - grpc_pass grpc://ml_training_service_backend; - grpc_set_header X-Real-IP $remote_addr; - grpc_set_header X-Forwarded-For $proxy_add_x_forwarded_for; - } - - # Health check endpoint (HTTP, not gRPC) - location /health { - access_log off; - return 200 "healthy\n"; - add_header Content-Type text/plain; - } - - # Metrics endpoint (for Prometheus scraping) - location /metrics { - # Restrict to monitoring IPs only - allow 10.0.0.0/8; - deny all; - - # Stub status (requires --with-http_stub_status_module) - stub_status; - } - - # Deny all other requests - location / { - return 404; - } -} - -# ============================================================================ -# MONITORING & OBSERVABILITY -# ============================================================================ - -# Prometheus metrics exporter (requires nginx-prometheus-exporter) -# Run separately as a sidecar container: -# docker run -p 9113:9113 nginx/nginx-prometheus-exporter:latest \ -# --nginx.scrape-uri=http://localhost:8080/metrics - -# Example Prometheus scrape config: -# scrape_configs: -# - job_name: 'nginx-lb' -# static_configs: -# - targets: ['nginx-lb:9113'] - -# ============================================================================ -# NOTES -# ============================================================================ - -# 1. Replace SSL certificate paths with your actual certificates: -# - Production: Use Let's Encrypt (certbot) or corporate CA -# - Development: Self-signed certificates (not recommended for production) -# -# 2. Adjust rate limits based on your traffic patterns: -# - Monitor 429 (Too Many Requests) error rate in Prometheus -# - Increase limits if legitimate users are throttled -# -# 3. Health checks: -# - Open-source nginx: Uses passive health checks (max_fails) -# - nginx Plus: Active health checks with /grpc.health.v1.Health/Check -# - Alternative: Use external health checker (e.g., Kubernetes liveness probes) -# -# 4. Load balancing algorithms: -# - least_conn: Best for gRPC (long-lived connections) -# - round_robin: Simpler, but can lead to uneven load -# - ip_hash: Use only if session affinity is required -# -# 5. Monitoring: -# - Use nginx-prometheus-exporter for metrics -# - Monitor: request rate, error rate, latency, active connections -# - Alerts: high error rate (> 1%), high latency (P99 > 100ms) -# -# 6. Scaling: -# - Add more backend servers to upstream blocks -# - Reload nginx config: nginx -s reload (zero-downtime) -# - For Kubernetes: Use Ingress controller (nginx-ingress) - -# ============================================================================ -# PRODUCTION CHECKLIST -# ============================================================================ - -# [ ] Replace SSL certificates with production certificates -# [ ] Configure DNS for foxhunt.trading -# [ ] Set up log rotation (logrotate) -# [ ] Enable Prometheus metrics exporter -# [ ] Configure alerts (high error rate, high latency) -# [ ] Test rate limiting with load testing tools -# [ ] Configure firewall rules (allow 80, 443; deny others) -# [ ] Enable nginx status page (for debugging) -# [ ] Document backend server IP addresses -# [ ] Set up automated certificate renewal (certbot cron) diff --git a/config/phase1_test_config.toml b/config/phase1_test_config.toml deleted file mode 100644 index 97d584495..000000000 --- a/config/phase1_test_config.toml +++ /dev/null @@ -1,115 +0,0 @@ -# Phase 1: Single Neuron Test Configuration -# Polygon API -> Event Bus -> DQN Model -> Log Output - -[strategy] -# Use AI Orchestration Strategy with DQN-only mode -strategy_type = "ai_orchestration" -strategy_id = "phase1_dqn_test" - -[backtesting] -# Test configuration -start_time = "2024-01-02T09:30:00Z" -end_time = "2024-01-02T16:00:00Z" -initial_capital = 100000.0 -commission_bps = 1.0 -slippage_bps = 0.5 -tick_size = 0.01 - -# Single symbol for testing -symbols = ["AAPL"] - -# Reduced latency for testing -strategy_to_exchange_latency_us = 100 -exchange_to_strategy_latency_us = 100 - -# Disable complex features for Phase 1 -enable_market_impact = false -enable_queue_position = false -enable_latency_modeling = false - -[data_source] -# PHASE 1: Deterministic testing with canned data -mode = "FromFile" -path = "tests/fixtures/canned_aapl_data.jsonl" - -[ai_orchestration] -# PHASE 1: DQN-ONLY MODE -enabled_models.dqn_enabled = true -enabled_models.tggn_enabled = false -enabled_models.tft_enabled = false -enabled_models.mamba_enabled = false -enabled_models.liquid_enabled = false - -# DQN Configuration -[ai_orchestration.dqn_config] -state_size = 20 -learning_rate = 0.001 -batch_size = 32 -memory_size = 10000 -epsilon = 0.1 -epsilon_decay = 0.995 -epsilon_min = 0.01 - -# Model weights (DQN = 1.0, others = 0.0) -[ai_orchestration.model_weights] -dqn_weight = 1.0 -tggn_weight = 0.0 -tft_weight = 0.0 -mamba_weight = 0.0 -liquid_weight = 0.0 - -# Risk limits -[ai_orchestration.risk_limits] -max_position_pct = 0.05 -max_daily_loss = 0.02 -max_trades_per_hour = 10 -stop_loss_pct = 0.01 -take_profit_pct = 0.02 - -# Performance requirements -max_inference_latency_us = 1000 # 1ms max for Phase 1 - - - -[logging] -# Enhanced logging for Phase 1 debugging -level = "debug" -filter = "backtesting=debug,ai_orchestration=trace" - -# Log specific events for Phase 1 validation -log_market_data = true -log_ai_predictions = true -log_signal_generation = true -log_order_events = true - -[validation] -# Phase 1 success criteria - ROBUST validation decoupled from model predictions -expected_log_messages = [ - "PIPELINE_SUCCESS: data_ingestion_complete", - "PIPELINE_SUCCESS: feature_extraction_complete", - "PIPELINE_SUCCESS: DQN_inference_complete", - "PIPELINE_SUCCESS: signal_processing_complete" -] - -# Performance thresholds -max_event_processing_time_us = 1000 -min_market_data_events = 20 # Reduced for canned data -expected_pipeline_completions = 10 - -# Deterministic test expectations -expected_canned_events = 20 # Number of events in canned data file -timeout_seconds = 10 # Reduced timeout for file-based testing - -[database] -# Use lightweight SQLite for Phase 1 testing -database_url = "sqlite:///tmp/phase1_test.db" -auto_migrate = true -log_queries = true - -[output] -# Save Phase 1 results for analysis -save_results = true -results_file = "/tmp/phase1_test_results.json" -save_performance_metrics = true -save_ai_predictions = true -save_market_data_sample = true \ No newline at end of file diff --git a/config/repomix.config.json b/config/repomix.config.json deleted file mode 100644 index 326f87960..000000000 --- a/config/repomix.config.json +++ /dev/null @@ -1,41 +0,0 @@ -{ - "$schema": "https://repomix.com/schemas/latest/schema.json", - "input": { - "maxFileSize": 52428800 - }, - "output": { - "filePath": "repomix-output.md", - "style": "markdown", - "parsableStyle": false, - "fileSummary": true, - "directoryStructure": true, - "files": true, - "removeComments": false, - "removeEmptyLines": false, - "compress": false, - "topFilesLength": 5, - "showLineNumbers": false, - "truncateBase64": false, - "copyToClipboard": false, - "tokenCountTree": false, - "git": { - "sortByChanges": true, - "sortByChangesMaxCommits": 100, - "includeDiffs": false, - "includeLogs": false, - "includeLogsCount": 50 - } - }, - "include": [], - "ignore": { - "useGitignore": true, - "useDefaultPatterns": true, - "customPatterns": [] - }, - "security": { - "enableSecurityCheck": true - }, - "tokenCount": { - "encoding": "o200k_base" - } -} \ No newline at end of file diff --git a/config/rust-toolchain-2024-security.toml b/config/rust-toolchain-2024-security.toml deleted file mode 100644 index f2a6885c2..000000000 --- a/config/rust-toolchain-2024-security.toml +++ /dev/null @@ -1,39 +0,0 @@ -# Rust 2024 Security-Hardened Toolchain Configuration for Foxhunt HFT System -# -# This configuration enables comprehensive Rust 2024 security features -# optimized for high-frequency trading financial systems. - -[toolchain] -channel = "1.78.0" # Latest stable with Rust 2024 features -components = ["rustfmt", "clippy", "miri", "rust-src", "llvm-tools-preview"] -targets = ["x86_64-unknown-linux-gnu"] -profile = "default" - -# Rust 2024 Edition Security Features -[profile.dev] -# Enable debug assertions for development security validation -debug-assertions = true -# Overflow checks catch arithmetic vulnerabilities -overflow-checks = true -# LTO for better security analysis -lto = "thin" - -[profile.release] -# Production security configuration -debug = 1 # Keep symbols for security monitoring -debug-assertions = false # Disabled for performance in release -overflow-checks = true # Keep overflow checks in financial systems -lto = "fat" # Full LTO for maximum security optimization -codegen-units = 1 # Single unit prevents TOCTOU between units -panic = "abort" # Security: Prevent unwinding exploitation - -[profile.release-with-debug] -# Security-hardened profile with debugging capabilities -inherits = "release" -debug = 2 -strip = "none" - -[profile.security-audit] -# Profile for security testing with all sanitizers -inherits = "dev" -# Sanitizers will be enabled via RUSTFLAGS \ No newline at end of file diff --git a/config/validate-production-config.sh b/config/validate-production-config.sh deleted file mode 100755 index f16bc0405..000000000 --- a/config/validate-production-config.sh +++ /dev/null @@ -1,180 +0,0 @@ -#!/bin/bash - -# Foxhunt HFT Trading System - Production Configuration Validator -# Validates that all required environment variables and configurations are set - -set -euo pipefail - -RED='\033[0;31m' -GREEN='\033[0;32m' -YELLOW='\033[1;33m' -NC='\033[0m' # No Color - -echo "πŸ” Foxhunt Production Configuration Validator" -echo "==============================================" -echo - -# Track validation status -VALIDATION_PASSED=true -MISSING_VARS=() -WARNING_VARS=() - -# Function to check if environment variable is set -check_env_var() { - local var_name="$1" - local description="$2" - local required="$3" - - if [[ -z "${!var_name:-}" ]]; then - if [[ "$required" == "true" ]]; then - echo -e "${RED}βœ—${NC} $var_name: $description" - MISSING_VARS+=("$var_name") - VALIDATION_PASSED=false - else - echo -e "${YELLOW}⚠${NC} $var_name: $description (optional)" - WARNING_VARS+=("$var_name") - fi - else - echo -e "${GREEN}βœ“${NC} $var_name: $description" - fi -} - -# Function to check if file exists -check_file() { - local file_path="$1" - local description="$2" - - if [[ ! -f "$file_path" ]]; then - echo -e "${RED}βœ—${NC} Missing file: $file_path ($description)" - VALIDATION_PASSED=false - else - echo -e "${GREEN}βœ“${NC} Found file: $file_path" - fi -} - -echo "πŸ“Š Database Configuration" -echo "========================" -check_env_var "DATABASE_URL" "PostgreSQL connection URL" true -check_env_var "REDIS_URL" "Redis connection URL" true -check_env_var "INFLUXDB_URL" "InfluxDB connection URL" true -check_env_var "INFLUXDB_TOKEN" "InfluxDB authentication token" true -echo - -echo "πŸ” Security Configuration" -echo "=========================" -check_env_var "FOXHUNT_JWT_SECRET" "JWT signing secret" true -check_env_var "FOXHUNT_SECRETS_ENCRYPTION_KEY" "Encryption key for secrets" true -check_env_var "VAULT_TOKEN" "HashiCorp Vault token" true -check_env_var "VAULT_ADDR" "HashiCorp Vault address" true -echo - -echo "πŸ’Ή Trading Configuration" -echo "========================" -check_env_var "FOXHUNT_TRADING_MODE" "Trading mode (paper/live)" true -check_env_var "POLYGON_API_KEY" "Polygon.io API key for market data" true -check_env_var "ICMARKETS_CLIENT_ID" "IC Markets client ID" true -check_env_var "ICMARKETS_CLIENT_SECRET" "IC Markets client secret" true -echo - -echo "⚑ Broker Configuration" -echo "=======================" -check_env_var "IB_HOST" "Interactive Brokers host" false -check_env_var "IB_PORT" "Interactive Brokers port" false -check_env_var "IB_CLIENT_ID" "Interactive Brokers client ID" false -check_env_var "BROKER_API_KEY" "Primary broker API key" false -check_env_var "BROKER_SECRET_KEY" "Primary broker secret key" false -echo - -echo "🌐 Service Endpoints" -echo "====================" -check_env_var "TRADING_ENGINE_ENDPOINT" "Trading engine gRPC endpoint" true -check_env_var "MARKET_DATA_ENDPOINT" "Market data service endpoint" true -check_env_var "RISK_MANAGEMENT_ENDPOINT" "Risk management service endpoint" true -check_env_var "BROKER_CONNECTOR_ENDPOINT" "Broker connector service endpoint" true -echo - -echo "🎯 Risk Management" -echo "==================" -check_env_var "FOXHUNT_MAX_DAILY_LOSS_PCT" "Maximum daily loss percentage" true -check_env_var "FOXHUNT_POSITION_LIMIT_PCT" "Position size limit percentage" true -check_env_var "FOXHUNT_LEVERAGE_LIMIT" "Maximum leverage limit" true -check_env_var "FOXHUNT_MAX_DRAWDOWN_PCT" "Maximum drawdown percentage" true -echo - -echo "🧠 ML Configuration" -echo "===================" -check_env_var "ML_MODEL_PATH" "ML model storage path" false -check_env_var "CUDA_DEVICE_ID" "CUDA device ID for GPU acceleration" false -check_env_var "ML_INFERENCE_TIMEOUT_MS" "ML inference timeout in milliseconds" false -echo - -echo "πŸ“ File Configuration Validation" -echo "==================================" -check_file "config/production.toml" "Main production configuration" -check_file "config/environments/production.env" "Production environment variables" -check_file "config/database/database.toml" "Database configuration" -check_file "config/security/security-hardening.toml" "Security hardening settings" -check_file "docker-compose.production.yml" "Production Docker Compose file" -echo - -echo "πŸ”§ Configuration File Validation" -echo "=================================" - -# Check if Docker Compose file is valid -if command -v docker-compose &> /dev/null; then - if docker-compose -f docker-compose.production.yml config &> /dev/null; then - echo -e "${GREEN}βœ“${NC} docker-compose.production.yml syntax is valid" - else - echo -e "${RED}βœ—${NC} docker-compose.production.yml has syntax errors" - VALIDATION_PASSED=false - fi -else - echo -e "${YELLOW}⚠${NC} docker-compose not available for validation" -fi - -# Check TOML files syntax -if command -v toml-test &> /dev/null; then - for toml_file in config/*.toml config/*/*.toml; do - if [[ -f "$toml_file" ]]; then - if toml-test "$toml_file" &> /dev/null; then - echo -e "${GREEN}βœ“${NC} $toml_file syntax is valid" - else - echo -e "${RED}βœ—${NC} $toml_file has syntax errors" - VALIDATION_PASSED=false - fi - fi - done -else - echo -e "${YELLOW}⚠${NC} toml-test not available for TOML validation" -fi - -echo - -echo "🏁 Validation Summary" -echo "====================+" - -if [[ ${#MISSING_VARS[@]} -gt 0 ]]; then - echo -e "${RED}Missing Required Variables:${NC}" - for var in "${MISSING_VARS[@]}"; do - echo -e " ${RED}β€’${NC} $var" - done - echo -fi - -if [[ ${#WARNING_VARS[@]} -gt 0 ]]; then - echo -e "${YELLOW}Optional Variables Not Set:${NC}" - for var in "${WARNING_VARS[@]}"; do - echo -e " ${YELLOW}β€’${NC} $var" - done - echo -fi - -if [[ "$VALIDATION_PASSED" == "true" ]]; then - echo -e "${GREEN}πŸŽ‰ Production Configuration Validation PASSED${NC}" - echo -e "${GREEN} All required configurations are present and valid${NC}" - exit 0 -else - echo -e "${RED}❌ Production Configuration Validation FAILED${NC}" - echo -e "${RED} Please fix the missing configurations above${NC}" - exit 1 -fi \ No newline at end of file diff --git a/docs/plans/2026-02-25-legacy-cleanup-design.md b/docs/plans/2026-02-25-legacy-cleanup-design.md new file mode 100644 index 000000000..dd1a9360d --- /dev/null +++ b/docs/plans/2026-02-25-legacy-cleanup-design.md @@ -0,0 +1,46 @@ +# Legacy Artifact Cleanup + +## Date: 2026-02-25 + +## Problem +Accumulated dead files from RunPod removal, tliβ†’fxt rename, disabled tests, +unused GitHub Actions workflows, and various development artifacts. + +## Tracked file deletions + +- `diagnostic_data/` β€” 4 DQN tuning artifacts +- `systemd/` β€” 5 files, unreferenced service definitions +- `.github/workflows/` β€” 24 GitHub Actions workflows (no GitHub remote, using GitLab CI) +- 20 `.disabled` files across tests/, common/, config/, data/, services/ +- `Dockerfile.foxhunt-build` β€” RunPod build, superseded by infra/docker/ +- `scripts/entrypoint.sh` β€” RunPod crash-log wrapper +- `scripts/entrypoint-self-terminate.sh` β€” RunPod pod auto-termination +- `.env.runpod.template` β€” RunPod S3 credentials template +- `.last_pod_id` β€” RunPod pod identifier +- `.gitignore.python` β€” Python gitignore, project is Rust-only +- `reports/` β€” 2 vim .swp files +- `Makefile` β€” convenience wrapper, not used in CI +- `justfile` β€” same +- `deploy.sh` β€” references deleted GitHub workflows +- `migrations/renumber_migrations.py` β€” one-off Python utility + +## .gitignore cleanup + +- Remove `!.env.runpod.template` exception +- Add `diagnostic_data/` + +## Untracked local deletions + +- `.venv/` (294M), `scripts/.venv` (101M) β€” Python virtualenvs +- `checkpoints/` (34M) β€” model checkpoints (already gitignored) +- `.swarm/` (4.4M), `.hive-mind/` (298K), `.claude-flow/` (254K) β€” tool runtime +- `coordination/` (4K), `memory/` (22K) β€” empty/session dirs +- `tli/` (55K), `runpod/` (69K) β€” leftover from deletions +- `.python-version`, `.env.runpod` β€” Python/RunPod local files + +## Confirmed kept + +- `storage/` β€” 9 crates depend on it +- `certs/` β€” referenced by 6 services + docker-compose +- `benches/` β€” active benchmarks +- `ml-data/`, `risk-data/`, `trading-data/` β€” workspace crates diff --git a/docs/plans/2026-02-25-repo-restructure-design.md b/docs/plans/2026-02-25-repo-restructure-design.md new file mode 100644 index 000000000..079e4dde4 --- /dev/null +++ b/docs/plans/2026-02-25-repo-restructure-design.md @@ -0,0 +1,163 @@ +# Repository Restructure β€” Approach C (Flat crates/ + Test Consolidation) + +## Date: 2026-02-25 + +## Problem +19 library crates at repo root mixed with 19+ non-crate directories. Root has 38+ dirs, +making it hard to distinguish code from config from infrastructure. + +## Target Structure + +``` +foxhunt/ +β”œβ”€β”€ Cargo.toml # workspace manifest only +β”œβ”€β”€ Cargo.lock +β”œβ”€β”€ .sqlx/ +β”œβ”€β”€ .gitignore +β”œβ”€β”€ .gitlab-ci.yml +β”œβ”€β”€ crates/ # ALL 17 library crates (flat) +β”‚ β”œβ”€β”€ common/ +β”‚ β”œβ”€β”€ config/ # Rust crate only (no config files) +β”‚ β”œβ”€β”€ storage/ +β”‚ β”œβ”€β”€ database/ +β”‚ β”œβ”€β”€ trading-engine/ # renamed from trading_engine +β”‚ β”œβ”€β”€ risk/ +β”‚ β”œβ”€β”€ backtesting/ +β”‚ β”œβ”€β”€ adaptive-strategy/ +β”‚ β”œβ”€β”€ ml/ +β”‚ β”œβ”€β”€ data/ +β”‚ β”œβ”€β”€ market-data/ +β”‚ β”œβ”€β”€ ml-data/ +β”‚ β”œβ”€β”€ risk-data/ +β”‚ β”œβ”€β”€ trading-data/ +β”‚ β”œβ”€β”€ model-loader/ # renamed from model_loader +β”‚ β”œβ”€β”€ ctrader-openapi/ +β”‚ └── web-gateway/ +β”œβ”€β”€ bin/fxt/ # CLI binary +β”œβ”€β”€ services/ # 8 microservices (internal paths unchanged) +β”‚ β”œβ”€β”€ api_gateway/ +β”‚ β”œβ”€β”€ backtesting_service/ +β”‚ β”œβ”€β”€ broker_gateway_service/ +β”‚ β”œβ”€β”€ data_acquisition_service/ +β”‚ β”œβ”€β”€ ml_training_service/ +β”‚ β”œβ”€β”€ trading_agent_service/ +β”‚ └── trading_service/ +β”œβ”€β”€ testing/ # consolidated test infrastructure +β”‚ β”œβ”€β”€ integration/ # was tests/ (main test crate) +β”‚ β”œβ”€β”€ e2e/ # was tests/e2e/ +β”‚ β”œβ”€β”€ vault-integration/ # was tests/e2e/vault_integration/ +β”‚ β”œβ”€β”€ load/ # was tests/load_tests/ +β”‚ β”œβ”€β”€ test-common/ # was tests/test_common/ +β”‚ β”œβ”€β”€ harness/ # was tests/harness/ +β”‚ β”œβ”€β”€ service-load/ # was services/load_tests/ +β”‚ β”œβ”€β”€ stress/ # was services/stress_tests/ +β”‚ β”œβ”€β”€ service-integration/# was services/integration_tests/ +β”‚ └── api-gateway-load/ # was services/api_gateway/load_tests/ +β”œβ”€β”€ benches/ # stays (Cargo convention) +β”œβ”€β”€ config/ # deployment config files only (split from crate) +β”‚ β”œβ”€β”€ alternative/ +β”‚ β”œβ”€β”€ base/ +β”‚ β”œβ”€β”€ database/ +β”‚ β”œβ”€β”€ grafana/ +β”‚ β”œβ”€β”€ k8s/ +β”‚ β”œβ”€β”€ ml/ +β”‚ β”œβ”€β”€ monitoring/ +β”‚ β”œβ”€β”€ prometheus/ +β”‚ β”œβ”€β”€ redis/ +β”‚ β”œβ”€β”€ security/ +β”‚ β”œβ”€β”€ sqlx/ +β”‚ └── tuning/ +β”œβ”€β”€ infra/ # terraform, docker, k8s helm +β”œβ”€β”€ web-dashboard/ # React frontend (not a Rust crate) +β”œβ”€β”€ scripts/ +β”œβ”€β”€ docs/ +β”œβ”€β”€ migrations/ +β”œβ”€β”€ sql/ +β”œβ”€β”€ certs/ +β”œβ”€β”€ monitoring/ +β”œβ”€β”€ test_data/ +└── vendor/ +``` + +Root: ~17 directories (down from 38+). + +## Crate Moves (git mv for history) + +### Library crates β†’ crates/ +| From | To | +|------|-----| +| common/ | crates/common/ | +| config/{Cargo.toml,src/,tests/,examples/} | crates/config/ | +| storage/ | crates/storage/ | +| database/ | crates/database/ | +| trading_engine/ | crates/trading-engine/ | +| risk/ | crates/risk/ | +| backtesting/ | crates/backtesting/ | +| adaptive-strategy/ | crates/adaptive-strategy/ | +| ml/ | crates/ml/ | +| data/ | crates/data/ | +| market-data/ | crates/market-data/ | +| ml-data/ | crates/ml-data/ | +| risk-data/ | crates/risk-data/ | +| trading-data/ | crates/trading-data/ | +| model_loader/ | crates/model-loader/ | +| ctrader-openapi/ | crates/ctrader-openapi/ | +| web-gateway/ | crates/web-gateway/ | + +### Binary β†’ bin/ +| From | To | +|------|-----| +| fxt/ | bin/fxt/ | + +### Config split +| From | To | +|------|-----| +| config/{Cargo.toml,src/,tests/,examples/} | crates/config/ | +| config/{base,database,grafana,k8s,ml,monitoring,prometheus,redis,security,sqlx,tuning,alternative}/ | config/ (stays at root) | + +### Test consolidation β†’ testing/ +| From | To | +|------|-----| +| tests/ (main crate) | testing/integration/ | +| tests/e2e/ | testing/e2e/ | +| tests/e2e/vault_integration/ | testing/vault-integration/ | +| tests/load_tests/ | testing/load/ | +| tests/test_common/ | testing/test-common/ | +| tests/harness/ | testing/harness/ | +| services/load_tests/ | testing/service-load/ | +| services/stress_tests/ | testing/stress/ | +| services/integration_tests/ | testing/service-integration/ | +| services/api_gateway/load_tests/ | testing/api-gateway-load/ | + +## Dependency Path Updates + +### A) Root Cargo.toml [workspace.dependencies] +14 entries: `path = "foo"` β†’ `path = "crates/foo"` (or `path = "bin/fxt"`) + +### B) Workspace members list +Explicit list with new paths. + +### C) Inter-crate path = "../foo" references +- Between crates in crates/: UNCHANGED (still siblings) +- Services β†’ crates: `../../foo` β†’ `../../crates/foo` +- Testing β†’ crates: similar depth adjustment +- bin/fxt β†’ crates: `../foo` β†’ `../crates/foo` +- Non-workspace test crates (test_common, harness, vault_integration): manual path update + +### D) build.rs proto paths +fxt/proto/ moves to bin/fxt/proto/ β€” all build.rs files that reference `fxt/proto/` need updating. + +## .claude/ +- Keep tracked (201 files β€” agent configs, skills, settings are useful across clones) + +## Unchanged +- services/ internal structure (only load_tests/stress_tests/integration_tests move out) +- benches/ stays at root (Cargo convention, [[bench]] in root Cargo.toml) +- infra/, scripts/, docs/, migrations/, sql/, certs/, monitoring/, test_data/, web-dashboard/ +- .sqlx/ stays at root + +## Risks +- Many Cargo.toml path edits β€” high chance of typos β†’ mitigated by cargo check +- build.rs proto paths may break β†’ verify with cargo check +- CI cache invalidation (one-time rebuild) +- IDE needs workspace reload after restructure diff --git a/docs/plans/2026-02-25-repo-restructure-plan.md b/docs/plans/2026-02-25-repo-restructure-plan.md new file mode 100644 index 000000000..c5f0a1be7 --- /dev/null +++ b/docs/plans/2026-02-25-repo-restructure-plan.md @@ -0,0 +1,640 @@ +# Repository Restructure Implementation Plan + +> **For Claude:** REQUIRED SUB-SKILL: Use superpowers:executing-plans to implement this plan task-by-task. + +**Goal:** Move 17 library crates into `crates/`, CLI binary into `bin/`, test infra into `testing/`, split config crate from config files. Reduce root from 38+ dirs to ~17. + +**Architecture:** Flat `crates/*` layout (rust-analyzer/Bevy convention). All moves use `git mv` for history preservation. Single atomic commit β€” everything moves together. + +**Tech Stack:** Cargo workspace, git mv, sed for bulk Cargo.toml updates, `SQLX_OFFLINE=true cargo check --workspace` for verification. + +**Design doc:** `docs/plans/2026-02-25-repo-restructure-design.md` + +--- + +## IMPORTANT: Execution Notes + +- **Execute in a worktree** (`isolation: "worktree"`) +- **ALL tasks must complete before cargo check** β€” partial moves break everything +- Directory names keep existing casing (`trading_engine` stays `trading_engine`, `model_loader` stays `model_loader`) to minimize rename churn +- Crate *names* in `Cargo.toml` are unchanged β€” only directory paths change +- The root `Cargo.toml` is package `foxhunt` (not virtual) β€” it has `[[bench]]` entries and stays at root + +--- + +### Task 1: Create target directories + +**Step 1: Create directory structure** + +```bash +mkdir -p crates bin testing +``` + +**Step 2: Verify** + +```bash +ls -d crates bin testing +``` + +--- + +### Task 2: Move library crates to crates/ + +All 17 library crates move from root into `crates/`. Use `git mv` for each. + +**Step 1: Move all library crates** + +```bash +git mv common crates/common +git mv storage crates/storage +git mv database crates/database +git mv trading_engine crates/trading_engine +git mv risk crates/risk +git mv backtesting crates/backtesting +git mv adaptive-strategy crates/adaptive-strategy +git mv ml crates/ml +git mv data crates/data +git mv market-data crates/market-data +git mv ml-data crates/ml-data +git mv risk-data crates/risk-data +git mv trading-data crates/trading-data +git mv model_loader crates/model_loader +git mv ctrader-openapi crates/ctrader-openapi +git mv web-gateway crates/web-gateway +``` + +Note: `config/` is special (split in Task 4). Don't move it here. + +**Step 2: Verify all moved** + +```bash +ls crates/ +# Should show: adaptive-strategy backtesting common ctrader-openapi data database +# market-data ml ml-data model_loader risk risk-data storage +# trading-data trading_engine web-gateway +``` + +--- + +### Task 3: Move CLI binary to bin/ + +**Step 1: Move fxt** + +```bash +git mv fxt bin/fxt +``` + +**Step 2: Verify** + +```bash +ls bin/fxt/Cargo.toml bin/fxt/src/main.rs bin/fxt/proto/ +``` + +--- + +### Task 4: Split config crate from config files + +The `config/` directory contains both the Rust crate AND deployment config files. +We need to move the crate to `crates/config/` while keeping config files at root. + +**Step 1: Move the entire config/ to crates/config/ first** + +```bash +git mv config crates/config +``` + +**Step 2: Move non-crate config directories back to root config/** + +```bash +mkdir -p config +git mv crates/config/alternative config/alternative +git mv crates/config/base config/base +git mv crates/config/database config/database +git mv crates/config/grafana config/grafana +git mv crates/config/k8s config/k8s +git mv crates/config/ml config/ml +git mv crates/config/monitoring config/monitoring +git mv crates/config/prometheus config/prometheus +git mv crates/config/redis config/redis +git mv crates/config/security config/security +git mv crates/config/sqlx config/sqlx +git mv crates/config/tuning config/tuning +``` + +**Step 3: Verify split** + +```bash +# Crate should have only code +ls crates/config/ +# Expected: Cargo.toml src/ tests/ examples/ + +# Config files at root +ls config/ +# Expected: alternative base database grafana k8s ml monitoring prometheus redis security sqlx tuning +``` + +--- + +### Task 5: Consolidate test crates into testing/ + +Move all test infrastructure from `tests/` and `services/` into `testing/`. + +**Step 1: Move test crates from tests/** + +```bash +# Move the main test crate (tests/ root files) +# First move the nested crates OUT, then move the parent +git mv tests/e2e/vault_integration testing/vault-integration +git mv tests/e2e testing/e2e +git mv tests/load_tests testing/load +git mv tests/test_common testing/test-common +git mv tests/harness testing/harness +# Now move the remaining tests/ (the main integration test crate) +git mv tests testing/integration +``` + +**Step 2: Move test crates from services/** + +```bash +git mv services/load_tests testing/service-load +git mv services/stress_tests testing/stress +git mv services/integration_tests testing/service-integration +git mv services/api_gateway/load_tests testing/api-gateway-load +``` + +**Step 3: Verify** + +```bash +ls testing/ +# Expected: api-gateway-load e2e harness integration load +# service-integration service-load stress test-common vault-integration +``` + +--- + +### Task 6: Update root Cargo.toml β€” workspace members + +**File:** `Cargo.toml` (root) + +Replace the entire `members = [...]` block with: + +```toml +members = [ + # Library crates + "crates/trading_engine", + "crates/risk", + "crates/risk-data", + "crates/trading-data", + "crates/ml", + "crates/ml-data", + "crates/data", + "crates/backtesting", + "crates/adaptive-strategy", + "crates/common", + "crates/storage", + "crates/model_loader", + "crates/market-data", + "crates/database", + "crates/config", + "crates/ctrader-openapi", + "crates/web-gateway", + # CLI binary + "bin/fxt", + # Services + "services/backtesting_service", + "services/broker_gateway_service", + "services/trading_service", + "services/ml_training_service", + "services/data_acquisition_service", + "services/trading_agent_service", + "services/api_gateway", + # Testing + "testing/integration", + "testing/e2e", + "testing/load", + "testing/service-load", + "testing/stress", + "testing/service-integration", + "testing/api-gateway-load", + "testing/vault-integration", +] +``` + +**Remove:** `"performance-tests"` (phantom member β€” directory doesn't exist). + +**Remove:** `"services/api_gateway/load_tests"`, `"services/load_tests"`, `"services/stress_tests"`, `"services/integration_tests"` (moved to testing/). + +**Remove:** `"tests"`, `"tests/e2e"`, `"tests/load_tests"`, `"tests/e2e/vault_integration"` (moved to testing/). + +**Note:** `testing/test-common` and `testing/harness` are NOT workspace members (they use direct path deps, not `workspace = true`). Keep them as non-members. + +--- + +### Task 7: Update root Cargo.toml β€” workspace dependency paths + +**File:** `Cargo.toml` (root), `[workspace.dependencies]` section + +Update all 14 path entries: + +| Old | New | +|-----|-----| +| `trading_engine = { path = "trading_engine" }` | `trading_engine = { path = "crates/trading_engine" }` | +| `data = { path = "data" }` | `data = { path = "crates/data" }` | +| `fxt = { path = "fxt" }` | `fxt = { path = "bin/fxt" }` | +| `risk = { path = "risk" }` | `risk = { path = "crates/risk" }` | +| `risk-data = { path = "risk-data" }` | `risk-data = { path = "crates/risk-data" }` | +| `backtesting = { path = "backtesting" }` | `backtesting = { path = "crates/backtesting" }` | +| `ml = { path = "ml", default-features = false }` | `ml = { path = "crates/ml", default-features = false }` | +| `adaptive-strategy = { path = "adaptive-strategy" }` | `adaptive-strategy = { path = "crates/adaptive-strategy" }` | +| `common = { path = "common" }` | `common = { path = "crates/common" }` | +| `storage = { path = "storage" }` | `storage = { path = "crates/storage" }` | +| `market-data = { path = "market-data" }` | `market-data = { path = "crates/market-data" }` | +| `config = { path = "config" }` | `config = { path = "crates/config" }` | +| `database = { path = "database" }` | `database = { path = "crates/database" }` | +| `ctrader-openapi = { path = "ctrader-openapi" }` | `ctrader-openapi = { path = "crates/ctrader-openapi" }` | + +--- + +### Task 8: Update inter-crate direct path dependencies (crates/) + +**Key insight:** All library crates are now siblings under `crates/`. References like `path = "../common"` between crates **stay the same** β€” they still resolve correctly. + +**NO CHANGES needed for these files:** +- `crates/risk/Cargo.toml`: `common = { path = "../common" }` βœ“ +- `crates/backtesting/Cargo.toml`: `common = { path = "../common" }` βœ“ +- `crates/database/Cargo.toml`: `common = { path = "../common" }`, `config = { path = "../config" }` βœ“ +- `crates/market-data/Cargo.toml`: `common = { path = "../common" }` βœ“ +- `crates/ml/Cargo.toml`: `risk = { path = "../risk" }`, `storage = { path = "../storage" }`, `data = { path = "../data" }` βœ“ +- `crates/trading_engine/Cargo.toml`: `common = { path = "../common" }` βœ“ +- `crates/data/Cargo.toml`: `common = { path = "../common" }` βœ“ +- `crates/model_loader/Cargo.toml`: `storage = { path = "../storage" }` βœ“ +- `crates/adaptive-strategy/Cargo.toml`: `common = { path = "../common" }`, `backtesting = { path = "../backtesting" }`, `data = { path = "../data" }` βœ“ +- `crates/ml-data/Cargo.toml`: `database = { path = "../database" }`, `config = { path = "../config" }` βœ“ +- `crates/common/Cargo.toml`: `config = { path = "../config" }`, `ml = { path = "../ml" }` βœ“ + +--- + +### Task 9: Update service Cargo.toml path dependencies + +Services stay at `services/` but now reference crates under `crates/` instead of root. + +**File: `services/trading_service/Cargo.toml`** + +| Old | New | +|-----|-----| +| `database = { path = "../../database" }` | `database = { path = "../../crates/database" }` | +| `ml-data = { path = "../../ml-data" }` | `ml-data = { path = "../../crates/ml-data" }` | +| `api_gateway = { path = "../api_gateway" }` | unchanged (serviceβ†’service) | + +**File: `services/backtesting_service/Cargo.toml`** + +| Old | New | +|-----|-----| +| `model_loader = { path = "../../model_loader" }` | `model_loader = { path = "../../crates/model_loader" }` | +| `ml-data = { path = "../../ml-data" }` | `ml-data = { path = "../../crates/ml-data" }` | + +**File: `services/ml_training_service/Cargo.toml`** + +| Old | New | +|-----|-----| +| `ml-data = { path = "../../ml-data" }` | `ml-data = { path = "../../crates/ml-data" }` | + +**File: `services/trading_agent_service/Cargo.toml`** + +| Old | New | +|-----|-----| +| `risk = { path = "../../risk" }` | `risk = { path = "../../crates/risk" }` | +| `ml = { path = "../../ml", ... }` | `ml = { path = "../../crates/ml", ... }` | + +**File: `services/integration_tests/Cargo.toml`** β†’ now at `testing/service-integration/Cargo.toml` + +| Old | New | +|-----|-----| +| `backtesting_service = { path = "../backtesting_service" }` | `backtesting_service = { path = "../services/backtesting_service" }` | + +Wait β€” this crate moved to `testing/service-integration/`. From there, `../` goes to `testing/`, not `services/`. So the path needs to go up two levels: `../../services/backtesting_service`. + +| Old (at `services/integration_tests/`) | New (at `testing/service-integration/`) | +|-----|-----| +| `backtesting_service = { path = "../backtesting_service" }` | `backtesting_service = { path = "../../services/backtesting_service" }` | + +--- + +### Task 10: Update testing crate path dependencies + +All test crates moved from `tests/` or `services/` to `testing/`. Path depths changed. + +**File: `testing/integration/Cargo.toml`** (was `tests/Cargo.toml`) + +| Old (at `tests/`) | New (at `testing/integration/`) | +|-----|-----| +| `config = { path = "../config" }` | `config = { path = "../../crates/config" }` | +| `trading_service = { path = "../services/trading_service" }` | `trading_service = { path = "../../services/trading_service" }` | + +**File: `testing/e2e/Cargo.toml`** (was `tests/e2e/`) + +| Old (at `tests/e2e/`) | New (at `testing/e2e/`) | +|-----|-----| +| `trading_engine = { path = "../../trading_engine" }` | `trading_engine = { path = "../../crates/trading_engine" }` | +| `data = { path = "../../data" }` | `data = { path = "../../crates/data" }` | +| `ml = { path = "../../ml" }` | `ml = { path = "../../crates/ml" }` | +| `ml_training_service = { path = "../../services/ml_training_service" }` | unchanged (same depth) | +| `risk = { path = "../../risk" }` | `risk = { path = "../../crates/risk" }` | +| `config = { path = "../../config" }` | `config = { path = "../../crates/config" }` | +| `common = { path = "../../common" }` | `common = { path = "../../crates/common" }` | + +**File: `testing/harness/Cargo.toml`** (was `tests/harness/`) + +| Old (at `tests/harness/`) | New (at `testing/harness/`) | +|-----|-----| +| `trading_engine = { path = "../../trading_engine" }` | `trading_engine = { path = "../../crates/trading_engine" }` | +| `data = { path = "../../data" }` | `data = { path = "../../crates/data" }` | +| `ml = { path = "../../ml" }` | `ml = { path = "../../crates/ml" }` | +| `risk = { path = "../../risk" }` | `risk = { path = "../../crates/risk" }` | +| `config = { path = "../../config" }` | `config = { path = "../../crates/config" }` | +| `common = { path = "../../common" }` | `common = { path = "../../crates/common" }` | + +**File: `testing/test-common/Cargo.toml`** (was `tests/test_common/`) + +| Old (at `tests/test_common/`) | New (at `testing/test-common/`) | +|-----|-----| +| `common = { path = "../../common" }` | `common = { path = "../../crates/common" }` | +| `ml = { path = "../../ml" }` | `ml = { path = "../../crates/ml" }` | +| `risk = { path = "../../risk" }` | `risk = { path = "../../crates/risk" }` | +| `config = { path = "../../config" }` | `config = { path = "../../crates/config" }` | + +**File: `testing/load/Cargo.toml`** (was `tests/load_tests/`) + +No direct path deps to crates (uses workspace = true for what it needs). Check build.rs β€” see Task 12. + +**File: `testing/api-gateway-load/Cargo.toml`** (was `services/api_gateway/load_tests/`) + +| Old (at `services/api_gateway/load_tests/`) | New (at `testing/api-gateway-load/`) | +|-----|-----| +| `common = { path = "../../../common" }` | `common = { path = "../../crates/common" }` | + +**File: `testing/vault-integration/Cargo.toml`** (was `tests/e2e/vault_integration/`) + +No direct path deps to crates (check and verify). + +--- + +### Task 11: Update build.rs proto paths β€” fxt references + +`fxt/proto/` moved to `bin/fxt/proto/`. All build.rs files referencing `fxt/proto/` from outside fxt need updating. + +**File: `services/api_gateway/build.rs`** + +| Old | New | +|-----|-----| +| `"../../fxt/proto/trading.proto"` | `"../../bin/fxt/proto/trading.proto"` | +| `"../../fxt/proto"` (include dir) | `"../../bin/fxt/proto"` | +| `cargo:rerun-if-changed=../../fxt/proto/trading.proto` | `cargo:rerun-if-changed=../../bin/fxt/proto/trading.proto` | + +**File: `services/ml_training_service/build.rs`** + +| Old | New | +|-----|-----| +| `"../../fxt/proto/trading.proto"` | `"../../bin/fxt/proto/trading.proto"` | +| `"../../fxt/proto"` (include dir) | `"../../bin/fxt/proto"` | +| `cargo:rerun-if-changed=../../fxt/proto/trading.proto` | `cargo:rerun-if-changed=../../bin/fxt/proto/trading.proto` | + +**File: `services/backtesting_service/build.rs`** + +| Old | New | +|-----|-----| +| `"../../fxt/proto/trading.proto"` | `"../../bin/fxt/proto/trading.proto"` | +| `"../../fxt/proto"` (include dir) | `"../../bin/fxt/proto"` | + +**File: `crates/web-gateway/build.rs`** (was `web-gateway/`, now in `crates/`) + +Old path was `../fxt/proto/`. Now from `crates/web-gateway/`, fxt is at `../../bin/fxt/`: + +| Old | New | +|-----|-----| +| `"../fxt/proto/trading.proto"` | `"../../bin/fxt/proto/trading.proto"` | +| `"../fxt/proto/health.proto"` | `"../../bin/fxt/proto/health.proto"` | +| `"../fxt/proto/ml.proto"` | `"../../bin/fxt/proto/ml.proto"` | +| `"../fxt/proto/config.proto"` | `"../../bin/fxt/proto/config.proto"` | +| `"../fxt/proto/ml_training.proto"` | `"../../bin/fxt/proto/ml_training.proto"` | +| `"../fxt/proto/trading_agent.proto"` | `"../../bin/fxt/proto/trading_agent.proto"` | +| `"../fxt/proto"` (include dir) | `"../../bin/fxt/proto"` | +| All `cargo:rerun-if-changed=../fxt/proto/...` | `cargo:rerun-if-changed=../../bin/fxt/proto/...` | + +**File: `bin/fxt/build.rs`** β€” NO CHANGE. Proto paths are relative to the crate dir (`proto/trading.proto` etc.) and moved with the crate. + +**File: `crates/ctrader-openapi/build.rs`** β€” NO CHANGE. Uses local `proto/` dir that moved with the crate. + +--- + +### Task 12: Update build.rs proto paths β€” test crate references + +**File: `testing/service-integration/build.rs`** (was `services/integration_tests/`) + +Old path from `services/integration_tests/` referenced `../../fxt/proto/`. New path from `testing/service-integration/` is also `../../bin/fxt/proto/` (same depth from root). + +| Old | New | +|-----|-----| +| `"../../fxt/proto/trading.proto"` | `"../../bin/fxt/proto/trading.proto"` | +| `"../../fxt/proto/ml.proto"` | `"../../bin/fxt/proto/ml.proto"` | +| `"../../fxt/proto/config.proto"` | `"../../bin/fxt/proto/config.proto"` | +| `"../../fxt/proto/health.proto"` | `"../../bin/fxt/proto/health.proto"` | +| `"../../fxt/proto"` (include dir) | `"../../bin/fxt/proto"` | +| All `cargo:rerun-if-changed=../../fxt/proto/...` | `cargo:rerun-if-changed=../../bin/fxt/proto/...` | + +**File: `testing/service-load/build.rs`** (was `services/load_tests/`) + +Old path from `services/load_tests/` referenced `../trading_service/proto/`. New path from `testing/service-load/`: + +| Old | New | +|-----|-----| +| `"../trading_service/proto/trading.proto"` | `"../../services/trading_service/proto/trading.proto"` | +| `cargo:rerun-if-changed=../trading_service/proto/trading.proto` | `cargo:rerun-if-changed=../../services/trading_service/proto/trading.proto` | + +**File: `testing/load/build.rs`** (was `tests/load_tests/`) + +Old path from `tests/load_tests/` referenced `../../services/trading_service/proto/`. New path from `testing/load/` is the same depth: + +| Old | New | +|-----|-----| +| `"../../services/trading_service/proto/trading.proto"` | unchanged (same depth from root) | + +**File: `testing/e2e/build.rs`** (was `tests/e2e/`) + +Old path from `tests/e2e/` referenced `../../services/...` and `../../fxt/proto/`. New path from `testing/e2e/` is same depth: + +| Old | New | +|-----|-----| +| `"../../services/trading_service/proto/trading.proto"` | unchanged (same depth) | +| `"../../services/trading_service/proto/config.proto"` | unchanged | +| `"../../services/trading_service/proto/risk.proto"` | unchanged | +| `"../../services/ml_training_service/proto/ml_training.proto"` | unchanged | +| `"../../services/trading_service/proto"` (include dir) | unchanged | +| `"../../services/ml_training_service/proto"` (include dir) | unchanged | +| `"../../fxt/proto/trading.proto"` | `"../../bin/fxt/proto/trading.proto"` | +| `"../../fxt/proto"` (include dir) | `"../../bin/fxt/proto"` | + +**File: `testing/harness/build.rs`** (was `tests/harness/`) + +Old referenced `../e2e/proto/`. Now from `testing/harness/`, e2e is a sibling: + +| Old | New | +|-----|-----| +| `"../e2e/proto/trading.proto"` | unchanged (e2e is still sibling in testing/) | +| `"../e2e/proto/ml_training.proto"` | unchanged | +| `"../e2e/proto"` (include dir) | unchanged | + +--- + +### Task 13: Update services build.rs β€” inter-service proto references + +Services that reference OTHER services' protos don't change (services stay in `services/`): + +**NO CHANGES needed:** +- `services/api_gateway/build.rs`: `../trading_service/proto/...` βœ“ +- `services/api_gateway/build.rs`: `../ml_training_service/proto/...` βœ“ +- `services/api_gateway/build.rs`: `../trading_agent_service/proto/...` βœ“ +- `services/trading_service/build.rs`: local `proto/` βœ“ +- `services/data_acquisition_service/build.rs`: local `proto/` βœ“ +- `services/broker_gateway_service/build.rs`: local `proto/` βœ“ +- `services/trading_agent_service/build.rs`: local `proto/` βœ“ +- `services/ml_training_service/build.rs`: local `proto/` βœ“ + +--- + +### Task 14: Verify with cargo check + +**Step 1: Full workspace check** + +```bash +SQLX_OFFLINE=true cargo check --workspace 2>&1 +``` + +Expected: compiles with 0 errors. Warnings are OK. + +**Step 2: If errors, iterate** + +Common issues: +- Typo in a path β†’ fix the specific Cargo.toml +- Proto path wrong β†’ fix the specific build.rs +- Workspace member missing β†’ add to members list + +--- + +### Task 15: Clean up stray references + +**Step 1: Search for any remaining old paths** + +```bash +# Find any Cargo.toml still referencing root-level crate dirs +grep -rn 'path = ".*\.\./trading_engine"' --include="Cargo.toml" . | grep -v crates/ +grep -rn 'path = ".*\.\./common"' --include="Cargo.toml" . | grep -v crates/ +grep -rn 'path = ".*\.\./risk"' --include="Cargo.toml" . | grep -v crates/ +grep -rn 'path = ".*\.\./ml"' --include="Cargo.toml" . | grep -v crates/ + +# Find any build.rs still referencing old fxt/ path +grep -rn 'fxt/proto' --include="build.rs" . | grep -v 'bin/fxt' +``` + +Fix any remaining references found. + +**Step 2: Check .gitignore for old paths** + +Review `.gitignore` for any paths that need updating (e.g., `data/cache/` is now `crates/data/cache/`). Most gitignore entries use patterns (not absolute paths) and should still work. + +--- + +### Task 16: Commit + +**Step 1: Stage all changes** + +```bash +git add -A +``` + +**Step 2: Review the diff** + +```bash +git diff --cached --stat | tail -5 +# Expect: many renames, Cargo.toml edits, build.rs edits +# NO source code (.rs) changes except build.rs files +``` + +**Step 3: Commit** + +```bash +git commit -m "$(cat <<'EOF' +refactor: restructure repo β€” crates/, bin/, testing/ layout + +Move 17 library crates into crates/, CLI binary into bin/fxt, +consolidate 10 test crates into testing/, split config crate +from deployment config files. + +Root directory reduced from 38+ to ~17 directories. +All Cargo.toml paths and build.rs proto refs updated. + +Co-Authored-By: Claude Opus 4.6 +EOF +)" +``` + +--- + +## Path Update Reference Tables + +### Services β†’ crates (Cargo.toml) + +All services are at depth 2 (`services/foo/`). Old root crates at depth 1. New crates at `crates/foo` depth 2. +Pattern: `../../foo` β†’ `../../crates/foo` + +| Service | Dependency | Old Path | New Path | +|---------|-----------|----------|----------| +| trading_service | database | `../../database` | `../../crates/database` | +| trading_service | ml-data | `../../ml-data` | `../../crates/ml-data` | +| backtesting_service | model_loader | `../../model_loader` | `../../crates/model_loader` | +| backtesting_service | ml-data | `../../ml-data` | `../../crates/ml-data` | +| ml_training_service | ml-data | `../../ml-data` | `../../crates/ml-data` | +| trading_agent_service | risk | `../../risk` | `../../crates/risk` | +| trading_agent_service | ml | `../../ml` | `../../crates/ml` | + +### Testing β†’ crates (Cargo.toml) + +All testing crates at depth 2 (`testing/foo/`). Pattern: `../../foo` β†’ `../../crates/foo` + +| Test Crate | Dependency | Old Path | New Path | +|-----------|-----------|----------|----------| +| integration | config | `../config` | `../../crates/config` | +| integration | trading_service | `../services/trading_service` | `../../services/trading_service` | +| e2e | trading_engine | `../../trading_engine` | `../../crates/trading_engine` | +| e2e | data | `../../data` | `../../crates/data` | +| e2e | ml | `../../ml` | `../../crates/ml` | +| e2e | risk | `../../risk` | `../../crates/risk` | +| e2e | config | `../../config` | `../../crates/config` | +| e2e | common | `../../common` | `../../crates/common` | +| e2e | ml_training_service | `../../services/ml_training_service` | unchanged | +| harness | trading_engine | `../../trading_engine` | `../../crates/trading_engine` | +| harness | data | `../../data` | `../../crates/data` | +| harness | ml | `../../ml` | `../../crates/ml` | +| harness | risk | `../../risk` | `../../crates/risk` | +| harness | config | `../../config` | `../../crates/config` | +| harness | common | `../../common` | `../../crates/common` | +| test-common | common | `../../common` | `../../crates/common` | +| test-common | ml | `../../ml` | `../../crates/ml` | +| test-common | risk | `../../risk` | `../../crates/risk` | +| test-common | config | `../../config` | `../../crates/config` | +| api-gateway-load | common | `../../../common` | `../../crates/common` | +| service-integration | backtesting_service | `../backtesting_service` | `../../services/backtesting_service` | +| service-load | (build.rs only) | `../trading_service/proto/` | `../../services/trading_service/proto/` | + +### build.rs fxt/proto references + +Pattern: `fxt/proto` β†’ `bin/fxt/proto` (from various depths) + +| File | Old | New | +|------|-----|-----| +| `services/api_gateway/build.rs` | `../../fxt/proto` | `../../bin/fxt/proto` | +| `services/ml_training_service/build.rs` | `../../fxt/proto` | `../../bin/fxt/proto` | +| `services/backtesting_service/build.rs` | `../../fxt/proto` | `../../bin/fxt/proto` | +| `crates/web-gateway/build.rs` | `../fxt/proto` | `../../bin/fxt/proto` | +| `testing/service-integration/build.rs` | `../../fxt/proto` | `../../bin/fxt/proto` | +| `testing/e2e/build.rs` | `../../fxt/proto` | `../../bin/fxt/proto` | diff --git a/migrations/002_risk_events.sql.broken b/migrations/002_risk_events.sql.broken deleted file mode 100644 index be5ce63d3..000000000 --- a/migrations/002_risk_events.sql.broken +++ /dev/null @@ -1,890 +0,0 @@ --- ================================================================================================ --- Migration 002: Risk Events Schema --- Comprehensive risk management event storage with real-time monitoring --- Production-ready with compliance, stress testing, and alert capabilities --- ================================================================================================ - --- ================================================================================================ --- RISK EVENT TYPES AND ENUMS --- Comprehensive classification for all risk-related events --- ================================================================================================ - -CREATE TYPE risk_event_type AS ENUM ( - 'var_breach', - 'exposure_limit_breach', - 'position_limit_breach', - 'concentration_risk', - 'leverage_excess', - 'margin_call', - 'drawdown_limit', - 'volatility_spike', - 'correlation_breakdown', - 'liquidity_shortage', - 'stress_test_failure', - 'compliance_violation', - 'model_validation_error', - 'circuit_breaker_triggered', - 'emergency_shutdown', - 'risk_limit_update', - 'model_recalibration', - 'backtest_failure' -); - -CREATE TYPE risk_severity AS ENUM ( - 'info', -- Information only - 'low', -- Minor risk, monitoring required - 'medium', -- Elevated risk, caution advised - 'high', -- Significant risk, action may be required - 'critical', -- Immediate action required - 'emergency' -- System shutdown level risk -); - -CREATE TYPE risk_action_type AS ENUM ( - 'alert_only', - 'reduce_position', - 'close_position', - 'halt_trading', - 'reduce_leverage', - 'increase_margin', - 'manual_intervention', - 'system_shutdown', - 'compliance_review' -); - -CREATE TYPE risk_metric_type AS ENUM ( - 'var_1d', - 'var_10d', - 'cvar_1d', - 'cvar_10d', - 'exposure_gross', - 'exposure_net', - 'leverage_ratio', - 'concentration_single', - 'concentration_sector', - 'beta_portfolio', - 'sharpe_ratio', - 'max_drawdown', - 'volatility_realized', - 'volatility_implied', - 'correlation_matrix', - 'margin_excess', - 'margin_requirement', - 'liquidity_score' -); - --- ================================================================================================ --- RISK EVENTS TABLE --- Immutable event store for all risk-related activities --- ================================================================================================ -CREATE TABLE risk_events ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - event_id BIGSERIAL NOT NULL, - correlation_id UUID NOT NULL, - - -- Timing with nanosecond precision - event_timestamp ns_timestamp NOT NULL, - detected_timestamp ns_timestamp NOT NULL, - acknowledged_timestamp ns_timestamp, - resolved_timestamp ns_timestamp, - - -- Risk event classification - event_type risk_event_type NOT NULL, - severity risk_severity NOT NULL, - risk_metric risk_metric_type, - - -- Risk context - symbol VARCHAR(32), - account_id VARCHAR(64), - strategy_id VARCHAR(100), - portfolio_id VARCHAR(100), - - -- Risk values and thresholds - threshold_value DECIMAL(20, 8), - actual_value DECIMAL(20, 8), - breach_percentage DECIMAL(8, 4), -- How much threshold was exceeded by - risk_score DECIMAL(10, 6), -- Normalized risk score 0-1 - - -- Event details - description TEXT NOT NULL, - risk_model VARCHAR(100), -- Which risk model detected this - model_version VARCHAR(50), - - -- Actions and responses - recommended_action risk_action_type, - action_taken risk_action_type, - action_details JSONB, - automated_response BOOLEAN DEFAULT FALSE, - - -- System context - source_system VARCHAR(100) NOT NULL, - node_id VARCHAR(50) NOT NULL, - process_id INTEGER NOT NULL, - - -- Audit and compliance - acknowledged_by VARCHAR(64), - resolved_by VARCHAR(64), - escalated_to VARCHAR(64), - compliance_notification_sent BOOLEAN DEFAULT FALSE, - - -- Additional data - event_data JSONB NOT NULL, -- Complete risk event payload - metadata JSONB, - - -- Partition key - event_date DATE GENERATED ALWAYS AS (ns_to_date_immutable(event_timestamp)) STORED, - - -- Constraints - CONSTRAINT chk_risk_timestamps CHECK ( - detected_timestamp >= event_timestamp AND - (acknowledged_timestamp IS NULL OR acknowledged_timestamp >= detected_timestamp) AND - (resolved_timestamp IS NULL OR resolved_timestamp >= COALESCE(acknowledged_timestamp, detected_timestamp)) - ), - CONSTRAINT chk_breach_percentage CHECK ( - breach_percentage IS NULL OR breach_percentage >= 0 - ) -) PARTITION BY RANGE (event_date); - --- ================================================================================================ --- RISK METRICS TABLE --- Current and historical risk metric values --- ================================================================================================ -CREATE TABLE risk_metrics ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - metric_name risk_metric_type NOT NULL, - - -- Scope identifiers - symbol VARCHAR(32), -- NULL for portfolio-level metrics - account_id VARCHAR(64), - strategy_id VARCHAR(100), - portfolio_id VARCHAR(100), - - -- Metric values - value DECIMAL(20, 8) NOT NULL, - confidence_interval_lower DECIMAL(20, 8), - confidence_interval_upper DECIMAL(20, 8), - confidence_level DECIMAL(5, 4) DEFAULT 0.95, -- 95% confidence by default - - -- Thresholds and limits - warning_threshold DECIMAL(20, 8), - breach_threshold DECIMAL(20, 8), - emergency_threshold DECIMAL(20, 8), - - -- Calculation context - calculation_timestamp ns_timestamp NOT NULL, - data_timestamp ns_timestamp NOT NULL, -- Timestamp of underlying data - model_name VARCHAR(100) NOT NULL, - model_version VARCHAR(50) NOT NULL, - calculation_method VARCHAR(200), - - -- Time horizon and parameters - time_horizon_days INTEGER, - lookback_days INTEGER, - confidence_level_pct DECIMAL(5, 2), - - -- Status and validation - is_valid BOOLEAN DEFAULT TRUE, - validation_errors TEXT[], - last_updated ns_timestamp NOT NULL, - - -- Additional context - market_conditions JSONB, -- Market state when calculated - calculation_details JSONB, -- Model parameters and inputs - - -- Partition key - metric_date DATE GENERATED ALWAYS AS (ns_to_date_immutable(calculation_timestamp)) STORED, - - -- Constraints - CONSTRAINT chk_confidence_level CHECK (confidence_level > 0 AND confidence_level <= 1), - CONSTRAINT chk_time_horizons CHECK ( - time_horizon_days IS NULL OR time_horizon_days > 0 - ), - CONSTRAINT chk_calculation_timestamps CHECK ( - calculation_timestamp >= data_timestamp - ) -) PARTITION BY RANGE (metric_date); - --- ================================================================================================ --- RISK LIMITS TABLE --- Configurable risk limits and thresholds --- ================================================================================================ -CREATE TABLE risk_limits ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - limit_name VARCHAR(200) NOT NULL, - limit_type risk_metric_type NOT NULL, - - -- Scope (hierarchy: global -> account -> strategy -> symbol) - scope_level VARCHAR(20) NOT NULL CHECK (scope_level IN ('global', 'account', 'strategy', 'symbol')), - account_id VARCHAR(64), - strategy_id VARCHAR(100), - symbol VARCHAR(32), - - -- Limit values - warning_threshold DECIMAL(20, 8), - breach_threshold DECIMAL(20, 8) NOT NULL, - emergency_threshold DECIMAL(20, 8), - - -- Time-based limits - intraday_limit DECIMAL(20, 8), - daily_limit DECIMAL(20, 8), - weekly_limit DECIMAL(20, 8), - monthly_limit DECIMAL(20, 8), - - -- Limit behavior - is_active BOOLEAN DEFAULT TRUE, - is_hard_limit BOOLEAN DEFAULT FALSE, -- If true, system enforces automatically - breach_action risk_action_type DEFAULT 'alert_only', - - -- Timing and validity - effective_from TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW(), - effective_to TIMESTAMP WITH TIME ZONE, - time_zone VARCHAR(50) DEFAULT 'UTC', - - -- Approval and audit - approved_by VARCHAR(64) NOT NULL, - approval_timestamp TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW(), - created_by VARCHAR(64) NOT NULL, - last_modified_by VARCHAR(64), - - -- Change tracking - created_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW(), - updated_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW(), - version INTEGER NOT NULL DEFAULT 1, - - -- Additional configuration - limit_details JSONB, -- Additional limit parameters - override_permissions TEXT[], -- Who can override this limit - - -- Constraints - CONSTRAINT uk_risk_limits_unique UNIQUE (limit_type, scope_level, COALESCE(account_id, ''), COALESCE(strategy_id, ''), COALESCE(symbol, '')), - CONSTRAINT chk_threshold_order CHECK ( - warning_threshold IS NULL OR breach_threshold IS NULL OR warning_threshold <= breach_threshold - ), - CONSTRAINT chk_scope_consistency CHECK ( - (scope_level = 'global' AND account_id IS NULL AND strategy_id IS NULL AND symbol IS NULL) OR - (scope_level = 'account' AND account_id IS NOT NULL AND strategy_id IS NULL AND symbol IS NULL) OR - (scope_level = 'strategy' AND account_id IS NOT NULL AND strategy_id IS NOT NULL AND symbol IS NULL) OR - (scope_level = 'symbol' AND symbol IS NOT NULL) - ) -); - --- ================================================================================================ --- STRESS TEST SCENARIOS TABLE --- Predefined stress test scenarios and results --- ================================================================================================ -CREATE TABLE stress_test_scenarios ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - scenario_name VARCHAR(200) NOT NULL UNIQUE, - scenario_type VARCHAR(100) NOT NULL, -- 'historical', 'hypothetical', 'monte_carlo' - - -- Scenario definition - description TEXT NOT NULL, - stress_parameters JSONB NOT NULL, -- Market movements, shocks, etc. - test_duration_days INTEGER DEFAULT 1, - - -- Execution details - is_active BOOLEAN DEFAULT TRUE, - frequency_hours INTEGER DEFAULT 24, -- How often to run this scenario - last_executed TIMESTAMP WITH TIME ZONE, - next_scheduled TIMESTAMP WITH TIME ZONE, - - -- Validation and approval - created_by VARCHAR(64) NOT NULL, - approved_by VARCHAR(64), - approval_date TIMESTAMP WITH TIME ZONE, - - -- Change tracking - created_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW(), - updated_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW(), - version INTEGER NOT NULL DEFAULT 1 -); - --- ================================================================================================ --- STRESS TEST RESULTS TABLE --- Results from stress test executions --- ================================================================================================ -CREATE TABLE stress_test_results ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - scenario_id UUID NOT NULL REFERENCES stress_test_scenarios(id), - execution_id UUID NOT NULL, -- Groups results from same execution - - -- Execution context - execution_timestamp ns_timestamp NOT NULL, - portfolio_snapshot_id UUID, -- Reference to portfolio state at test time - market_data_timestamp ns_timestamp, - - -- Scope of test - account_id VARCHAR(64), - strategy_id VARCHAR(100), - symbol VARCHAR(32), - - -- Results - base_value DECIMAL(20, 8) NOT NULL, -- Portfolio value before stress - stressed_value DECIMAL(20, 8) NOT NULL, -- Portfolio value after stress - pnl_impact DECIMAL(20, 8) NOT NULL, -- Profit/Loss impact - percentage_impact DECIMAL(8, 4) NOT NULL, -- Percentage change - - -- Risk metrics under stress - stressed_var DECIMAL(20, 8), - stressed_volatility DECIMAL(10, 6), - stressed_correlation DECIMAL(6, 4), - max_drawdown DECIMAL(8, 4), - - -- Test verdict - test_passed BOOLEAN NOT NULL, - failure_reason TEXT, - risk_score DECIMAL(10, 6), -- Overall risk score after stress - - -- Additional details - detailed_results JSONB, -- Breakdown by position, factor, etc. - calculation_time_ms INTEGER, -- How long the calculation took - - -- Partition key - execution_date DATE GENERATED ALWAYS AS (ns_to_date_immutable(execution_timestamp)) STORED -) PARTITION BY RANGE (execution_date); - --- ================================================================================================ --- RISK DASHBOARD MATERIALIZED VIEW --- Real-time risk monitoring dashboard --- ================================================================================================ -CREATE MATERIALIZED VIEW mv_risk_dashboard AS -SELECT - -- Scope identifiers - COALESCE(rm.account_id, 'ALL') as account_id, - COALESCE(rm.strategy_id, 'ALL') as strategy_id, - COALESCE(rm.symbol, 'ALL') as symbol, - - -- Current risk metrics - rm.metric_name, - rm.value as current_value, - rm.warning_threshold, - rm.breach_threshold, - rm.emergency_threshold, - - -- Risk status - CASE - WHEN rm.value > COALESCE(rm.emergency_threshold, rm.breach_threshold) THEN 'emergency' - WHEN rm.value > rm.breach_threshold THEN 'critical' - WHEN rm.value > COALESCE(rm.warning_threshold, rm.breach_threshold * 0.8) THEN 'warning' - ELSE 'normal' - END as risk_status, - - -- Utilization percentages - CASE - WHEN rm.breach_threshold > 0 THEN (rm.value / rm.breach_threshold * 100) - ELSE 0 - END as threshold_utilization_pct, - - -- Timing - rm.calculation_timestamp, - rm.last_updated, - - -- Recent events - (SELECT COUNT(*) - FROM risk_events re - WHERE re.event_timestamp >= EXTRACT(EPOCH FROM (NOW() - INTERVAL '1 hour')) * 1000000000 - AND re.severity IN ('high', 'critical', 'emergency') - AND (re.account_id = rm.account_id OR rm.account_id IS NULL) - AND (re.strategy_id = rm.strategy_id OR rm.strategy_id IS NULL) - AND (re.symbol = rm.symbol OR rm.symbol IS NULL) - ) as recent_high_severity_events, - - -- Model information - rm.model_name, - rm.model_version, - rm.is_valid - -FROM risk_metrics rm -WHERE rm.calculation_timestamp >= EXTRACT(EPOCH FROM (NOW() - INTERVAL '4 hours')) * 1000000000 - AND rm.is_valid = TRUE - --- Get the most recent metric for each combination -AND rm.calculation_timestamp = ( - SELECT MAX(rm2.calculation_timestamp) - FROM risk_metrics rm2 - WHERE rm2.metric_name = rm.metric_name - AND COALESCE(rm2.account_id, '') = COALESCE(rm.account_id, '') - AND COALESCE(rm2.strategy_id, '') = COALESCE(rm.strategy_id, '') - AND COALESCE(rm2.symbol, '') = COALESCE(rm.symbol, '') - AND rm2.calculation_timestamp >= EXTRACT(EPOCH FROM (NOW() - INTERVAL '4 hours')) * 1000000000 - AND rm2.is_valid = TRUE -); - --- ================================================================================================ --- HIGH-PERFORMANCE INDEXES --- ================================================================================================ - --- Risk events indexes -CREATE INDEX idx_risk_events_timestamp ON risk_events USING BTREE (event_timestamp); -CREATE INDEX idx_risk_events_severity_timestamp ON risk_events USING BTREE (severity, event_timestamp); -CREATE INDEX idx_risk_events_type_timestamp ON risk_events USING BTREE (event_type, event_timestamp); -CREATE INDEX idx_risk_events_account ON risk_events USING BTREE (account_id, event_timestamp) WHERE account_id IS NOT NULL; -CREATE INDEX idx_risk_events_symbol ON risk_events USING BTREE (symbol, event_timestamp) WHERE symbol IS NOT NULL; -CREATE INDEX idx_risk_events_unresolved ON risk_events USING BTREE (severity, event_timestamp) WHERE resolved_timestamp IS NULL; -CREATE INDEX idx_risk_events_correlation ON risk_events USING HASH (correlation_id); - --- GIN indexes for JSONB fields -CREATE INDEX idx_risk_events_data_gin ON risk_events USING GIN (event_data); -CREATE INDEX idx_risk_events_action_details_gin ON risk_events USING GIN (action_details); - --- Risk metrics indexes -CREATE INDEX idx_risk_metrics_timestamp ON risk_metrics USING BTREE (calculation_timestamp); -CREATE INDEX idx_risk_metrics_name_scope ON risk_metrics USING BTREE (metric_name, account_id, strategy_id, symbol); -CREATE INDEX idx_risk_metrics_account_timestamp ON risk_metrics USING BTREE (account_id, calculation_timestamp) WHERE account_id IS NOT NULL; -CREATE INDEX idx_risk_metrics_symbol_timestamp ON risk_metrics USING BTREE (symbol, calculation_timestamp) WHERE symbol IS NOT NULL; -CREATE INDEX idx_risk_metrics_valid ON risk_metrics USING BTREE (metric_name, calculation_timestamp) WHERE is_valid = TRUE; - --- Risk limits indexes -CREATE INDEX idx_risk_limits_scope ON risk_limits USING BTREE (limit_type, scope_level); -CREATE INDEX idx_risk_limits_account ON risk_limits USING BTREE (account_id) WHERE account_id IS NOT NULL; -CREATE INDEX idx_risk_limits_active ON risk_limits USING BTREE (limit_type, is_active) WHERE is_active = TRUE; -CREATE INDEX idx_risk_limits_effective ON risk_limits USING BTREE (effective_from, effective_to); - --- Stress test indexes -CREATE INDEX idx_stress_test_results_execution ON stress_test_results USING BTREE (execution_id, execution_timestamp); -CREATE INDEX idx_stress_test_results_scenario ON stress_test_results USING BTREE (scenario_id, execution_timestamp); -CREATE INDEX idx_stress_test_results_account ON stress_test_results USING BTREE (account_id, execution_timestamp) WHERE account_id IS NOT NULL; - --- ================================================================================================ --- AUTOMATIC PARTITIONING --- ================================================================================================ - --- Function to create daily partitions for risk events -CREATE OR REPLACE FUNCTION create_risk_events_partition(target_date DATE) -RETURNS VOID AS $$ -DECLARE - partition_name TEXT; - start_date DATE; - end_date DATE; -BEGIN - start_date := target_date; - end_date := target_date + INTERVAL '1 day'; - partition_name := 'risk_events_' || to_char(start_date, 'YYYY_MM_DD'); - - IF NOT EXISTS ( - SELECT 1 FROM information_schema.tables - WHERE table_name = partition_name - ) THEN - EXECUTE format('CREATE TABLE %I PARTITION OF risk_events - FOR VALUES FROM (%L) TO (%L)', - partition_name, start_date, end_date); - - -- Add partition-specific indexes - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (event_timestamp)', - 'idx_' || partition_name || '_timestamp', partition_name); - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (severity, event_timestamp)', - 'idx_' || partition_name || '_severity_ts', partition_name); - END IF; -END; -$$ LANGUAGE plpgsql; - --- Function to create monthly partitions for risk metrics -CREATE OR REPLACE FUNCTION create_risk_metrics_partition(target_date DATE) -RETURNS VOID AS $$ -DECLARE - partition_name TEXT; - start_date DATE; - end_date DATE; -BEGIN - start_date := date_trunc('month', target_date); - end_date := start_date + INTERVAL '1 month'; - partition_name := 'risk_metrics_' || to_char(start_date, 'YYYY_MM'); - - IF NOT EXISTS ( - SELECT 1 FROM information_schema.tables - WHERE table_name = partition_name - ) THEN - EXECUTE format('CREATE TABLE %I PARTITION OF risk_metrics - FOR VALUES FROM (%L) TO (%L)', - partition_name, start_date, end_date); - - -- Add partition-specific indexes - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (calculation_timestamp)', - 'idx_' || partition_name || '_timestamp', partition_name); - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (metric_name, calculation_timestamp)', - 'idx_' || partition_name || '_name_ts', partition_name); - END IF; -END; -$$ LANGUAGE plpgsql; - --- Create initial partitions -DO $$ -DECLARE - i INTEGER; -BEGIN - -- Create risk_events partitions for current and next 7 days - FOR i IN 0..7 LOOP - PERFORM create_risk_events_partition(CURRENT_DATE + i); - END LOOP; - - -- Create risk_metrics partitions for current and next 2 months - FOR i IN 0..2 LOOP - PERFORM create_risk_metrics_partition(CURRENT_DATE + (i || ' months')::INTERVAL); - END LOOP; - - -- Create stress_test_results partitions - FOR i IN 0..7 LOOP - PERFORM create_trading_events_partition(CURRENT_DATE + i); -- Reuse function with same logic - END LOOP; -END $$; - --- ================================================================================================ --- TRIGGER FUNCTIONS FOR AUTOMATION --- ================================================================================================ - --- Function to automatically update risk limits timestamp -CREATE OR REPLACE FUNCTION update_risk_limits_timestamp() -RETURNS TRIGGER AS $$ -BEGIN - NEW.updated_at := NOW(); - NEW.version := OLD.version + 1; - RETURN NEW; -END; -$$ LANGUAGE plpgsql; - --- Function to validate risk limit hierarchy -CREATE OR REPLACE FUNCTION validate_risk_limit_hierarchy() -RETURNS TRIGGER AS $$ -DECLARE - parent_limit DECIMAL(20, 8); -BEGIN - -- Check that child limits don't exceed parent limits - IF NEW.scope_level = 'account' THEN - SELECT breach_threshold INTO parent_limit - FROM risk_limits - WHERE limit_type = NEW.limit_type - AND scope_level = 'global' - AND is_active = TRUE; - - IF parent_limit IS NOT NULL AND NEW.breach_threshold > parent_limit THEN - RAISE EXCEPTION 'Account limit cannot exceed global limit for %', NEW.limit_type; - END IF; - END IF; - - RETURN NEW; -END; -$$ LANGUAGE plpgsql; - --- Function to generate risk events from metric breaches -CREATE OR REPLACE FUNCTION check_risk_metric_breach() -RETURNS TRIGGER AS $$ -DECLARE - applicable_limit RECORD; - breach_detected BOOLEAN := FALSE; - severity_level risk_severity; - event_type_val risk_event_type; -BEGIN - -- Find applicable risk limit (most specific first) - SELECT * INTO applicable_limit - FROM risk_limits rl - WHERE rl.limit_type = NEW.metric_name - AND rl.is_active = TRUE - AND NOW() BETWEEN rl.effective_from AND COALESCE(rl.effective_to, 'infinity'::TIMESTAMP WITH TIME ZONE) - AND ( - (rl.scope_level = 'symbol' AND rl.symbol = NEW.symbol) OR - (rl.scope_level = 'strategy' AND rl.strategy_id = NEW.strategy_id) OR - (rl.scope_level = 'account' AND rl.account_id = NEW.account_id) OR - (rl.scope_level = 'global') - ) - ORDER BY - CASE rl.scope_level - WHEN 'symbol' THEN 1 - WHEN 'strategy' THEN 2 - WHEN 'account' THEN 3 - WHEN 'global' THEN 4 - END - LIMIT 1; - - -- Check for breaches - IF applicable_limit.id IS NOT NULL THEN - IF NEW.value > COALESCE(applicable_limit.emergency_threshold, applicable_limit.breach_threshold) THEN - breach_detected := TRUE; - severity_level := 'emergency'; - event_type_val := CASE NEW.metric_name - WHEN 'var_1d', 'var_10d' THEN 'var_breach' - WHEN 'exposure_gross', 'exposure_net' THEN 'exposure_limit_breach' - WHEN 'leverage_ratio' THEN 'leverage_excess' - WHEN 'concentration_single', 'concentration_sector' THEN 'concentration_risk' - ELSE 'stress_test_failure' - END; - ELSIF NEW.value > applicable_limit.breach_threshold THEN - breach_detected := TRUE; - severity_level := 'critical'; - event_type_val := CASE NEW.metric_name - WHEN 'var_1d', 'var_10d' THEN 'var_breach' - WHEN 'exposure_gross', 'exposure_net' THEN 'exposure_limit_breach' - WHEN 'leverage_ratio' THEN 'leverage_excess' - WHEN 'concentration_single', 'concentration_sector' THEN 'concentration_risk' - ELSE 'stress_test_failure' - END; - ELSIF NEW.value > COALESCE(applicable_limit.warning_threshold, applicable_limit.breach_threshold * 0.8) THEN - breach_detected := TRUE; - severity_level := 'medium'; - event_type_val := CASE NEW.metric_name - WHEN 'var_1d', 'var_10d' THEN 'var_breach' - WHEN 'exposure_gross', 'exposure_net' THEN 'exposure_limit_breach' - WHEN 'leverage_ratio' THEN 'leverage_excess' - WHEN 'concentration_single', 'concentration_sector' THEN 'concentration_risk' - ELSE 'stress_test_failure' - END; - END IF; - - -- Generate risk event if breach detected - IF breach_detected THEN - INSERT INTO risk_events ( - correlation_id, event_timestamp, detected_timestamp, - event_type, severity, risk_metric, - symbol, account_id, strategy_id, - threshold_value, actual_value, breach_percentage, - description, risk_model, model_version, - recommended_action, source_system, node_id, process_id, - event_data - ) VALUES ( - NEW.id, - NEW.calculation_timestamp, - EXTRACT(EPOCH FROM NOW()) * 1000000000, - event_type_val, - severity_level, - NEW.metric_name, - NEW.symbol, - NEW.account_id, - NEW.strategy_id, - applicable_limit.breach_threshold, - NEW.value, - ((NEW.value - applicable_limit.breach_threshold) / applicable_limit.breach_threshold * 100), - format('Risk metric %s breached: %s > %s', NEW.metric_name, NEW.value, applicable_limit.breach_threshold), - NEW.model_name, - NEW.model_version, - applicable_limit.breach_action, - 'risk_engine', - 'risk-node-01', - pg_backend_pid(), - jsonb_build_object( - 'metric_id', NEW.id, - 'limit_id', applicable_limit.id, - 'calculation_details', NEW.calculation_details, - 'confidence_level', NEW.confidence_level - ) - ); - END IF; - END IF; - - RETURN NEW; -END; -$$ LANGUAGE plpgsql; - --- ================================================================================================ --- CREATE TRIGGERS --- ================================================================================================ - --- Risk limits triggers -CREATE TRIGGER tg_update_risk_limits_timestamp - BEFORE UPDATE ON risk_limits - FOR EACH ROW - EXECUTE FUNCTION update_risk_limits_timestamp(); - -CREATE TRIGGER tg_validate_risk_limit_hierarchy - BEFORE INSERT OR UPDATE ON risk_limits - FOR EACH ROW - EXECUTE FUNCTION validate_risk_limit_hierarchy(); - --- Risk metrics breach detection -CREATE TRIGGER tg_check_risk_metric_breach - AFTER INSERT OR UPDATE ON risk_metrics - FOR EACH ROW - WHEN (NEW.is_valid = TRUE) - EXECUTE FUNCTION check_risk_metric_breach(); - --- Update stress test scenario timestamp -CREATE TRIGGER tg_update_stress_scenarios_timestamp - BEFORE UPDATE ON stress_test_scenarios - FOR EACH ROW - EXECUTE FUNCTION update_risk_limits_timestamp(); -- Reuse same function - --- ================================================================================================ --- RISK MANAGEMENT FUNCTIONS --- ================================================================================================ - --- Function to calculate portfolio VaR -CREATE OR REPLACE FUNCTION calculate_portfolio_var( - p_account_id VARCHAR(64) DEFAULT NULL, - p_strategy_id VARCHAR(100) DEFAULT NULL, - p_confidence_level DECIMAL(5,4) DEFAULT 0.95, - p_time_horizon_days INTEGER DEFAULT 1 -) RETURNS DECIMAL(20,8) AS $$ -DECLARE - portfolio_var DECIMAL(20,8) := 0; - position_count INTEGER; -BEGIN - -- Simple VaR calculation based on current positions - -- In production, this would use more sophisticated models - - SELECT COUNT(*) INTO position_count - FROM positions p - WHERE (p_account_id IS NULL OR p.account_id = p_account_id) - AND (p_strategy_id IS NULL OR p.strategy_id = p_strategy_id) - AND p.quantity != 0; - - IF position_count = 0 THEN - RETURN 0; - END IF; - - -- Placeholder calculation - implement actual VaR model - SELECT COALESCE(SUM(ABS(p.market_value) * 0.02), 0) -- 2% daily volatility assumption - INTO portfolio_var - FROM positions p - WHERE (p_account_id IS NULL OR p.account_id = p_account_id) - AND (p_strategy_id IS NULL OR p.strategy_id = p_strategy_id) - AND p.quantity != 0; - - -- Adjust for confidence level and time horizon - portfolio_var := portfolio_var * SQRT(p_time_horizon_days) * - (CASE - WHEN p_confidence_level >= 0.99 THEN 2.33 - WHEN p_confidence_level >= 0.95 THEN 1.65 - ELSE 1.28 - END); - - RETURN portfolio_var; -END; -$$ LANGUAGE plpgsql; - --- Function to refresh risk dashboard -CREATE OR REPLACE FUNCTION refresh_risk_dashboard() -RETURNS VOID AS $$ -BEGIN - REFRESH MATERIALIZED VIEW CONCURRENTLY mv_risk_dashboard; -END; -$$ LANGUAGE plpgsql; - --- Function to get active risk alerts -CREATE OR REPLACE FUNCTION get_active_risk_alerts( - p_severity risk_severity[] DEFAULT ARRAY['high', 'critical', 'emergency'] -) RETURNS TABLE ( - event_id UUID, - event_type risk_event_type, - severity risk_severity, - symbol VARCHAR(32), - account_id VARCHAR(64), - description TEXT, - event_timestamp ns_timestamp, - age_minutes INTEGER -) AS $$ -BEGIN - RETURN QUERY - SELECT - re.id, - re.event_type, - re.severity, - re.symbol, - re.account_id, - re.description, - re.event_timestamp, - EXTRACT(EPOCH FROM (NOW() - TO_TIMESTAMP(re.event_timestamp / 1000000000.0))) / 60 AS age_minutes - FROM risk_events re - WHERE re.resolved_timestamp IS NULL - AND re.severity = ANY(p_severity) - AND re.event_timestamp >= EXTRACT(EPOCH FROM (NOW() - INTERVAL '24 hours')) * 1000000000 - ORDER BY re.severity DESC, re.event_timestamp DESC; -END; -$$ LANGUAGE plpgsql; - --- ================================================================================================ --- REPORTING VIEWS --- ================================================================================================ - --- Active risk alerts view -CREATE VIEW v_active_risk_alerts AS -SELECT - re.id, - re.event_type, - re.severity, - re.symbol, - re.account_id, - re.strategy_id, - re.description, - re.actual_value, - re.threshold_value, - re.breach_percentage, - TO_TIMESTAMP(re.event_timestamp / 1000000000.0) as event_time, - TO_TIMESTAMP(re.detected_timestamp / 1000000000.0) as detected_time, - EXTRACT(EPOCH FROM (NOW() - TO_TIMESTAMP(re.event_timestamp / 1000000000.0))) / 60 as age_minutes, - re.recommended_action, - re.acknowledged_by IS NOT NULL as is_acknowledged -FROM risk_events re -WHERE re.resolved_timestamp IS NULL - AND re.severity IN ('medium', 'high', 'critical', 'emergency') - AND re.event_timestamp >= EXTRACT(EPOCH FROM (NOW() - INTERVAL '7 days')) * 1000000000 -ORDER BY - CASE re.severity - WHEN 'emergency' THEN 1 - WHEN 'critical' THEN 2 - WHEN 'high' THEN 3 - WHEN 'medium' THEN 4 - ELSE 5 - END, - re.event_timestamp DESC; - --- Risk metrics summary view -CREATE VIEW v_risk_metrics_summary AS -SELECT - rm.metric_name, - rm.account_id, - rm.strategy_id, - rm.symbol, - rm.value as current_value, - rl.warning_threshold, - rl.breach_threshold, - rl.emergency_threshold, - CASE - WHEN rm.value > COALESCE(rl.emergency_threshold, rl.breach_threshold) THEN 'EMERGENCY' - WHEN rm.value > rl.breach_threshold THEN 'CRITICAL' - WHEN rm.value > COALESCE(rl.warning_threshold, rl.breach_threshold * 0.8) THEN 'WARNING' - ELSE 'NORMAL' - END as status, - TO_TIMESTAMP(rm.calculation_timestamp / 1000000000.0) as calculated_at, - rm.model_name, - rm.is_valid -FROM risk_metrics rm -LEFT JOIN risk_limits rl ON ( - rl.limit_type = rm.metric_name - AND rl.is_active = TRUE - AND NOW() BETWEEN rl.effective_from AND COALESCE(rl.effective_to, 'infinity'::TIMESTAMP WITH TIME ZONE) - AND ( - (rl.scope_level = 'symbol' AND rl.symbol = rm.symbol) OR - (rl.scope_level = 'strategy' AND rl.strategy_id = rm.strategy_id) OR - (rl.scope_level = 'account' AND rl.account_id = rm.account_id) OR - (rl.scope_level = 'global' AND rl.account_id IS NULL AND rl.strategy_id IS NULL AND rl.symbol IS NULL) - ) -) -WHERE rm.calculation_timestamp >= EXTRACT(EPOCH FROM (NOW() - INTERVAL '4 hours')) * 1000000000 - AND rm.is_valid = TRUE - -- Get most recent calculation for each metric/scope combination - AND rm.calculation_timestamp = ( - SELECT MAX(rm2.calculation_timestamp) - FROM risk_metrics rm2 - WHERE rm2.metric_name = rm.metric_name - AND COALESCE(rm2.account_id, '') = COALESCE(rm.account_id, '') - AND COALESCE(rm2.strategy_id, '') = COALESCE(rm.strategy_id, '') - AND COALESCE(rm2.symbol, '') = COALESCE(rm.symbol, '') - AND rm2.is_valid = TRUE - ); - --- ================================================================================================ --- COMMENTS AND DOCUMENTATION --- ================================================================================================ - -COMMENT ON TABLE risk_events IS 'Immutable event store for all risk management events including breaches, alerts, and stress test results. Critical for compliance and risk monitoring.'; -COMMENT ON TABLE risk_metrics IS 'Historical and current risk metric calculations with confidence intervals. Partitioned by date for performance.'; -COMMENT ON TABLE risk_limits IS 'Configurable risk limits with hierarchical scope (global > account > strategy > symbol). Supports time-based limits and automatic enforcement.'; -COMMENT ON TABLE stress_test_scenarios IS 'Predefined stress test scenarios including historical events, hypothetical shocks, and Monte Carlo simulations.'; -COMMENT ON TABLE stress_test_results IS 'Results from stress test executions showing portfolio impact under various scenarios. Critical for regulatory reporting.'; - -COMMENT ON MATERIALIZED VIEW mv_risk_dashboard IS 'Real-time risk monitoring dashboard with current metrics, thresholds, and alert counts. Refresh every 5 minutes in production.'; - -COMMENT ON FUNCTION calculate_portfolio_var IS 'Calculate portfolio Value at Risk using specified confidence level and time horizon. Implement with actual risk models in production.'; -COMMENT ON FUNCTION get_active_risk_alerts IS 'Get currently active risk alerts filtered by severity. Used by monitoring systems and dashboards.'; \ No newline at end of file diff --git a/migrations/003_audit_system.sql.broken b/migrations/003_audit_system.sql.broken deleted file mode 100644 index 5a7b4e957..000000000 --- a/migrations/003_audit_system.sql.broken +++ /dev/null @@ -1,1006 +0,0 @@ --- ================================================================================================ --- Migration 003: Comprehensive Audit System Schema --- Complete audit trail for regulatory compliance with immutable logging --- Includes ML events, system events, and complete change tracking --- ================================================================================================ - --- ================================================================================================ --- AUDIT EVENT TYPES AND ENUMS --- Comprehensive classification for all auditable events --- ================================================================================================ - -CREATE TYPE audit_event_type AS ENUM ( - -- Trading audit events - 'order_created', - 'order_modified', - 'order_cancelled', - 'order_executed', - 'trade_settled', - 'position_updated', - - -- Risk management audit events - 'risk_limit_breached', - 'risk_limit_updated', - 'emergency_action_taken', - 'compliance_check_failed', - 'model_validation_failed', - - -- ML and Algorithm audit events - 'model_prediction', - 'model_training_started', - 'model_training_completed', - 'model_deployed', - 'model_rollback', - 'signal_generated', - 'algorithm_decision', - 'backtest_executed', - - -- System audit events - 'system_startup', - 'system_shutdown', - 'service_restart', - 'configuration_changed', - 'user_login', - 'user_logout', - 'permission_granted', - 'permission_revoked', - 'data_export', - 'data_import', - - -- Security audit events - 'authentication_success', - 'authentication_failure', - 'authorization_failure', - 'suspicious_activity', - 'security_breach_detected', - 'encryption_key_rotated', - - -- Market data audit events - 'market_data_received', - 'market_data_gap_detected', - 'circuit_breaker_triggered', - 'trading_halt_detected', - - -- Compliance audit events - 'regulatory_report_generated', - 'audit_trail_accessed', - 'data_retention_policy_applied', - 'compliance_validation_completed' -); - -CREATE TYPE audit_severity AS ENUM ( - 'trace', -- Detailed debugging information - 'debug', -- General debugging information - 'info', -- General information - 'notice', -- Normal but significant condition - 'warning', -- Warning conditions - 'error', -- Error conditions - 'critical', -- Critical conditions requiring immediate attention - 'alert', -- Action must be taken immediately - 'emergency' -- System is unusable -); - -CREATE TYPE system_component AS ENUM ( - 'trading_engine', - 'risk_management', - 'market_data', - 'order_management', - 'portfolio_management', - 'ml_engine', - 'execution_engine', - 'compliance_engine', - 'authentication', - 'configuration', - 'database', - 'api_gateway', - 'user_interface', - 'reporting', - 'monitoring', - 'backup_system' -); - --- ================================================================================================ --- COMPREHENSIVE AUDIT LOG TABLE --- Immutable audit trail for all system activities --- ================================================================================================ -CREATE TABLE audit_log ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - audit_id BIGSERIAL NOT NULL, -- Sequential audit ID for ordering - correlation_id UUID, -- Links related audit entries - - -- Timing with nanosecond precision - event_timestamp ns_timestamp NOT NULL, -- When the event actually occurred - recorded_timestamp ns_timestamp NOT NULL, -- When the audit entry was created - processing_timestamp ns_timestamp NOT NULL, -- When the system processed the event - - -- Event classification - event_type audit_event_type NOT NULL, - severity audit_severity NOT NULL DEFAULT 'info', - component system_component NOT NULL, - - -- Event context and scope - entity_type VARCHAR(100), -- Type of entity affected (order, position, user, etc.) - entity_id UUID, -- ID of the affected entity - parent_entity_type VARCHAR(100), -- Parent entity type - parent_entity_id UUID, -- Parent entity ID - - -- User and session context - user_id VARCHAR(64), - session_id UUID, - impersonated_user_id VARCHAR(64), -- If acting on behalf of another user - client_id VARCHAR(100), -- Application or API client - - -- Request context - request_id UUID, -- Original request that triggered this event - trace_id UUID, -- Distributed tracing ID - span_id UUID, -- Distributed tracing span ID - - -- Network and system context - source_ip INET, - user_agent TEXT, - source_host VARCHAR(255), - source_port INTEGER, - - -- Event details - action VARCHAR(100) NOT NULL, -- Specific action taken - resource VARCHAR(200), -- Resource or endpoint accessed - method VARCHAR(20), -- HTTP method or operation type - - -- Data changes (for compliance) - old_values JSONB, -- Previous state before change - new_values JSONB, -- New state after change - affected_fields TEXT[], -- List of fields that were modified - - -- Event payload and metadata - event_data JSONB NOT NULL, -- Complete event details - tags JSONB, -- Searchable tags and labels - metadata JSONB, -- Additional metadata - - -- System and performance context - node_id VARCHAR(50) NOT NULL, -- Which system node recorded this - process_id INTEGER NOT NULL, -- OS process ID - thread_id INTEGER, -- Thread ID - execution_time_ns BIGINT, -- How long the operation took (nanoseconds) - memory_usage_bytes BIGINT, -- Memory usage at time of event - cpu_usage_percent DECIMAL(5,2), -- CPU usage percentage - - -- Security and integrity - checksum VARCHAR(64) NOT NULL, -- SHA-256 hash of event content - digital_signature TEXT, -- Digital signature for non-repudiation - encryption_key_id VARCHAR(100), -- If event data is encrypted - - -- Compliance and retention - retention_category VARCHAR(50) DEFAULT 'standard', -- Retention policy category - retention_until DATE, -- When this record can be archived/deleted - is_sensitive BOOLEAN DEFAULT FALSE, -- Contains sensitive data - compliance_flags TEXT[], -- Regulatory compliance flags - - -- Error handling - is_error BOOLEAN DEFAULT FALSE, - error_code VARCHAR(50), - error_message TEXT, - stack_trace TEXT, - - -- Partition key for performance - audit_date DATE GENERATED ALWAYS AS (DATE(TO_TIMESTAMP(event_timestamp / 1000000000.0))) STORED, - - -- Constraints - CONSTRAINT chk_audit_timestamps CHECK ( - recorded_timestamp >= event_timestamp AND - processing_timestamp >= recorded_timestamp - ), - CONSTRAINT chk_execution_time CHECK (execution_time_ns IS NULL OR execution_time_ns >= 0), - CONSTRAINT chk_cpu_usage CHECK (cpu_usage_percent IS NULL OR (cpu_usage_percent >= 0 AND cpu_usage_percent <= 100)) -) PARTITION BY RANGE (audit_date); - --- ================================================================================================ --- ML EVENTS TABLE --- Specialized tracking for machine learning operations --- ================================================================================================ -CREATE TABLE ml_events ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - model_id VARCHAR(200) NOT NULL, -- Unique model identifier - model_version VARCHAR(50) NOT NULL, - - -- Timing - event_timestamp ns_timestamp NOT NULL, - - -- ML event classification - event_type VARCHAR(100) NOT NULL, -- prediction, training, validation, deployment, etc. - event_subtype VARCHAR(100), -- More specific classification - - -- Model context - model_name VARCHAR(200) NOT NULL, - model_architecture VARCHAR(100), -- transformer, lstm, cnn, etc. - framework VARCHAR(50), -- tensorflow, pytorch, etc. - framework_version VARCHAR(50), - - -- Input/Output data - input_features JSONB, -- Model input features and values - predictions JSONB, -- Model predictions/outputs - confidence_scores JSONB, -- Confidence levels for predictions - feature_importance JSONB, -- Feature importance scores - - -- Performance metrics - inference_time_ns BIGINT, -- Time taken for inference - model_accuracy DECIMAL(8,6), -- Model accuracy if available - model_loss DECIMAL(12,8), -- Model loss/error - prediction_confidence DECIMAL(8,6), -- Overall prediction confidence - - -- Training context (for training events) - training_dataset_id VARCHAR(200), - training_dataset_size INTEGER, - training_epochs INTEGER, - learning_rate DECIMAL(10,8), - batch_size INTEGER, - - -- Validation and testing - validation_score DECIMAL(8,6), - test_score DECIMAL(8,6), - cross_validation_scores DECIMAL(8,6)[], - - -- Model drift and monitoring - data_drift_score DECIMAL(8,6), -- How much input data has drifted - model_drift_score DECIMAL(8,6), -- How much model performance has drifted - feature_drift_scores JSONB, -- Per-feature drift scores - anomaly_score DECIMAL(8,6), -- Anomaly detection score - - -- Business context - symbol VARCHAR(32), -- Trading symbol if applicable - strategy_id VARCHAR(100), - account_id VARCHAR(64), - signal_strength DECIMAL(8,6), -- Trading signal strength - - -- System context - node_id VARCHAR(50) NOT NULL, - gpu_device_id INTEGER, -- GPU device used - memory_usage_mb INTEGER, - gpu_memory_usage_mb INTEGER, - - -- Model artifacts and references - model_artifact_path TEXT, -- Path to model file - checkpoint_id VARCHAR(200), -- Training checkpoint reference - experiment_id VARCHAR(200), -- ML experiment tracking ID - - -- Additional metadata - hyperparameters JSONB, -- Model hyperparameters - environment_info JSONB, -- Runtime environment details - custom_metrics JSONB, -- Domain-specific metrics - - -- Partition key - event_date DATE GENERATED ALWAYS AS (DATE(TO_TIMESTAMP(event_timestamp / 1000000000.0))) STORED -) PARTITION BY RANGE (event_date); - --- ================================================================================================ --- SYSTEM EVENTS TABLE --- Specialized tracking for system health and performance --- ================================================================================================ -CREATE TABLE system_events ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - event_timestamp ns_timestamp NOT NULL, - - -- Event classification - event_type VARCHAR(100) NOT NULL, -- startup, shutdown, health_check, performance_alert, etc. - severity audit_severity NOT NULL, - component system_component NOT NULL, - service_name VARCHAR(100), - - -- System metrics - cpu_usage_percent DECIMAL(5,2), - memory_usage_mb INTEGER, - memory_total_mb INTEGER, - disk_usage_gb INTEGER, - disk_total_gb INTEGER, - network_rx_bytes BIGINT, - network_tx_bytes BIGINT, - load_average_1m DECIMAL(8,4), - load_average_5m DECIMAL(8,4), - load_average_15m DECIMAL(8,4), - - -- Performance metrics - latency_p50_ns BIGINT, -- 50th percentile latency - latency_p95_ns BIGINT, -- 95th percentile latency - latency_p99_ns BIGINT, -- 99th percentile latency - throughput_ops_per_sec DECIMAL(12,2), - error_rate_percent DECIMAL(5,2), - - -- Application-specific metrics - active_connections INTEGER, - pending_requests INTEGER, - orders_per_second DECIMAL(10,2), - fills_per_second DECIMAL(10,2), - market_data_messages_per_second DECIMAL(12,2), - - -- Health check details - health_status VARCHAR(50), -- healthy, degraded, unhealthy, unknown - health_checks JSONB, -- Individual health check results - dependencies_status JSONB, -- Status of external dependencies - - -- Configuration and version info - application_version VARCHAR(100), - configuration_version VARCHAR(100), - database_version VARCHAR(100), - - -- Error and debugging information - error_details JSONB, - debug_info JSONB, - - -- System context - node_id VARCHAR(50) NOT NULL, - process_id INTEGER NOT NULL, - container_id VARCHAR(100), -- Docker/Kubernetes container ID - pod_name VARCHAR(100), -- Kubernetes pod name - namespace VARCHAR(100), -- Kubernetes namespace - - -- Partition key - event_date DATE GENERATED ALWAYS AS (DATE(TO_TIMESTAMP(event_timestamp / 1000000000.0))) STORED -) PARTITION BY RANGE (event_date); - --- ================================================================================================ --- CHANGE TRACKING TABLE --- Detailed tracking of all data changes for compliance --- ================================================================================================ -CREATE TABLE change_tracking ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - change_timestamp ns_timestamp NOT NULL, - - -- Change context - table_name VARCHAR(100) NOT NULL, - operation VARCHAR(20) NOT NULL CHECK (operation IN ('INSERT', 'UPDATE', 'DELETE')), - primary_key_values JSONB NOT NULL, -- Primary key values of affected record - - -- Change details - changed_columns TEXT[], -- Names of columns that changed - old_row_data JSONB, -- Complete old row data (for UPDATE/DELETE) - new_row_data JSONB, -- Complete new row data (for INSERT/UPDATE) - column_changes JSONB, -- Detailed before/after for each changed column - - -- User and session context - user_id VARCHAR(64), - session_id UUID, - application_name VARCHAR(100), - - -- Transaction context - transaction_id BIGINT, -- Database transaction ID - statement_id INTEGER, -- Statement within transaction - - -- System context - node_id VARCHAR(50) NOT NULL, - process_id INTEGER NOT NULL, - - -- Audit metadata - audit_log_id UUID, -- Reference to audit_log entry - checksum VARCHAR(64) NOT NULL, -- Integrity check - - -- Partition key - change_date DATE GENERATED ALWAYS AS (DATE(TO_TIMESTAMP(change_timestamp / 1000000000.0))) STORED -) PARTITION BY RANGE (change_date); - --- ================================================================================================ --- COMPLIANCE ANNOTATIONS TABLE --- Additional compliance metadata for audit entries --- ================================================================================================ -CREATE TABLE compliance_annotations ( - -- Primary identifiers - id UUID PRIMARY KEY DEFAULT uuid_generate_v4(), - audit_log_id UUID NOT NULL REFERENCES audit_log(id), - - -- Compliance framework - regulation_name VARCHAR(100) NOT NULL, -- MiFID II, GDPR, SOX, etc. - requirement_section VARCHAR(100), -- Specific section/article - compliance_category VARCHAR(100), -- trade_reporting, record_keeping, etc. - - -- Annotation details - annotation_type VARCHAR(50) NOT NULL, -- tag, note, exemption, etc. - annotation_value TEXT, - is_required BOOLEAN DEFAULT TRUE, - - -- Validation and review - validated_by VARCHAR(64), - validated_at TIMESTAMP WITH TIME ZONE, - review_status VARCHAR(50), -- pending, approved, rejected - reviewer_notes TEXT, - - -- Retention and archival - retention_years INTEGER NOT NULL DEFAULT 7, - archive_after_years INTEGER DEFAULT 10, - - -- Timing - created_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW(), - updated_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW() -); - --- ================================================================================================ --- HIGH-PERFORMANCE INDEXES --- ================================================================================================ - --- Audit log indexes (optimized for compliance queries) -CREATE INDEX idx_audit_log_timestamp ON audit_log USING BTREE (event_timestamp); -CREATE INDEX idx_audit_log_user_timestamp ON audit_log USING BTREE (user_id, event_timestamp) WHERE user_id IS NOT NULL; -CREATE INDEX idx_audit_log_entity ON audit_log USING BTREE (entity_type, entity_id) WHERE entity_type IS NOT NULL AND entity_id IS NOT NULL; -CREATE INDEX idx_audit_log_component_type ON audit_log USING BTREE (component, event_type); -CREATE INDEX idx_audit_log_severity ON audit_log USING BTREE (severity, event_timestamp) WHERE severity IN ('error', 'critical', 'alert', 'emergency'); -CREATE INDEX idx_audit_log_session ON audit_log USING HASH (session_id) WHERE session_id IS NOT NULL; -CREATE INDEX idx_audit_log_correlation ON audit_log USING HASH (correlation_id) WHERE correlation_id IS NOT NULL; -CREATE INDEX idx_audit_log_trace ON audit_log USING HASH (trace_id) WHERE trace_id IS NOT NULL; -CREATE INDEX idx_audit_log_sensitive ON audit_log USING BTREE (event_timestamp) WHERE is_sensitive = TRUE; - --- GIN indexes for JSONB columns (flexible querying) -CREATE INDEX idx_audit_log_event_data_gin ON audit_log USING GIN (event_data); -CREATE INDEX idx_audit_log_old_values_gin ON audit_log USING GIN (old_values); -CREATE INDEX idx_audit_log_new_values_gin ON audit_log USING GIN (new_values); -CREATE INDEX idx_audit_log_tags_gin ON audit_log USING GIN (tags); - --- ML events indexes -CREATE INDEX idx_ml_events_timestamp ON ml_events USING BTREE (event_timestamp); -CREATE INDEX idx_ml_events_model ON ml_events USING BTREE (model_name, model_version, event_timestamp); -CREATE INDEX idx_ml_events_symbol ON ml_events USING BTREE (symbol, event_timestamp) WHERE symbol IS NOT NULL; -CREATE INDEX idx_ml_events_strategy ON ml_events USING BTREE (strategy_id, event_timestamp) WHERE strategy_id IS NOT NULL; -CREATE INDEX idx_ml_events_type ON ml_events USING BTREE (event_type, event_timestamp); - --- System events indexes -CREATE INDEX idx_system_events_timestamp ON system_events USING BTREE (event_timestamp); -CREATE INDEX idx_system_events_component ON system_events USING BTREE (component, event_timestamp); -CREATE INDEX idx_system_events_severity ON system_events USING BTREE (severity, event_timestamp); -CREATE INDEX idx_system_events_health ON system_events USING BTREE (health_status, event_timestamp) WHERE health_status IS NOT NULL; -CREATE INDEX idx_system_events_node ON system_events USING BTREE (node_id, event_timestamp); - --- Change tracking indexes -CREATE INDEX idx_change_tracking_timestamp ON change_tracking USING BTREE (change_timestamp); -CREATE INDEX idx_change_tracking_table ON change_tracking USING BTREE (table_name, change_timestamp); -CREATE INDEX idx_change_tracking_user ON change_tracking USING BTREE (user_id, change_timestamp) WHERE user_id IS NOT NULL; -CREATE INDEX idx_change_tracking_audit_log ON change_tracking USING HASH (audit_log_id) WHERE audit_log_id IS NOT NULL; - --- Compliance annotations indexes -CREATE INDEX idx_compliance_annotations_audit_log ON compliance_annotations USING HASH (audit_log_id); -CREATE INDEX idx_compliance_annotations_regulation ON compliance_annotations USING BTREE (regulation_name, compliance_category); -CREATE INDEX idx_compliance_annotations_review ON compliance_annotations USING BTREE (review_status, created_at); - --- ================================================================================================ --- AUTOMATIC PARTITIONING --- ================================================================================================ - --- Function to create daily partitions for audit_log -CREATE OR REPLACE FUNCTION create_audit_log_partition(target_date DATE) -RETURNS VOID AS $$ -DECLARE - partition_name TEXT; - start_date DATE; - end_date DATE; -BEGIN - start_date := target_date; - end_date := target_date + INTERVAL '1 day'; - partition_name := 'audit_log_' || to_char(start_date, 'YYYY_MM_DD'); - - IF NOT EXISTS ( - SELECT 1 FROM information_schema.tables - WHERE table_name = partition_name - ) THEN - EXECUTE format('CREATE TABLE %I PARTITION OF audit_log - FOR VALUES FROM (%L) TO (%L)', - partition_name, start_date, end_date); - - -- Add partition-specific indexes - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (event_timestamp)', - 'idx_' || partition_name || '_timestamp', partition_name); - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (user_id, event_timestamp) WHERE user_id IS NOT NULL', - 'idx_' || partition_name || '_user_ts', partition_name); - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (component, event_type)', - 'idx_' || partition_name || '_comp_type', partition_name); - END IF; -END; -$$ LANGUAGE plpgsql; - --- Function to create daily partitions for ML events -CREATE OR REPLACE FUNCTION create_ml_events_partition(target_date DATE) -RETURNS VOID AS $$ -DECLARE - partition_name TEXT; - start_date DATE; - end_date DATE; -BEGIN - start_date := target_date; - end_date := target_date + INTERVAL '1 day'; - partition_name := 'ml_events_' || to_char(start_date, 'YYYY_MM_DD'); - - IF NOT EXISTS ( - SELECT 1 FROM information_schema.tables - WHERE table_name = partition_name - ) THEN - EXECUTE format('CREATE TABLE %I PARTITION OF ml_events - FOR VALUES FROM (%L) TO (%L)', - partition_name, start_date, end_date); - - -- Add partition-specific indexes - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (event_timestamp)', - 'idx_' || partition_name || '_timestamp', partition_name); - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (model_name, event_timestamp)', - 'idx_' || partition_name || '_model_ts', partition_name); - END IF; -END; -$$ LANGUAGE plpgsql; - --- Function to create daily partitions for system events -CREATE OR REPLACE FUNCTION create_system_events_partition(target_date DATE) -RETURNS VOID AS $$ -DECLARE - partition_name TEXT; - start_date DATE; - end_date DATE; -BEGIN - start_date := target_date; - end_date := target_date + INTERVAL '1 day'; - partition_name := 'system_events_' || to_char(start_date, 'YYYY_MM_DD'); - - IF NOT EXISTS ( - SELECT 1 FROM information_schema.tables - WHERE table_name = partition_name - ) THEN - EXECUTE format('CREATE TABLE %I PARTITION OF system_events - FOR VALUES FROM (%L) TO (%L)', - partition_name, start_date, end_date); - - -- Add partition-specific indexes - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (event_timestamp)', - 'idx_' || partition_name || '_timestamp', partition_name); - EXECUTE format('CREATE INDEX %I ON %I USING BTREE (component, event_timestamp)', - 'idx_' || partition_name || '_comp_ts', partition_name); - END IF; -END; -$$ LANGUAGE plpgsql; - --- Create initial partitions for all audit tables -DO $$ -DECLARE - i INTEGER; -BEGIN - -- Create partitions for current and next 30 days - FOR i IN 0..30 LOOP - PERFORM create_audit_log_partition(CURRENT_DATE + i); - PERFORM create_ml_events_partition(CURRENT_DATE + i); - PERFORM create_system_events_partition(CURRENT_DATE + i); - -- Reuse trading_events partition function for change_tracking - PERFORM create_trading_events_partition(CURRENT_DATE + i); - END LOOP; -END $$; - --- ================================================================================================ --- AUDIT HELPER FUNCTIONS --- ================================================================================================ - --- Function to create audit log entry -CREATE OR REPLACE FUNCTION create_audit_entry( - p_event_type audit_event_type, - p_component system_component, - p_action VARCHAR(100), - p_entity_type VARCHAR(100) DEFAULT NULL, - p_entity_id UUID DEFAULT NULL, - p_user_id VARCHAR(64) DEFAULT NULL, - p_session_id UUID DEFAULT NULL, - p_event_data JSONB DEFAULT '{}'::jsonb, - p_severity audit_severity DEFAULT 'info', - p_old_values JSONB DEFAULT NULL, - p_new_values JSONB DEFAULT NULL -) RETURNS UUID AS $$ -DECLARE - audit_entry_id UUID; - current_timestamp_ns ns_timestamp; -BEGIN - audit_entry_id := uuid_generate_v4(); - current_timestamp_ns := EXTRACT(EPOCH FROM NOW()) * 1000000000; - - INSERT INTO audit_log ( - id, event_timestamp, recorded_timestamp, processing_timestamp, - event_type, severity, component, entity_type, entity_id, - user_id, session_id, action, event_data, - old_values, new_values, node_id, process_id, checksum - ) VALUES ( - audit_entry_id, - current_timestamp_ns, - current_timestamp_ns, - current_timestamp_ns, - p_event_type, - p_severity, - p_component, - p_entity_type, - p_entity_id, - p_user_id, - p_session_id, - p_action, - p_event_data, - p_old_values, - p_new_values, - 'audit-node-01', -- TODO: Get from environment - pg_backend_pid(), - encode(sha256(audit_entry_id::text::bytea), 'hex') - ); - - RETURN audit_entry_id; -END; -$$ LANGUAGE plpgsql; - --- Function to log ML event -CREATE OR REPLACE FUNCTION log_ml_event( - p_model_name VARCHAR(200), - p_model_version VARCHAR(50), - p_event_type VARCHAR(100), - p_predictions JSONB DEFAULT NULL, - p_confidence_scores JSONB DEFAULT NULL, - p_symbol VARCHAR(32) DEFAULT NULL, - p_strategy_id VARCHAR(100) DEFAULT NULL, - p_inference_time_ns BIGINT DEFAULT NULL, - p_additional_data JSONB DEFAULT '{}'::jsonb -) RETURNS UUID AS $$ -DECLARE - ml_event_id UUID; - current_timestamp_ns ns_timestamp; -BEGIN - ml_event_id := uuid_generate_v4(); - current_timestamp_ns := EXTRACT(EPOCH FROM NOW()) * 1000000000; - - INSERT INTO ml_events ( - id, model_id, model_version, event_timestamp, - event_type, model_name, predictions, confidence_scores, - symbol, strategy_id, inference_time_ns, node_id, - custom_metrics - ) VALUES ( - ml_event_id, - p_model_name || ':' || p_model_version, - p_model_version, - current_timestamp_ns, - p_event_type, - p_model_name, - p_predictions, - p_confidence_scores, - p_symbol, - p_strategy_id, - p_inference_time_ns, - 'ml-node-01', -- TODO: Get from environment - p_additional_data - ); - - RETURN ml_event_id; -END; -$$ LANGUAGE plpgsql; - --- Function to log system event -CREATE OR REPLACE FUNCTION log_system_event( - p_event_type VARCHAR(100), - p_component system_component, - p_severity audit_severity DEFAULT 'info', - p_service_name VARCHAR(100) DEFAULT NULL, - p_health_status VARCHAR(50) DEFAULT NULL, - p_metrics JSONB DEFAULT '{}'::jsonb, - p_error_details JSONB DEFAULT NULL -) RETURNS UUID AS $$ -DECLARE - system_event_id UUID; - current_timestamp_ns ns_timestamp; -BEGIN - system_event_id := uuid_generate_v4(); - current_timestamp_ns := EXTRACT(EPOCH FROM NOW()) * 1000000000; - - INSERT INTO system_events ( - id, event_timestamp, event_type, severity, component, - service_name, health_status, node_id, process_id, - error_details, debug_info - ) VALUES ( - system_event_id, - current_timestamp_ns, - p_event_type, - p_severity, - p_component, - p_service_name, - p_health_status, - 'system-node-01', -- TODO: Get from environment - pg_backend_pid(), - p_error_details, - p_metrics - ); - - RETURN system_event_id; -END; -$$ LANGUAGE plpgsql; - --- ================================================================================================ --- AUTOMATED CHANGE TRACKING TRIGGERS --- ================================================================================================ - --- Generic function to track changes on any table -CREATE OR REPLACE FUNCTION track_table_changes() -RETURNS TRIGGER AS $$ -DECLARE - change_record_id UUID; - current_timestamp_ns ns_timestamp; - old_data JSONB; - new_data JSONB; - changed_cols TEXT[]; -BEGIN - change_record_id := uuid_generate_v4(); - current_timestamp_ns := EXTRACT(EPOCH FROM NOW()) * 1000000000; - - -- Convert row data to JSONB - IF TG_OP = 'DELETE' THEN - old_data := to_jsonb(OLD); - new_data := NULL; - ELSIF TG_OP = 'INSERT' THEN - old_data := NULL; - new_data := to_jsonb(NEW); - ELSE -- UPDATE - old_data := to_jsonb(OLD); - new_data := to_jsonb(NEW); - - -- Identify changed columns - SELECT array_agg(key) INTO changed_cols - FROM ( - SELECT key - FROM jsonb_each_text(old_data) o - FULL OUTER JOIN jsonb_each_text(new_data) n USING (key) - WHERE o.value IS DISTINCT FROM n.value - ) t; - END IF; - - -- Insert change tracking record - INSERT INTO change_tracking ( - id, change_timestamp, table_name, operation, - primary_key_values, changed_columns, old_row_data, new_row_data, - node_id, process_id, checksum - ) VALUES ( - change_record_id, - current_timestamp_ns, - TG_TABLE_NAME, - TG_OP, - CASE - WHEN TG_OP = 'DELETE' THEN jsonb_build_object('id', OLD.id) - ELSE jsonb_build_object('id', NEW.id) - END, - changed_cols, - old_data, - new_data, - 'change-tracker-01', -- TODO: Get from environment - pg_backend_pid(), - encode(sha256(change_record_id::text::bytea), 'hex') - ); - - RETURN COALESCE(NEW, OLD); -END; -$$ LANGUAGE plpgsql; - --- Function to automatically update compliance annotations timestamp -CREATE OR REPLACE FUNCTION update_compliance_annotations_timestamp() -RETURNS TRIGGER AS $$ -BEGIN - NEW.updated_at := NOW(); - RETURN NEW; -END; -$$ LANGUAGE plpgsql; - --- ================================================================================================ --- CREATE TRIGGERS FOR CHANGE TRACKING --- ================================================================================================ - --- Enable change tracking on core trading tables -CREATE TRIGGER tg_track_orders_changes - AFTER INSERT OR UPDATE OR DELETE ON orders - FOR EACH ROW - EXECUTE FUNCTION track_table_changes(); - -CREATE TRIGGER tg_track_fills_changes - AFTER INSERT OR UPDATE OR DELETE ON fills - FOR EACH ROW - EXECUTE FUNCTION track_table_changes(); - -CREATE TRIGGER tg_track_positions_changes - AFTER INSERT OR UPDATE OR DELETE ON positions - FOR EACH ROW - EXECUTE FUNCTION track_table_changes(); - -CREATE TRIGGER tg_track_risk_limits_changes - AFTER INSERT OR UPDATE OR DELETE ON risk_limits - FOR EACH ROW - EXECUTE FUNCTION track_table_changes(); - --- Compliance annotations timestamp trigger -CREATE TRIGGER tg_update_compliance_annotations_timestamp - BEFORE UPDATE ON compliance_annotations - FOR EACH ROW - EXECUTE FUNCTION update_compliance_annotations_timestamp(); - --- ================================================================================================ --- AUDIT SEARCH AND REPORTING FUNCTIONS --- ================================================================================================ - --- Function to search audit log with flexible filters -CREATE OR REPLACE FUNCTION search_audit_log( - p_start_time TIMESTAMP WITH TIME ZONE DEFAULT NOW() - INTERVAL '24 hours', - p_end_time TIMESTAMP WITH TIME ZONE DEFAULT NOW(), - p_user_id VARCHAR(64) DEFAULT NULL, - p_component system_component DEFAULT NULL, - p_event_type audit_event_type DEFAULT NULL, - p_severity audit_severity DEFAULT NULL, - p_entity_type VARCHAR(100) DEFAULT NULL, - p_entity_id UUID DEFAULT NULL, - p_search_text TEXT DEFAULT NULL, - p_limit INTEGER DEFAULT 1000 -) RETURNS TABLE ( - id UUID, - event_timestamp TIMESTAMP WITH TIME ZONE, - event_type audit_event_type, - severity audit_severity, - component system_component, - user_id VARCHAR(64), - action VARCHAR(100), - entity_type VARCHAR(100), - entity_id UUID, - description TEXT -) AS $$ -BEGIN - RETURN QUERY - SELECT - al.id, - TO_TIMESTAMP(al.event_timestamp / 1000000000.0), - al.event_type, - al.severity, - al.component, - al.user_id, - al.action, - al.entity_type, - al.entity_id, - COALESCE(al.event_data->>'description', al.action) as description - FROM audit_log al - WHERE al.event_timestamp >= EXTRACT(EPOCH FROM p_start_time) * 1000000000 - AND al.event_timestamp <= EXTRACT(EPOCH FROM p_end_time) * 1000000000 - AND (p_user_id IS NULL OR al.user_id = p_user_id) - AND (p_component IS NULL OR al.component = p_component) - AND (p_event_type IS NULL OR al.event_type = p_event_type) - AND (p_severity IS NULL OR al.severity = p_severity) - AND (p_entity_type IS NULL OR al.entity_type = p_entity_type) - AND (p_entity_id IS NULL OR al.entity_id = p_entity_id) - AND (p_search_text IS NULL OR - al.event_data::text ILIKE '%' || p_search_text || '%' OR - al.action ILIKE '%' || p_search_text || '%') - ORDER BY al.event_timestamp DESC - LIMIT p_limit; -END; -$$ LANGUAGE plpgsql; - --- Function to get audit trail for specific entity -CREATE OR REPLACE FUNCTION get_entity_audit_trail( - p_entity_type VARCHAR(100), - p_entity_id UUID, - p_start_time TIMESTAMP WITH TIME ZONE DEFAULT NOW() - INTERVAL '30 days' -) RETURNS TABLE ( - event_timestamp TIMESTAMP WITH TIME ZONE, - event_type audit_event_type, - action VARCHAR(100), - user_id VARCHAR(64), - old_values JSONB, - new_values JSONB, - description TEXT -) AS $$ -BEGIN - RETURN QUERY - SELECT - TO_TIMESTAMP(al.event_timestamp / 1000000000.0), - al.event_type, - al.action, - al.user_id, - al.old_values, - al.new_values, - COALESCE(al.event_data->>'description', al.action) as description - FROM audit_log al - WHERE al.entity_type = p_entity_type - AND al.entity_id = p_entity_id - AND al.event_timestamp >= EXTRACT(EPOCH FROM p_start_time) * 1000000000 - ORDER BY al.event_timestamp ASC; -END; -$$ LANGUAGE plpgsql; - --- ================================================================================================ --- REPORTING VIEWS --- ================================================================================================ - --- Daily audit summary view -CREATE VIEW v_daily_audit_summary AS -SELECT - DATE(TO_TIMESTAMP(al.event_timestamp / 1000000000.0)) as audit_date, - al.component, - al.event_type, - al.severity, - COUNT(*) as event_count, - COUNT(DISTINCT al.user_id) as unique_users, - COUNT(*) FILTER (WHERE al.is_error = TRUE) as error_count, - COUNT(*) FILTER (WHERE al.severity IN ('critical', 'alert', 'emergency')) as critical_count -FROM audit_log al -WHERE al.event_timestamp >= EXTRACT(EPOCH FROM (NOW() - INTERVAL '30 days')) * 1000000000 -GROUP BY - DATE(TO_TIMESTAMP(al.event_timestamp / 1000000000.0)), - al.component, - al.event_type, - al.severity -ORDER BY audit_date DESC, event_count DESC; - --- User activity summary view -CREATE VIEW v_user_activity_summary AS -SELECT - al.user_id, - DATE(TO_TIMESTAMP(al.event_timestamp / 1000000000.0)) as activity_date, - COUNT(*) as total_actions, - COUNT(DISTINCT al.component) as components_accessed, - COUNT(*) FILTER (WHERE al.severity = 'error') as error_count, - MIN(TO_TIMESTAMP(al.event_timestamp / 1000000000.0)) as first_activity, - MAX(TO_TIMESTAMP(al.event_timestamp / 1000000000.0)) as last_activity, - array_agg(DISTINCT al.event_type ORDER BY al.event_type) as event_types -FROM audit_log al -WHERE al.user_id IS NOT NULL - AND al.event_timestamp >= EXTRACT(EPOCH FROM (NOW() - INTERVAL '30 days')) * 1000000000 -GROUP BY al.user_id, DATE(TO_TIMESTAMP(al.event_timestamp / 1000000000.0)) -ORDER BY activity_date DESC, total_actions DESC; - --- System health summary view -CREATE VIEW v_system_health_summary AS -SELECT - se.component, - se.health_status, - COUNT(*) as status_count, - AVG(se.cpu_usage_percent) as avg_cpu_usage, - AVG(se.memory_usage_mb) as avg_memory_usage, - AVG(se.latency_p95_ns) / 1000000.0 as avg_p95_latency_ms, - MAX(TO_TIMESTAMP(se.event_timestamp / 1000000000.0)) as last_reported -FROM system_events se -WHERE se.event_timestamp >= EXTRACT(EPOCH FROM (NOW() - INTERVAL '4 hours')) * 1000000000 - AND se.health_status IS NOT NULL -GROUP BY se.component, se.health_status -ORDER BY se.component, status_count DESC; - --- ================================================================================================ --- RETENTION AND ARCHIVAL POLICIES --- ================================================================================================ - --- Function to archive old audit data -CREATE OR REPLACE FUNCTION archive_audit_data(retention_years INTEGER DEFAULT 7) -RETURNS INTEGER AS $$ -DECLARE - archive_date DATE; - archived_count INTEGER := 0; - partition_name TEXT; -BEGIN - archive_date := CURRENT_DATE - INTERVAL '1 year' * retention_years; - - -- Archive old partitions (implementation depends on archival strategy) - -- This is a placeholder for actual archival implementation - - -- Drop partitions older than retention period - FOR partition_name IN - SELECT table_name - FROM information_schema.tables - WHERE table_name LIKE 'audit_log_%' - AND table_name < 'audit_log_' || to_char(archive_date, 'YYYY_MM_DD') - LOOP - -- Move to archive or drop (implement based on requirements) - EXECUTE format('DROP TABLE %I', partition_name); - archived_count := archived_count + 1; - END LOOP; - - RETURN archived_count; -END; -$$ LANGUAGE plpgsql; - --- ================================================================================================ --- COMMENTS AND DOCUMENTATION --- ================================================================================================ - -COMMENT ON TABLE audit_log IS 'Comprehensive immutable audit trail for all system activities. Partitioned by date with 7+ year retention for regulatory compliance.'; -COMMENT ON TABLE ml_events IS 'Specialized audit log for machine learning operations including model predictions, training, and deployment events.'; -COMMENT ON TABLE system_events IS 'System health and performance event tracking with metrics and health status monitoring.'; -COMMENT ON TABLE change_tracking IS 'Detailed change tracking for all data modifications with before/after values for compliance reporting.'; -COMMENT ON TABLE compliance_annotations IS 'Additional compliance metadata and annotations for audit entries to support regulatory requirements.'; - -COMMENT ON FUNCTION create_audit_entry IS 'Helper function to create standardized audit log entries with proper formatting and security.'; -COMMENT ON FUNCTION log_ml_event IS 'Helper function to log machine learning events with standardized schema and performance metrics.'; -COMMENT ON FUNCTION search_audit_log IS 'Flexible audit log search function supporting various filters for compliance reporting and investigation.'; -COMMENT ON FUNCTION get_entity_audit_trail IS 'Get complete audit trail for a specific entity showing all changes and activities over time.'; \ No newline at end of file diff --git a/migrations/023_*.sql.skip b/migrations/023_*.sql.skip deleted file mode 100644 index 7313afe5c..000000000 --- a/migrations/023_*.sql.skip +++ /dev/null @@ -1,376 +0,0 @@ --- ================================================================================================ --- Migration 023: Ensemble ML Performance Tuning --- High-frequency write optimization for 1000+ predictions/sec --- ================================================================================================ --- Target: >1000 inserts/sec, <100ms P99 query latency, >5x compression ratio --- ================================================================================================ - --- ================================================================================================ --- PART 1: ENHANCED INDEXING STRATEGY --- ================================================================================================ - --- Drop redundant indexes from migration 022 that are covered by hypertable time-space indexes -DROP INDEX IF EXISTS idx_ensemble_predictions_symbol_timestamp; -DROP INDEX IF EXISTS idx_model_performance_symbol_timestamp; - --- Composite index for high-frequency writes (model_id + prediction_timestamp for fast lookups) -CREATE INDEX IF NOT EXISTS idx_model_performance_composite -ON model_performance_attribution (model_id, symbol, window_hours, prediction_timestamp DESC); - --- Index for ensemble prediction lookups by symbol (most common query pattern) -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_symbol_action_timestamp -ON ensemble_predictions (symbol, ensemble_action, prediction_timestamp DESC); - --- Index for model checkpoint tracking (frequent lookup by checkpoint_id) -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_checkpoints -ON ensemble_predictions (dqn_checkpoint_id, ppo_checkpoint_id, mamba2_checkpoint_id, tft_checkpoint_id); - --- Index for real-time performance monitoring (last 24 hours only) -CREATE INDEX IF NOT EXISTS idx_model_performance_realtime -ON model_performance_attribution (model_id, prediction_timestamp DESC); - --- Covering index for P&L attribution queries (includes all needed columns) -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_pnl_covering -ON ensemble_predictions (symbol, prediction_timestamp DESC) -INCLUDE (ensemble_action, ensemble_signal, pnl, order_id); - --- Index for inference latency monitoring (P99 latency tracking) -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_latency -ON ensemble_predictions (inference_latency_us DESC); - --- ================================================================================================ --- PART 2: TIMESCALEDB COMPRESSION OPTIMIZATION --- ================================================================================================ - --- Drop old compression policies -SELECT remove_compression_policy('ensemble_predictions', if_exists => true); -SELECT remove_compression_policy('model_performance_attribution', if_exists => true); - --- Enhanced compression for ensemble_predictions (compress after 7 days, target >5x ratio) -ALTER TABLE ensemble_predictions SET ( - timescaledb.compress = true, - timescaledb.compress_segmentby = 'symbol, ensemble_action', - timescaledb.compress_orderby = 'prediction_timestamp DESC, id', - timescaledb.compress_chunk_time_interval = '1 day' -); - --- Aggressive compression policy (7-day retention for hot data) -SELECT add_compression_policy('ensemble_predictions', - compress_after => INTERVAL '7 days', - if_not_exists => true -); - --- Enhanced compression for model_performance_attribution (compress after 14 days) -ALTER TABLE model_performance_attribution SET ( - timescaledb.compress = true, - timescaledb.compress_segmentby = 'model_id, symbol, window_hours', - timescaledb.compress_orderby = 'prediction_timestamp DESC, id', - timescaledb.compress_chunk_time_interval = '1 day' -); - -SELECT add_compression_policy('model_performance_attribution', - compress_after => INTERVAL '14 days', - if_not_exists => true -); - --- ================================================================================================ --- PART 3: CONTINUOUS AGGREGATES FOR REAL-TIME DASHBOARDS --- ================================================================================================ - --- Drop old continuous aggregates if they exist (to recreate with better config) -DROP MATERIALIZED VIEW IF EXISTS ensemble_performance_5min CASCADE; -DROP MATERIALIZED VIEW IF EXISTS model_performance_hourly CASCADE; - --- 5-minute ensemble performance (for real-time dashboards) -CREATE MATERIALIZED VIEW ensemble_performance_5min -WITH (timescaledb.continuous) AS -SELECT - time_bucket('5 minutes', prediction_timestamp) AS bucket, - symbol, - ensemble_action, - COUNT(*) AS prediction_count, - AVG(ensemble_confidence) AS avg_confidence, - STDDEV(ensemble_confidence) AS stddev_confidence, - AVG(disagreement_rate) AS avg_disagreement, - MAX(disagreement_rate) AS max_disagreement, - AVG(inference_latency_us) AS avg_latency_us, - PERCENTILE_CONT(0.99) WITHIN GROUP (ORDER BY inference_latency_us) AS p99_latency_us, - SUM(CASE WHEN pnl IS NOT NULL THEN pnl ELSE 0 END) AS total_pnl, - COUNT(CASE WHEN pnl > 0 THEN 1 END) AS winning_trades, - COUNT(CASE WHEN pnl < 0 THEN 1 END) AS losing_trades, - COUNT(CASE WHEN pnl IS NOT NULL THEN 1 END) AS total_trades, - AVG(CASE WHEN pnl IS NOT NULL THEN pnl END) AS avg_trade_pnl -FROM ensemble_predictions -GROUP BY bucket, symbol, ensemble_action; - --- Refresh every 5 minutes (near real-time) -SELECT add_continuous_aggregate_policy('ensemble_performance_5min', - start_offset => INTERVAL '30 minutes', - end_offset => INTERVAL '5 minutes', - schedule_interval => INTERVAL '5 minutes', - if_not_exists => true -); - --- Hourly model performance comparison (detailed attribution) -CREATE MATERIALIZED VIEW model_performance_hourly -WITH (timescaledb.continuous) AS -SELECT - time_bucket('1 hour', prediction_timestamp) AS bucket, - model_id, - symbol, - window_hours, - SUM(total_predictions) AS total_predictions, - SUM(correct_predictions) AS correct_predictions, - AVG(accuracy) AS avg_accuracy, - STDDEV(accuracy) AS stddev_accuracy, - SUM(total_pnl) AS total_pnl, - AVG(sharpe_ratio) AS avg_sharpe_ratio, - MAX(sharpe_ratio) AS max_sharpe_ratio, - AVG(sortino_ratio) AS avg_sortino_ratio, - AVG(max_drawdown) AS avg_max_drawdown, - AVG(win_rate) AS avg_win_rate, - AVG(avg_weight) AS avg_weight, - AVG(avg_confidence) AS avg_confidence, - AVG(disagreement_rate) AS avg_disagreement_rate, - SUM(disagreement_count) AS total_disagreements -FROM model_performance_attribution -GROUP BY bucket, model_id, symbol, window_hours; - --- Refresh every hour -SELECT add_continuous_aggregate_policy('model_performance_hourly', - start_offset => INTERVAL '6 hours', - end_offset => INTERVAL '1 hour', - schedule_interval => INTERVAL '1 hour', - if_not_exists => true -); - --- Weekly ensemble summary (for long-term trend analysis) -CREATE MATERIALIZED VIEW ensemble_performance_weekly -WITH (timescaledb.continuous) AS -SELECT - time_bucket('1 week', prediction_timestamp) AS bucket, - symbol, - COUNT(*) AS total_predictions, - AVG(ensemble_confidence) AS avg_confidence, - AVG(disagreement_rate) AS avg_disagreement, - SUM(CASE WHEN pnl IS NOT NULL THEN pnl ELSE 0 END) AS total_pnl, - COUNT(CASE WHEN pnl > 0 THEN 1 END) AS winning_trades, - COUNT(CASE WHEN pnl IS NOT NULL THEN 1 END) AS total_trades, - -- Sharpe ratio approximation - AVG(CASE WHEN pnl IS NOT NULL THEN pnl END) / - NULLIF(STDDEV(CASE WHEN pnl IS NOT NULL THEN pnl END), 0) AS sharpe_approx, - -- Max drawdown approximation - MIN(pnl) AS worst_trade, - MAX(pnl) AS best_trade -FROM ensemble_predictions -GROUP BY bucket, symbol; - --- Refresh daily -SELECT add_continuous_aggregate_policy('ensemble_performance_weekly', - start_offset => INTERVAL '1 month', - end_offset => INTERVAL '1 day', - schedule_interval => INTERVAL '1 day', - if_not_exists => true -); - --- ================================================================================================ --- PART 4: STATISTICS COLLECTION TUNING --- ================================================================================================ - --- Increase statistics target for critical columns (better query planning) -ALTER TABLE ensemble_predictions ALTER COLUMN prediction_timestamp SET STATISTICS 1000; -ALTER TABLE ensemble_predictions ALTER COLUMN symbol SET STATISTICS 500; -ALTER TABLE ensemble_predictions ALTER COLUMN ensemble_action SET STATISTICS 200; -ALTER TABLE ensemble_predictions ALTER COLUMN disagreement_rate SET STATISTICS 200; - -ALTER TABLE model_performance_attribution ALTER COLUMN model_id SET STATISTICS 500; -ALTER TABLE model_performance_attribution ALTER COLUMN prediction_timestamp SET STATISTICS 1000; -ALTER TABLE model_performance_attribution ALTER COLUMN symbol SET STATISTICS 500; -ALTER TABLE model_performance_attribution ALTER COLUMN sharpe_ratio SET STATISTICS 500; - --- Force statistics update -ANALYZE ensemble_predictions; -ANALYZE model_performance_attribution; - --- ================================================================================================ --- PART 5: RETENTION POLICIES (DATA LIFECYCLE MANAGEMENT) --- ================================================================================================ - --- Drop old retention policies if they exist -SELECT remove_retention_policy('ensemble_predictions', if_exists => true); -SELECT remove_retention_policy('model_performance_attribution', if_exists => true); - --- Retain ensemble predictions for 90 days (compressed after 7 days, deleted after 90) -SELECT add_retention_policy('ensemble_predictions', - drop_after => INTERVAL '90 days', - if_not_exists => true -); - --- Retain model performance for 180 days (6 months for long-term analysis) -SELECT add_retention_policy('model_performance_attribution', - drop_after => INTERVAL '180 days', - if_not_exists => true -); - --- ================================================================================================ --- PART 6: WRITE OPTIMIZATION FUNCTIONS --- ================================================================================================ - --- Bulk insert function for high-frequency predictions (batched writes) -CREATE OR REPLACE FUNCTION insert_ensemble_predictions_bulk( - p_predictions JSONB -- Array of prediction objects -) -RETURNS INTEGER AS $$ -DECLARE - v_count INTEGER; -BEGIN - -- Insert all predictions from JSONB array - INSERT INTO ensemble_predictions ( - prediction_timestamp, symbol, account_id, strategy_id, - ensemble_action, ensemble_signal, ensemble_confidence, disagreement_rate, - dqn_signal, dqn_confidence, dqn_weight, dqn_vote, - ppo_signal, ppo_confidence, ppo_weight, ppo_vote, - mamba2_signal, mamba2_confidence, mamba2_weight, mamba2_vote, - tft_signal, tft_confidence, tft_weight, tft_vote, - inference_latency_us, aggregation_latency_us, - feature_snapshot, metadata - ) - SELECT - (pred->>'prediction_timestamp')::TIMESTAMPTZ, - pred->>'symbol', - pred->>'account_id', - pred->>'strategy_id', - pred->>'ensemble_action', - (pred->>'ensemble_signal')::DOUBLE PRECISION, - (pred->>'ensemble_confidence')::DOUBLE PRECISION, - (pred->>'disagreement_rate')::DOUBLE PRECISION, - (pred->>'dqn_signal')::DOUBLE PRECISION, - (pred->>'dqn_confidence')::DOUBLE PRECISION, - (pred->>'dqn_weight')::DOUBLE PRECISION, - pred->>'dqn_vote', - (pred->>'ppo_signal')::DOUBLE PRECISION, - (pred->>'ppo_confidence')::DOUBLE PRECISION, - (pred->>'ppo_weight')::DOUBLE PRECISION, - pred->>'ppo_vote', - (pred->>'mamba2_signal')::DOUBLE PRECISION, - (pred->>'mamba2_confidence')::DOUBLE PRECISION, - (pred->>'mamba2_weight')::DOUBLE PRECISION, - pred->>'mamba2_vote', - (pred->>'tft_signal')::DOUBLE PRECISION, - (pred->>'tft_confidence')::DOUBLE PRECISION, - (pred->>'tft_weight')::DOUBLE PRECISION, - pred->>'tft_vote', - (pred->>'inference_latency_us')::INTEGER, - (pred->>'aggregation_latency_us')::INTEGER, - (pred->'feature_snapshot')::JSONB, - (pred->'metadata')::JSONB - FROM jsonb_array_elements(p_predictions) AS pred; - - GET DIAGNOSTICS v_count = ROW_COUNT; - RETURN v_count; -END; -$$ LANGUAGE plpgsql; - -COMMENT ON FUNCTION insert_ensemble_predictions_bulk IS 'Bulk insert function for high-frequency predictions (batched writes, >1000/sec)'; - --- Bulk update function for P&L attribution (post-trade execution) -CREATE OR REPLACE FUNCTION update_ensemble_pnl_bulk( - p_updates JSONB -- Array of {id, order_id, executed_price, position_size, pnl, commission, slippage_bps} -) -RETURNS INTEGER AS $$ -DECLARE - v_count INTEGER := 0; - v_update JSONB; -BEGIN - -- Update each prediction with execution data - FOR v_update IN SELECT jsonb_array_elements(p_updates) - LOOP - UPDATE ensemble_predictions - SET - order_id = (v_update->>'order_id')::UUID, - executed_price = (v_update->>'executed_price')::BIGINT, - position_size = (v_update->>'position_size')::BIGINT, - pnl = (v_update->>'pnl')::BIGINT, - commission = COALESCE((v_update->>'commission')::BIGINT, 0), - slippage_bps = (v_update->>'slippage_bps')::INTEGER - WHERE id = (v_update->>'id')::UUID; - - v_count := v_count + 1; - END LOOP; - - RETURN v_count; -END; -$$ LANGUAGE plpgsql; - -COMMENT ON FUNCTION update_ensemble_pnl_bulk IS 'Bulk update P&L for executed predictions (post-trade attribution)'; - --- ================================================================================================ --- PART 7: MONITORING VIEWS --- ================================================================================================ - --- Real-time write throughput view (last 5 minutes) -CREATE OR REPLACE VIEW ensemble_write_throughput_5min AS -SELECT - time_bucket('1 minute', prediction_timestamp) AS minute, - COUNT(*) AS inserts_per_minute, - COUNT(*) / 60.0 AS inserts_per_second, - AVG(inference_latency_us) AS avg_inference_us, - PERCENTILE_CONT(0.99) WITHIN GROUP (ORDER BY inference_latency_us) AS p99_inference_us -FROM ensemble_predictions -WHERE prediction_timestamp >= NOW() - INTERVAL '5 minutes' -GROUP BY minute -ORDER BY minute DESC; - -COMMENT ON VIEW ensemble_write_throughput_5min IS 'Real-time write throughput monitoring (last 5 minutes)'; - --- Compression efficiency view -CREATE OR REPLACE VIEW ensemble_compression_stats AS -SELECT - hypertable_name, - chunk_name, - before_compression_total_bytes / 1024.0 / 1024.0 AS uncompressed_mb, - after_compression_total_bytes / 1024.0 / 1024.0 AS compressed_mb, - before_compression_total_bytes::FLOAT / NULLIF(after_compression_total_bytes, 0) AS compression_ratio, - number_compressed_rows, - pg_size_pretty(before_compression_total_bytes) AS uncompressed_size, - pg_size_pretty(after_compression_total_bytes) AS compressed_size -FROM timescaledb_information.compressed_chunk_stats -WHERE hypertable_name IN ('ensemble_predictions', 'model_performance_attribution') -ORDER BY before_compression_total_bytes DESC; - -COMMENT ON VIEW ensemble_compression_stats IS 'TimescaleDB compression efficiency metrics'; - --- Query performance view (pg_stat_statements required) -CREATE OR REPLACE VIEW ensemble_query_performance AS -SELECT - LEFT(query, 100) AS query_preview, - calls, - total_exec_time / 1000.0 AS total_time_sec, - mean_exec_time AS avg_time_ms, - max_exec_time AS max_time_ms, - stddev_exec_time AS stddev_time_ms, - rows / NULLIF(calls, 0) AS avg_rows_per_call -FROM pg_stat_statements -WHERE query LIKE '%ensemble_predictions%' OR query LIKE '%model_performance_attribution%' -ORDER BY mean_exec_time DESC -LIMIT 20; - -COMMENT ON VIEW ensemble_query_performance IS 'Top 20 slowest ensemble queries (requires pg_stat_statements)'; - --- ================================================================================================ --- PART 8: GRANT PERMISSIONS --- ================================================================================================ - -GRANT SELECT ON ensemble_performance_5min TO foxhunt; -GRANT SELECT ON model_performance_hourly TO foxhunt; -GRANT SELECT ON ensemble_performance_weekly TO foxhunt; -GRANT SELECT ON ensemble_write_throughput_5min TO foxhunt; -GRANT SELECT ON ensemble_compression_stats TO foxhunt; -GRANT SELECT ON ensemble_query_performance TO foxhunt; - -GRANT EXECUTE ON FUNCTION insert_ensemble_predictions_bulk TO foxhunt; -GRANT EXECUTE ON FUNCTION update_ensemble_pnl_bulk TO foxhunt; - --- ================================================================================================ --- END MIGRATION 023 --- ================================================================================================ diff --git a/migrations/024_*.sql.skip b/migrations/024_*.sql.skip deleted file mode 100644 index cbf058ab1..000000000 --- a/migrations/024_*.sql.skip +++ /dev/null @@ -1,70 +0,0 @@ --- Migration: ML Security Events Table --- Agent: Agent 122 --- Date: 2025-10-14 --- Purpose: Security event logging for ML inference system (SEC-001, SEC-002, SEC-003 fixes) - --- ML security events table for tracking security incidents -CREATE TABLE IF NOT EXISTS ml_security_events ( - id UUID PRIMARY KEY DEFAULT gen_random_uuid(), - timestamp TIMESTAMPTZ NOT NULL DEFAULT NOW(), - - -- Event classification - event_type VARCHAR(50) NOT NULL, -- signature_failure, outlier_detected, etc. - severity VARCHAR(20) NOT NULL, -- low, medium, high, critical - - -- Context - model_id VARCHAR(50), - checkpoint_id VARCHAR(255), - prediction_id UUID, -- References ensemble_predictions(id) if available - - -- Event details - description TEXT NOT NULL, - metadata JSONB, - - -- Response - action_taken VARCHAR(100), -- rejected, flagged, alerted, rollback - - CONSTRAINT chk_severity CHECK (severity IN ('low', 'medium', 'high', 'critical')), - CONSTRAINT chk_event_type CHECK (event_type IN ( - 'checkpoint_signature_failure', - 'checkpoint_signature_missing', - 'checkpoint_tampering_detected', - 'prediction_outlier_detected', - 'prediction_out_of_bounds', - 'extreme_rate_exceeded', - 'ensemble_sudden_shift', - 'coordinated_attack_suspected', - 'model_behavioral_drift', - 'automatic_rollback', - 'manual_intervention' - )) -); - --- Indexes for fast querying -CREATE INDEX idx_ml_security_events_timestamp ON ml_security_events (timestamp DESC); -CREATE INDEX idx_ml_security_events_severity ON ml_security_events (severity) - WHERE severity IN ('high', 'critical'); -CREATE INDEX idx_ml_security_events_type ON ml_security_events (event_type); -CREATE INDEX idx_ml_security_events_model ON ml_security_events (model_id) - WHERE model_id IS NOT NULL; - --- TimescaleDB hypertable for time-series data (if TimescaleDB is available) -DO $$ -BEGIN - -- Check if TimescaleDB extension exists - IF EXISTS (SELECT 1 FROM pg_extension WHERE extname = 'timescaledb') THEN - PERFORM create_hypertable('ml_security_events', 'timestamp', if_not_exists => TRUE); - - -- Retention policy: Keep high/critical events for 1 year, others for 90 days - -- Note: Actual retention requires setting up TimescaleDB retention policies - -- This can be done later via: - -- SELECT add_retention_policy('ml_security_events', INTERVAL '90 days'); - END IF; -END $$; - --- Add comment for documentation -COMMENT ON TABLE ml_security_events IS 'Security events for ML inference system - tracks checkpoint tampering, model poisoning, and ensemble anomalies'; -COMMENT ON COLUMN ml_security_events.event_type IS 'Type of security event (see CHECK constraint for valid values)'; -COMMENT ON COLUMN ml_security_events.severity IS 'Severity level: low, medium, high, critical'; -COMMENT ON COLUMN ml_security_events.metadata IS 'Additional event-specific metadata (JSON)'; -COMMENT ON COLUMN ml_security_events.action_taken IS 'Response action: rejected, flagged, alerted, rollback'; diff --git a/migrations/025_*.sql.skip b/migrations/025_*.sql.skip deleted file mode 100644 index f1a5d2208..000000000 --- a/migrations/025_*.sql.skip +++ /dev/null @@ -1,293 +0,0 @@ --- ================================================================================================ --- Migration 025: Query Performance Optimization --- Additional optimizations for paper trading validation queries --- ================================================================================================ --- Target: <5ms P99 query latency for aggregation queries, optimize TimescaleDB chunk exclusion --- ================================================================================================ - --- ================================================================================================ --- PART 1: ENHANCED COMPOSITE INDEXES FOR COMMON QUERY PATTERNS --- ================================================================================================ - --- Note: TimescaleDB hypertables do not support CONCURRENTLY, using regular CREATE INDEX --- Optimize symbol-filtered aggregation queries (from paper trading validation) --- Pattern: WHERE symbol = ? AND timestamp > ? -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_symbol_time -ON ensemble_predictions (symbol, timestamp DESC) -WHERE timestamp > NOW() - INTERVAL '30 days'; - --- Optimize real-time dashboard queries (last 24 hours) --- Pattern: WHERE timestamp > NOW() - INTERVAL '1 day' --- Note: This is a partial index covering only recent data for faster scans -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_recent_24h -ON ensemble_predictions (timestamp DESC) -INCLUDE (ensemble_confidence, disagreement_rate, ensemble_action, symbol) -WHERE timestamp > NOW() - INTERVAL '24 hours'; - --- Optimize model-specific queries (individual model performance) --- Pattern: WHERE dqn_signal IS NOT NULL -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_dqn_active -ON ensemble_predictions (timestamp DESC) -WHERE dqn_signal IS NOT NULL; - -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_ppo_active -ON ensemble_predictions (timestamp DESC) -WHERE ppo_signal IS NOT NULL; - -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_mamba2_active -ON ensemble_predictions (timestamp DESC) -WHERE mamba2_signal IS NOT NULL; - -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_tft_active -ON ensemble_predictions (timestamp DESC) -WHERE tft_signal IS NOT NULL; - --- Optimize order execution tracking (predictions that converted to orders) -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_executed -ON ensemble_predictions (timestamp DESC) -INCLUDE (order_id, executed_price, position_size, pnl) -WHERE order_id IS NOT NULL; - --- ================================================================================================ --- PART 2: MATERIALIZED VIEWS FOR SLOW AGGREGATION QUERIES --- ================================================================================================ - --- Real-time model activity summary (for debugging NULL model votes) --- Refreshes every minute to catch inactive models quickly -DROP MATERIALIZED VIEW IF EXISTS model_activity_realtime CASCADE; - -CREATE MATERIALIZED VIEW model_activity_realtime AS -SELECT - time_bucket('1 minute', timestamp) AS minute, - symbol, - COUNT(*) AS total_predictions, - COUNT(dqn_signal) AS dqn_active_count, - COUNT(ppo_signal) AS ppo_active_count, - COUNT(mamba2_signal) AS mamba2_active_count, - COUNT(tft_signal) AS tft_active_count, - ROUND(100.0 * COUNT(dqn_signal) / NULLIF(COUNT(*), 0), 2) AS dqn_active_pct, - ROUND(100.0 * COUNT(ppo_signal) / NULLIF(COUNT(*), 0), 2) AS ppo_active_pct, - ROUND(100.0 * COUNT(mamba2_signal) / NULLIF(COUNT(*), 0), 2) AS mamba2_active_pct, - ROUND(100.0 * COUNT(tft_signal) / NULLIF(COUNT(*), 0), 2) AS tft_active_pct, - AVG(ensemble_confidence) AS avg_confidence, - AVG(disagreement_rate) AS avg_disagreement -FROM ensemble_predictions -WHERE timestamp > NOW() - INTERVAL '1 hour' -GROUP BY minute, symbol -ORDER BY minute DESC; - -CREATE INDEX ON model_activity_realtime (minute DESC); -CREATE INDEX ON model_activity_realtime (symbol); - -COMMENT ON MATERIALIZED VIEW model_activity_realtime IS 'Real-time model activity tracking (last 1 hour, 1-minute buckets)'; - --- Paper trading execution summary (for monitoring order conversion rate) -DROP MATERIALIZED VIEW IF EXISTS paper_trading_execution_summary CASCADE; - -CREATE MATERIALIZED VIEW paper_trading_execution_summary AS -SELECT - time_bucket('5 minutes', timestamp) AS bucket, - symbol, - COUNT(*) AS total_predictions, - COUNT(order_id) AS executed_orders, - ROUND(100.0 * COUNT(order_id) / NULLIF(COUNT(*), 0), 2) AS execution_rate_pct, - COUNT(CASE WHEN pnl > 0 THEN 1 END) AS winning_trades, - COUNT(CASE WHEN pnl < 0 THEN 1 END) AS losing_trades, - COUNT(CASE WHEN pnl IS NOT NULL THEN 1 END) AS total_trades, - ROUND(100.0 * COUNT(CASE WHEN pnl > 0 THEN 1 END) / NULLIF(COUNT(CASE WHEN pnl IS NOT NULL THEN 1 END), 0), 2) AS win_rate_pct, - SUM(pnl) AS total_pnl, - AVG(pnl) AS avg_pnl, - STDDEV(pnl) AS stddev_pnl, - MIN(pnl) AS worst_trade, - MAX(pnl) AS best_trade -FROM ensemble_predictions -WHERE timestamp > NOW() - INTERVAL '24 hours' -GROUP BY bucket, symbol -ORDER BY bucket DESC; - -CREATE INDEX ON paper_trading_execution_summary (bucket DESC); -CREATE INDEX ON paper_trading_execution_summary (symbol); - -COMMENT ON MATERIALIZED VIEW paper_trading_execution_summary IS 'Paper trading execution rate and P&L summary (last 24 hours)'; - --- ================================================================================================ --- PART 3: OPTIMIZED QUERY FUNCTIONS (PRE-COMPUTED AGGREGATIONS) --- ================================================================================================ - --- Fast aggregation function for real-time dashboard queries --- Uses continuous aggregates instead of scanning raw table -CREATE OR REPLACE FUNCTION get_ensemble_performance_summary( - p_interval INTERVAL DEFAULT INTERVAL '1 day', - p_symbol VARCHAR(20) DEFAULT NULL -) -RETURNS TABLE ( - avg_confidence DOUBLE PRECISION, - avg_disagreement DOUBLE PRECISION, - total_predictions BIGINT, - total_trades BIGINT, - win_rate DOUBLE PRECISION, - total_pnl NUMERIC, - avg_latency_us NUMERIC, - p99_latency_us DOUBLE PRECISION -) AS $$ -BEGIN - RETURN QUERY - SELECT - AVG(ep5m.avg_confidence)::DOUBLE PRECISION, - AVG(ep5m.avg_disagreement)::DOUBLE PRECISION, - SUM(ep5m.prediction_count)::BIGINT, - SUM(ep5m.total_trades)::BIGINT, - (100.0 * SUM(ep5m.winning_trades) / NULLIF(SUM(ep5m.total_trades), 0))::DOUBLE PRECISION, - SUM(ep5m.total_pnl), - AVG(ep5m.avg_latency_us), - MAX(ep5m.p99_latency_us)::DOUBLE PRECISION - FROM ensemble_performance_5min ep5m - WHERE ep5m.bucket > NOW() - p_interval - AND (p_symbol IS NULL OR ep5m.symbol = p_symbol); -END; -$$ LANGUAGE plpgsql STABLE; - -COMMENT ON FUNCTION get_ensemble_performance_summary IS 'Fast aggregation using continuous aggregates (avoids raw table scan)'; - --- Model activity health check function --- Quickly identifies inactive models -CREATE OR REPLACE FUNCTION check_model_activity_health( - p_lookback_minutes INTEGER DEFAULT 60 -) -RETURNS TABLE ( - model_name VARCHAR(20), - is_active BOOLEAN, - last_prediction_time TIMESTAMPTZ, - minutes_since_last_prediction INTEGER, - predictions_in_window BIGINT, - activity_rate_pct DOUBLE PRECISION -) AS $$ -BEGIN - RETURN QUERY - WITH recent_predictions AS ( - SELECT - timestamp, - dqn_signal IS NOT NULL AS dqn_active, - ppo_signal IS NOT NULL AS ppo_active, - mamba2_signal IS NOT NULL AS mamba2_active, - tft_signal IS NOT NULL AS tft_active - FROM ensemble_predictions - WHERE timestamp > NOW() - INTERVAL '1 minute' * p_lookback_minutes - ), - model_stats AS ( - SELECT - 'DQN' AS model, - MAX(CASE WHEN dqn_active THEN timestamp END) AS last_pred, - COUNT(CASE WHEN dqn_active THEN 1 END) AS pred_count, - COUNT(*) AS total_count - FROM recent_predictions - UNION ALL - SELECT - 'PPO', - MAX(CASE WHEN ppo_active THEN timestamp END), - COUNT(CASE WHEN ppo_active THEN 1 END), - COUNT(*) - FROM recent_predictions - UNION ALL - SELECT - 'MAMBA-2', - MAX(CASE WHEN mamba2_active THEN timestamp END), - COUNT(CASE WHEN mamba2_active THEN 1 END), - COUNT(*) - FROM recent_predictions - UNION ALL - SELECT - 'TFT', - MAX(CASE WHEN tft_active THEN timestamp END), - COUNT(CASE WHEN tft_active THEN 1 END), - COUNT(*) - FROM recent_predictions - ) - SELECT - ms.model::VARCHAR(20), - (ms.pred_count > 0)::BOOLEAN, - ms.last_pred, - EXTRACT(EPOCH FROM (NOW() - COALESCE(ms.last_pred, NOW() - INTERVAL '1 year')))::INTEGER / 60, - ms.pred_count::BIGINT, - (100.0 * ms.pred_count / NULLIF(ms.total_count, 0))::DOUBLE PRECISION - FROM model_stats ms; -END; -$$ LANGUAGE plpgsql STABLE; - -COMMENT ON FUNCTION check_model_activity_health IS 'Quickly identifies inactive models (NULL signal issue)'; - --- ================================================================================================ --- PART 4: QUERY PERFORMANCE MONITORING --- ================================================================================================ - --- Create extension for query statistics if not exists -CREATE EXTENSION IF NOT EXISTS pg_stat_statements; - --- View for monitoring slow queries (updated from migration 023) -CREATE OR REPLACE VIEW ensemble_slow_queries AS -SELECT - LEFT(query, 150) AS query_preview, - calls, - ROUND(total_exec_time::NUMERIC / 1000.0, 2) AS total_time_sec, - ROUND(mean_exec_time::NUMERIC, 2) AS avg_time_ms, - ROUND(max_exec_time::NUMERIC, 2) AS max_time_ms, - ROUND(stddev_exec_time::NUMERIC, 2) AS stddev_time_ms, - rows / NULLIF(calls, 0) AS avg_rows_per_call, - ROUND(100.0 * shared_blks_hit / NULLIF(shared_blks_hit + shared_blks_read, 0), 2) AS cache_hit_ratio -FROM pg_stat_statements -WHERE query LIKE '%ensemble_predictions%' - OR query LIKE '%model_performance_attribution%' - OR query LIKE '%paper_trading_predictions%' -ORDER BY mean_exec_time DESC -LIMIT 30; - -COMMENT ON VIEW ensemble_slow_queries IS 'Top 30 slowest ensemble/paper trading queries with cache hit ratio'; - --- ================================================================================================ --- PART 5: VACUUM AND ANALYZE OPTIMIZATION --- ================================================================================================ - --- Optimize autovacuum settings for high-write tables -ALTER TABLE ensemble_predictions SET ( - autovacuum_vacuum_scale_factor = 0.05, -- Vacuum when 5% of rows change (default 20%) - autovacuum_analyze_scale_factor = 0.025, -- Analyze when 2.5% change (default 10%) - autovacuum_vacuum_cost_delay = 10 -- Speed up vacuum (default 20ms) -); - -ALTER TABLE model_performance_attribution SET ( - autovacuum_vacuum_scale_factor = 0.05, - autovacuum_analyze_scale_factor = 0.025, - autovacuum_vacuum_cost_delay = 10 -); - -ALTER TABLE paper_trading_predictions SET ( - autovacuum_vacuum_scale_factor = 0.05, - autovacuum_analyze_scale_factor = 0.025, - autovacuum_vacuum_cost_delay = 10 -); - --- Force immediate vacuum and analyze -VACUUM ANALYZE ensemble_predictions; -VACUUM ANALYZE model_performance_attribution; -VACUUM ANALYZE paper_trading_predictions; - --- ================================================================================================ --- PART 6: GRANT PERMISSIONS --- ================================================================================================ - -GRANT SELECT ON model_activity_realtime TO foxhunt; -GRANT SELECT ON paper_trading_execution_summary TO foxhunt; -GRANT SELECT ON ensemble_slow_queries TO foxhunt; -GRANT EXECUTE ON FUNCTION get_ensemble_performance_summary TO foxhunt; -GRANT EXECUTE ON FUNCTION check_model_activity_health TO foxhunt; - --- ================================================================================================ --- PART 7: REFRESH MATERIALIZED VIEWS --- ================================================================================================ - -REFRESH MATERIALIZED VIEW model_activity_realtime; -REFRESH MATERIALIZED VIEW paper_trading_execution_summary; - --- ================================================================================================ --- END MIGRATION 025 --- ================================================================================================ diff --git a/migrations/026_*.sql.skip b/migrations/026_*.sql.skip deleted file mode 100644 index bbbd5cda5..000000000 --- a/migrations/026_*.sql.skip +++ /dev/null @@ -1,51 +0,0 @@ --- ================================================================================================ --- Migration 026: Add account_id column to ensemble_predictions table --- Fixes schema drift - column expected by code but missing from table --- ================================================================================================ - --- Add account_id column (nullable for backward compatibility) -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS account_id VARCHAR(64); - --- Add index for account_id queries -CREATE INDEX IF NOT EXISTS idx_ensemble_predictions_account_id -ON ensemble_predictions(account_id) -WHERE account_id IS NOT NULL; - --- Add missing columns that may have been skipped -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS ab_variant VARCHAR(50); - -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS dqn_checkpoint_id VARCHAR(255); - -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS ppo_checkpoint_id VARCHAR(255); - -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS mamba2_checkpoint_id VARCHAR(255); - -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS tft_checkpoint_id VARCHAR(255); - -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS node_id VARCHAR(50); - -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS user_id VARCHAR(64); - -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS session_id UUID; - -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS request_id UUID; - -ALTER TABLE ensemble_predictions -ADD COLUMN IF NOT EXISTS strategy_id VARCHAR(100); - --- Add comment -COMMENT ON COLUMN ensemble_predictions.account_id IS 'Trading account identifier for multi-account tracking'; - --- ================================================================================================ --- END MIGRATION 026 --- ================================================================================================ diff --git a/migrations/027_*.sql.skip b/migrations/027_*.sql.skip deleted file mode 100644 index 33ca789b2..000000000 --- a/migrations/027_*.sql.skip +++ /dev/null @@ -1,41 +0,0 @@ --- ================================================================================================ --- Migration 027: Create get_top_models_24h() PostgreSQL function --- Utility function for retrieving top performing models in last 24 hours --- ================================================================================================ - --- Function: Get top performing models in last 24 hours -CREATE OR REPLACE FUNCTION get_top_models_24h( - p_limit INT, - p_min_predictions INT -) -RETURNS TABLE ( - model_id VARCHAR, - total_predictions BIGINT, - accuracy FLOAT, - sharpe_ratio FLOAT, - total_pnl FLOAT, - avg_weight FLOAT -) AS $$ -BEGIN - RETURN QUERY - SELECT - COALESCE(mpa.model_id, 'UNKNOWN')::VARCHAR as model_id, - COALESCE(mpa.total_predictions, 0)::BIGINT as total_predictions, - COALESCE(mpa.accuracy, 0.0)::FLOAT as accuracy, - COALESCE(mpa.sharpe_ratio, 0.0)::FLOAT as sharpe_ratio, - COALESCE(mpa.total_pnl::FLOAT, 0.0) as total_pnl, - COALESCE(mpa.avg_weight, 0.0)::FLOAT as avg_weight - FROM model_performance_attribution mpa - WHERE - mpa.timestamp >= NOW() - INTERVAL '24 hours' - AND mpa.total_predictions >= p_min_predictions - ORDER BY mpa.sharpe_ratio DESC NULLS LAST - LIMIT p_limit; -END; -$$ LANGUAGE plpgsql; - -COMMENT ON FUNCTION get_top_models_24h IS 'Get top N performing models in last 24 hours by Sharpe ratio (requires minimum prediction count)'; - --- ================================================================================================ --- END MIGRATION 027 --- ================================================================================================ diff --git a/migrations/028_*.sql.skip b/migrations/028_*.sql.skip deleted file mode 100644 index b2e12d00a..000000000 --- a/migrations/028_*.sql.skip +++ /dev/null @@ -1,54 +0,0 @@ --- ================================================================================================ --- Migration 028: Create get_high_disagreement_events_24h() PostgreSQL function --- Utility function for retrieving high model disagreement events --- ================================================================================================ - --- Drop existing function if it exists -DROP FUNCTION IF EXISTS get_high_disagreement_events_24h(FLOAT, INT, INT); -DROP FUNCTION IF EXISTS get_high_disagreement_events_24h(VARCHAR, FLOAT, INT); - --- Function: Get high disagreement events in last 24 hours --- Parameters match code usage: symbol (optional), disagreement_threshold, limit -CREATE OR REPLACE FUNCTION get_high_disagreement_events_24h( - p_symbol VARCHAR, - p_disagreement_threshold FLOAT, - p_limit INT -) -RETURNS TABLE ( - event_timestamp TIMESTAMPTZ, - event_symbol VARCHAR, - ensemble_action VARCHAR, - ensemble_confidence FLOAT, - disagreement_rate FLOAT, - dqn_vote VARCHAR, - ppo_vote VARCHAR, - mamba2_vote VARCHAR, - tft_vote VARCHAR -) AS $$ -BEGIN - RETURN QUERY - SELECT - ep.timestamp as event_timestamp, - ep.symbol::VARCHAR as event_symbol, - ep.ensemble_action::VARCHAR, - ep.ensemble_confidence::FLOAT, - ep.disagreement_rate::FLOAT, - ep.dqn_vote::VARCHAR, - ep.ppo_vote::VARCHAR, - ep.mamba2_vote::VARCHAR, - ep.tft_vote::VARCHAR - FROM ensemble_predictions ep - WHERE - ep.timestamp >= NOW() - INTERVAL '24 hours' - AND (p_symbol IS NULL OR ep.symbol = p_symbol) - AND ep.disagreement_rate >= p_disagreement_threshold - ORDER BY ep.disagreement_rate DESC, ep.timestamp DESC - LIMIT p_limit; -END; -$$ LANGUAGE plpgsql; - -COMMENT ON FUNCTION get_high_disagreement_events_24h IS 'Get predictions with high model disagreement (possible regime shifts or market transitions)'; - --- ================================================================================================ --- END MIGRATION 028 --- ================================================================================================ diff --git a/migrations/029_*.sql.skip b/migrations/029_*.sql.skip deleted file mode 100644 index ef5e49223..000000000 --- a/migrations/029_*.sql.skip +++ /dev/null @@ -1,43 +0,0 @@ --- ================================================================================================ --- Migration 029: Fix order_side enum type compatibility --- Ensures order_side enum accepts lowercase values and text casting --- ================================================================================================ - --- Verify order_side enum exists and has correct values -DO $$ -BEGIN - -- Check if enum type exists - IF NOT EXISTS (SELECT 1 FROM pg_type WHERE typname = 'order_side') THEN - RAISE EXCEPTION 'order_side enum type does not exist'; - END IF; - - -- Verify enum has lowercase values (buy, sell) - IF NOT EXISTS ( - SELECT 1 FROM pg_enum - WHERE enumtypid = 'order_side'::regtype - AND enumlabel IN ('buy', 'sell') - ) THEN - RAISE NOTICE 'order_side enum values are not lowercase, this is expected'; - END IF; -END $$; - --- Add comment documenting type casting requirement -COMMENT ON TYPE order_side IS 'Order side enum (buy, sell) - Use ::order_side cast or "as _" override in SQLx queries'; - --- Create helper function to normalize order side strings -CREATE OR REPLACE FUNCTION normalize_order_side(side_text TEXT) -RETURNS order_side AS $$ -BEGIN - RETURN CASE LOWER(TRIM(side_text)) - WHEN 'buy' THEN 'buy'::order_side - WHEN 'sell' THEN 'sell'::order_side - ELSE NULL::order_side - END; -END; -$$ LANGUAGE plpgsql IMMUTABLE; - -COMMENT ON FUNCTION normalize_order_side IS 'Convert text to order_side enum with case normalization'; - --- ================================================================================================ --- END MIGRATION 029 --- ================================================================================================ diff --git a/migrations/030_*.sql.skip b/migrations/030_*.sql.skip deleted file mode 100644 index 7820c0926..000000000 --- a/migrations/030_*.sql.skip +++ /dev/null @@ -1,80 +0,0 @@ --- Migration 030: Create A/B Test Results Table --- Creates table for storing A/B testing pipeline results and deployment decisions - -CREATE TABLE IF NOT EXISTS ab_test_results ( - -- Primary identifiers - test_id VARCHAR(100) PRIMARY KEY, - - -- Test configuration - control_model VARCHAR(100) NOT NULL, - treatment_model VARCHAR(100) NOT NULL, - symbol VARCHAR(20) NOT NULL, - traffic_split DOUBLE PRECISION NOT NULL DEFAULT 0.5, - min_sample_size INTEGER NOT NULL DEFAULT 1000, - - -- Test status - status VARCHAR(50) NOT NULL DEFAULT 'running', - start_time TIMESTAMPTZ NOT NULL DEFAULT NOW(), - end_time TIMESTAMPTZ, - - -- Control group metrics - control_predictions BIGINT DEFAULT 0, - control_correct_predictions BIGINT DEFAULT 0, - control_win_rate DOUBLE PRECISION DEFAULT 0.0, - control_total_pnl DOUBLE PRECISION DEFAULT 0.0, - control_sharpe DOUBLE PRECISION DEFAULT 0.0, - control_avg_latency_us DOUBLE PRECISION DEFAULT 0.0, - - -- Treatment group metrics - treatment_predictions BIGINT DEFAULT 0, - treatment_correct_predictions BIGINT DEFAULT 0, - treatment_win_rate DOUBLE PRECISION DEFAULT 0.0, - treatment_total_pnl DOUBLE PRECISION DEFAULT 0.0, - treatment_sharpe DOUBLE PRECISION DEFAULT 0.0, - treatment_avg_latency_us DOUBLE PRECISION DEFAULT 0.0, - - -- Statistical test results - sharpe_diff DOUBLE PRECISION, - sharpe_p_value DOUBLE PRECISION, - sharpe_significant BOOLEAN, - pnl_diff DOUBLE PRECISION, - pnl_p_value DOUBLE PRECISION, - pnl_significant BOOLEAN, - - -- Deployment decision (JSON) - decision JSONB, - - -- Audit trail - created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(), - updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW() -); - --- Indexes for querying -CREATE INDEX idx_ab_test_results_status ON ab_test_results (status); -CREATE INDEX idx_ab_test_results_symbol ON ab_test_results (symbol); -CREATE INDEX idx_ab_test_results_start_time ON ab_test_results (start_time DESC); -CREATE INDEX idx_ab_test_results_control_model ON ab_test_results (control_model); -CREATE INDEX idx_ab_test_results_treatment_model ON ab_test_results (treatment_model); - --- Trigger to update updated_at timestamp -CREATE OR REPLACE FUNCTION update_ab_test_results_timestamp() -RETURNS TRIGGER AS $$ -BEGIN - NEW.updated_at = NOW(); - RETURN NEW; -END; -$$ LANGUAGE plpgsql; - -CREATE TRIGGER trigger_update_ab_test_results_timestamp - BEFORE UPDATE ON ab_test_results - FOR EACH ROW - EXECUTE FUNCTION update_ab_test_results_timestamp(); - --- Comments -COMMENT ON TABLE ab_test_results IS 'A/B testing pipeline results for automated model deployment decisions'; -COMMENT ON COLUMN ab_test_results.test_id IS 'Unique test identifier'; -COMMENT ON COLUMN ab_test_results.control_model IS 'Baseline model ID (e.g., DQN_v1.0.0)'; -COMMENT ON COLUMN ab_test_results.treatment_model IS 'New model ID under test (e.g., DQN_v2.0.0)'; -COMMENT ON COLUMN ab_test_results.traffic_split IS 'Traffic split ratio (0.5 = 50/50)'; -COMMENT ON COLUMN ab_test_results.status IS 'Test status: running, completed_rollout, completed_revert, completed_neutral, completed_inconclusive'; -COMMENT ON COLUMN ab_test_results.decision IS 'JSON deployment decision: RolloutTreatment, RevertToControl, Neutral, Inconclusive'; diff --git a/monitoring/alertmanager/alertmanager.yml b/monitoring/alertmanager/alertmanager.yml deleted file mode 100644 index 70821dd6f..000000000 --- a/monitoring/alertmanager/alertmanager.yml +++ /dev/null @@ -1,257 +0,0 @@ -# AlertManager Configuration for Foxhunt -# -# Routes alerts to appropriate notification channels - -global: - resolve_timeout: 5m - slack_api_url: 'https://hooks.slack.com/services/YOUR/SLACK/WEBHOOK' - pagerduty_url: 'https://events.pagerduty.com/v2/enqueue' - -# Alert routing tree -route: - receiver: 'default' - group_by: ['alertname', 'cluster', 'service'] - group_wait: 10s - group_interval: 5m - repeat_interval: 4h - - # Route alerts based on severity - routes: - # Critical ensemble alerts - IMMEDIATE PagerDuty + Slack - - match: - severity: critical - component: ensemble - receiver: 'ensemble-critical' - group_wait: 0s - repeat_interval: 30m - continue: false - - # Warning ensemble alerts - Slack only - - match: - severity: warning - component: ensemble - receiver: 'ensemble-warnings' - group_wait: 15s - repeat_interval: 2h - continue: false - - # Info ensemble alerts (A/B test results) - Slack only - - match: - severity: info - component: ensemble - receiver: 'ensemble-info' - group_wait: 5m - repeat_interval: 24h - continue: false - - # Critical alerts (non-ensemble) - PagerDuty + Slack - - match: - severity: critical - receiver: 'critical-alerts' - group_wait: 0s - repeat_interval: 1h - - # Warning alerts - less urgent - - match: - severity: warning - receiver: 'warning-alerts' - group_wait: 30s - repeat_interval: 4h - - # Auth-specific alerts - - match: - component: auth - receiver: 'auth-alerts' - group_by: ['alertname'] - - # Backend proxy alerts - - match: - component: proxy - receiver: 'backend-alerts' - - # Configuration alerts - - match: - component: config - receiver: 'config-alerts' - -# Alert receivers (notification channels) -receivers: - # Default receiver (logs only) - - name: 'default' - webhook_configs: - - url: 'http://localhost:9090/api/v1/alerts' - - # ========== ENSEMBLE ALERTS ========== - # Ensemble critical alerts - PagerDuty + Slack - - name: 'ensemble-critical' - slack_configs: - - channel: '#foxhunt-ensemble-critical' - title: '🚨 ENSEMBLE CRITICAL: {{ .GroupLabels.alertname }}' - text: | - *Alert:* {{ .GroupLabels.alertname }} - *Type:* {{ .CommonLabels.alert_type }} - *Symbol:* {{ .CommonLabels.symbol }} - {{ range .Alerts }} - *Summary:* {{ .Annotations.summary }} - *Description:* {{ .Annotations.description }} - *Impact:* {{ .Annotations.impact }} - *Action Required:* - {{ .Annotations.action }} - *Runbook:* {{ .Annotations.runbook_url }} - {{ end }} - send_resolved: true - color: '{{ if eq .Status "firing" }}danger{{ else }}good{{ end }}' - - pagerduty_configs: - - routing_key: 'YOUR_PAGERDUTY_ENSEMBLE_INTEGRATION_KEY' - severity: 'critical' - description: '{{ .GroupLabels.alertname }}: {{ .CommonAnnotations.summary }}' - details: - alert_type: '{{ .CommonLabels.alert_type }}' - symbol: '{{ .CommonLabels.symbol }}' - impact: '{{ .CommonAnnotations.impact }}' - action: '{{ .CommonAnnotations.action }}' - runbook_url: '{{ .CommonAnnotations.runbook_url }}' - client: 'Foxhunt Ensemble Monitoring' - client_url: 'http://localhost:3000/d/ensemble-ml-production' - - # Ensemble warning alerts - Slack only - - name: 'ensemble-warnings' - slack_configs: - - channel: '#foxhunt-ensemble-warnings' - title: '⚠️ ENSEMBLE WARNING: {{ .GroupLabels.alertname }}' - text: | - *Alert:* {{ .GroupLabels.alertname }} - *Type:* {{ .CommonLabels.alert_type }} - *Symbol:* {{ .CommonLabels.symbol }} - {{ range .Alerts }} - *Summary:* {{ .Annotations.summary }} - *Description:* {{ .Annotations.description }} - {{ if .Annotations.action }}*Action:* {{ .Annotations.action }}{{ end }} - {{ end }} - send_resolved: true - color: 'warning' - - # Ensemble info alerts (A/B tests, model updates) - - name: 'ensemble-info' - slack_configs: - - channel: '#foxhunt-ensemble-info' - title: 'ℹ️ ENSEMBLE INFO: {{ .GroupLabels.alertname }}' - text: | - *Alert:* {{ .GroupLabels.alertname }} - {{ range .Alerts }} - *Summary:* {{ .Annotations.summary }} - *Description:* {{ .Annotations.description }} - {{ end }} - send_resolved: true - color: 'good' - - # ========== GENERAL ALERTS ========== - # Critical alerts (non-ensemble) - PagerDuty + Slack - - name: 'critical-alerts' - slack_configs: - - channel: '#foxhunt-critical' - title: '🚨 CRITICAL: {{ .GroupLabels.alertname }}' - text: '{{ range .Alerts }}{{ .Annotations.summary }}\n{{ .Annotations.description }}{{ end }}' - send_resolved: true - - # PagerDuty for on-call rotation - pagerduty_configs: - - routing_key: 'YOUR_PAGERDUTY_SERVICE_KEY' - severity: 'critical' - description: '{{ .GroupLabels.alertname }}' - - # Warning alerts - Slack only - - name: 'warning-alerts' - slack_configs: - - channel: '#foxhunt-warnings' - title: '⚠️ Warning: {{ .GroupLabels.alertname }}' - text: '{{ range .Alerts }}{{ .Annotations.summary }}{{ end }}' - send_resolved: true - - # Auth-specific alerts - - name: 'auth-alerts' - slack_configs: - - channel: '#foxhunt-auth' - title: 'πŸ” Auth Alert: {{ .GroupLabels.alertname }}' - text: '{{ range .Alerts }}{{ .Annotations.description }}{{ end }}' - - # Backend alerts - - name: 'backend-alerts' - slack_configs: - - channel: '#foxhunt-backend' - title: 'πŸ”Œ Backend Alert: {{ .GroupLabels.alertname }}' - text: '{{ range .Alerts }}{{ .Annotations.description }}{{ end }}' - - # Config alerts - - name: 'config-alerts' - slack_configs: - - channel: '#foxhunt-config' - title: 'βš™οΈ Config Alert: {{ .GroupLabels.alertname }}' - text: '{{ range .Alerts }}{{ .Annotations.description }}{{ end }}' - -# Inhibition rules (suppress redundant alerts) -inhibit_rules: - # ========== ENSEMBLE INHIBITION RULES ========== - # If cascade failure detected, suppress individual model failures - - source_match: - alertname: 'EnsembleCascadeFailureDetected' - target_match: - alertname: 'EnsembleModelFailureDetected' - equal: ['cluster'] - - # If Sharpe ratio drop critical, suppress warning - - source_match: - alertname: 'EnsembleSharpeRatioDropCritical' - target_match: - alertname: 'EnsembleSharpeRatioDropWarning' - equal: ['symbol'] - - # If high memory, suppress latency alerts (latency caused by swapping) - - source_match: - alertname: 'EnsembleServiceMemoryHigh' - target_match: - alertname: 'EnsembleAggregationLatencyP99High' - equal: ['instance'] - - # If checkpoint rollback rate high, suppress individual swap failures - - source_match: - alertname: 'EnsembleCheckpointRollbackRateHigh' - target_match: - alertname: 'EnsembleCheckpointSwapFailed' - equal: ['cluster'] - - # If high disagreement critical, suppress warning - - source_match: - alertname: 'EnsembleHighDisagreementCritical' - target_match: - alertname: 'EnsembleHighDisagreementWarning' - equal: ['symbol'] - - # If low confidence + high disagreement, suppress both individual alerts - - source_match: - alertname: 'EnsembleLowConfidenceHighDisagreement' - target_match_re: - alertname: 'EnsembleHighDisagreementCritical|EnsembleHighDisagreementWarning' - equal: ['symbol'] - - # ========== GENERAL INHIBITION RULES ========== - # If circuit breaker is open, suppress high latency alerts - - source_match: - alertname: 'CircuitBreakerOpen' - target_match: - alertname: 'HighBackendLatency' - equal: ['service'] - - # If backend is unhealthy, suppress other backend alerts - - source_match: - alertname: 'BackendServiceUnhealthy' - target_match_re: - alertname: 'HighBackendLatency|CircuitBreakerOpen' - equal: ['service'] - - # If NOTIFY listener is down, suppress config alerts - - source_match: - alertname: 'NotifyListenerDisconnected' - target_match_re: - alertname: 'HighConfigReloadLatency|ConfigValidationFailures' diff --git a/monitoring/alertmanager/ml_notification_config.yml b/monitoring/alertmanager/ml_notification_config.yml deleted file mode 100644 index eb8db7a35..000000000 --- a/monitoring/alertmanager/ml_notification_config.yml +++ /dev/null @@ -1,166 +0,0 @@ -# ML Training Service Notification Configuration -# -# Integrated with main AlertManager configuration for ML-specific routing - -# ML Training Service Alert Routes (add to alertmanager.yml) -ml_training_routes: - # Critical ML alerts - PagerDuty + Slack - - match: - severity: critical - component: ml - receiver: 'ml-critical-alerts' - group_wait: 0s - repeat_interval: 30m - continue: false - - # High severity ML alerts - Slack + Email - - match: - severity: high - component: ml - receiver: 'ml-high-alerts' - group_wait: 15s - repeat_interval: 1h - continue: false - - # Warning ML alerts - Slack only - - match: - severity: warning - component: ml - receiver: 'ml-warning-alerts' - group_wait: 30s - repeat_interval: 4h - continue: false - - # Info ML alerts - Slack #ml-info channel - - match: - severity: info - component: ml - receiver: 'ml-info-alerts' - group_wait: 5m - repeat_interval: 24h - continue: false - -# ML Training Service Alert Receivers -ml_receivers: - # Critical ML alerts - PagerDuty + Slack - - name: 'ml-critical-alerts' - slack_configs: - - channel: '#foxhunt-ml-critical' - api_url: '${SLACK_WEBHOOK_URL}' - title: '🚨 ML TRAINING CRITICAL: {{ .GroupLabels.alertname }}' - text: | - *Alert:* {{ .GroupLabels.alertname }} - *Model Type:* {{ .CommonLabels.model_type }} - *Job ID:* {{ .CommonLabels.job_id }} - {{ range .Alerts }} - *Summary:* {{ .Annotations.summary }} - *Description:* {{ .Annotations.description }} - *Impact:* {{ .Annotations.impact }} - *Action Required:* - {{ .Annotations.action }} - *Runbook:* {{ .Annotations.runbook_url }} - {{ end }} - send_resolved: true - color: '{{ if eq .Status "firing" }}danger{{ else }}good{{ end }}' - - pagerduty_configs: - - routing_key: '${PAGERDUTY_ML_INTEGRATION_KEY}' - severity: 'critical' - description: '{{ .GroupLabels.alertname }}: {{ .CommonAnnotations.summary }}' - details: - alert_type: '{{ .CommonLabels.alert_type }}' - model_type: '{{ .CommonLabels.model_type }}' - job_id: '{{ .CommonLabels.job_id }}' - impact: '{{ .CommonAnnotations.impact }}' - action: '{{ .CommonAnnotations.action }}' - runbook_url: '{{ .CommonAnnotations.runbook_url }}' - client: 'Foxhunt ML Training Service' - client_url: 'http://localhost:3000/d/ml-training-monitoring' - - # High severity ML alerts - - name: 'ml-high-alerts' - slack_configs: - - channel: '#foxhunt-ml-high' - api_url: '${SLACK_WEBHOOK_URL}' - title: '⚠️ ML TRAINING HIGH: {{ .GroupLabels.alertname }}' - text: | - *Alert:* {{ .GroupLabels.alertname }} - {{ range .Alerts }} - *Summary:* {{ .Annotations.summary }} - *Description:* {{ .Annotations.description }} - *Impact:* {{ .Annotations.impact }} - {{ if .Annotations.action }}*Action:* {{ .Annotations.action }}{{ end }} - {{ end }} - send_resolved: true - color: 'danger' - - # Warning ML alerts - - name: 'ml-warning-alerts' - slack_configs: - - channel: '#foxhunt-ml-warnings' - api_url: '${SLACK_WEBHOOK_URL}' - title: '⚠️ ML TRAINING WARNING: {{ .GroupLabels.alertname }}' - text: | - *Alert:* {{ .GroupLabels.alertname }} - {{ range .Alerts }} - *Summary:* {{ .Annotations.summary }} - *Description:* {{ .Annotations.description }} - {{ end }} - send_resolved: true - color: 'warning' - - # Info ML alerts - - name: 'ml-info-alerts' - slack_configs: - - channel: '#foxhunt-ml-info' - api_url: '${SLACK_WEBHOOK_URL}' - title: 'ℹ️ ML TRAINING INFO: {{ .GroupLabels.alertname }}' - text: | - *Alert:* {{ .GroupLabels.alertname }} - {{ range .Alerts }} - *Summary:* {{ .Annotations.summary }} - *Description:* {{ .Annotations.description }} - {{ end }} - send_resolved: true - color: 'good' - -# Inhibition Rules for ML Training Service -ml_inhibit_rules: - # If GPU memory exhausted, suppress GPU memory high warning - - source_match: - alertname: 'GPUMemoryExhausted' - target_match: - alertname: 'GPUMemoryUsageHigh' - equal: ['gpu_id'] - - # If training job failed, suppress progress/slowdown alerts - - source_match: - alertname: 'TrainingJobFailed' - target_match_re: - alertname: 'TrainingSlowdown|ModelConvergenceStalled' - equal: ['job_id'] - - # If automated job stuck, suppress other job-related alerts - - source_match: - alertname: 'AutomatedTrainingJobStuck' - target_match_re: - alertname: 'TrainingSlowdown|TrainingIterationTimeSlow' - equal: ['job_id'] - - # If data drift detected, suppress model accuracy degraded - - source_match: - alertname: 'ModelDriftDetected' - target_match: - alertname: 'MLModelAccuracyDegraded' - equal: ['model'] - - # If S3 connection errors, suppress checkpoint save failures - - source_match: - alertname: 'S3ConnectionErrors' - target_match: - alertname: 'CheckpointSaveFailures' - equal: ['job_id'] - -# Environment Variables (set in deployment environment) -# export SLACK_WEBHOOK_URL=https://hooks.slack.com/services/YOUR/SLACK/WEBHOOK -# export PAGERDUTY_ML_INTEGRATION_KEY=your_pagerduty_integration_key diff --git a/monitoring/docker-compose.yml b/monitoring/docker-compose.yml deleted file mode 100644 index d85a5e082..000000000 --- a/monitoring/docker-compose.yml +++ /dev/null @@ -1,126 +0,0 @@ -# Foxhunt Monitoring Stack -# -# Services: -# - Prometheus: Metrics collection and alerting -# - Grafana: Metrics visualization -# - AlertManager: Alert routing and notification -# - PostgreSQL Exporter: Database metrics -# - Redis Exporter: Cache metrics - -version: '3.8' - -services: - # Prometheus - Metrics collection - prometheus: - image: prom/prometheus:v2.48.0 - container_name: foxhunt-prometheus - restart: unless-stopped - ports: - - "9099:9090" - volumes: - - ./prometheus/prometheus.yml:/etc/prometheus/prometheus.yml:ro - - ./prometheus/alerts:/etc/prometheus/alerts:ro - - prometheus-data:/prometheus - command: - - '--config.file=/etc/prometheus/prometheus.yml' - - '--storage.tsdb.path=/prometheus' - - '--storage.tsdb.retention.time=30d' - - '--web.enable-lifecycle' - - '--web.enable-admin-api' - networks: - - foxhunt-monitoring - - # Grafana - Metrics visualization - grafana: - image: grafana/grafana:10.2.2 - container_name: foxhunt-grafana - restart: unless-stopped - ports: - - "3000:3000" - environment: - - GF_SECURITY_ADMIN_USER=admin - - GF_SECURITY_ADMIN_PASSWORD=foxhunt2025 - - GF_USERS_ALLOW_SIGN_UP=false - - GF_SERVER_ROOT_URL=http://localhost:3000 - - GF_INSTALL_PLUGINS= - volumes: - - ./grafana:/etc/grafana/provisioning/dashboards:ro - - grafana-data:/var/lib/grafana - depends_on: - - prometheus - networks: - - foxhunt-monitoring - - # AlertManager - Alert routing - alertmanager: - image: prom/alertmanager:v0.26.0 - container_name: foxhunt-alertmanager - restart: unless-stopped - ports: - - "9093:9093" - volumes: - - ./alertmanager/alertmanager.yml:/etc/alertmanager/alertmanager.yml:ro - - alertmanager-data:/alertmanager - command: - - '--config.file=/etc/alertmanager/alertmanager.yml' - - '--storage.path=/alertmanager' - networks: - - foxhunt-monitoring - - # PostgreSQL Exporter - Database metrics - postgres-exporter: - image: prometheuscommunity/postgres-exporter:v0.15.0 - container_name: foxhunt-postgres-exporter - restart: unless-stopped - ports: - - "9187:9187" - environment: - - DATA_SOURCE_NAME=postgresql://foxhunt:foxhunt_dev_password@postgres:5432/foxhunt?sslmode=disable - networks: - - foxhunt-monitoring - - foxhunt_foxhunt-network - - # Redis Exporter - Cache metrics - redis-exporter: - image: oliver006/redis_exporter:v1.55.0 - container_name: foxhunt-redis-exporter - restart: unless-stopped - ports: - - "9121:9121" - environment: - - REDIS_ADDR=redis://redis:6379 - networks: - - foxhunt-monitoring - - # Node Exporter - System metrics (API Gateway host) - node-exporter-gateway: - image: prom/node-exporter:v1.7.0 - container_name: foxhunt-node-exporter-gateway - restart: unless-stopped - ports: - - "9100:9100" - volumes: - - /proc:/host/proc:ro - - /sys:/host/sys:ro - - /:/rootfs:ro - command: - - '--path.procfs=/host/proc' - - '--path.sysfs=/host/sys' - - '--collector.filesystem.mount-points-exclude=^/(sys|proc|dev|host|etc)($$|/)' - networks: - - foxhunt-monitoring - -networks: - foxhunt-monitoring: - name: foxhunt-monitoring - driver: bridge - foxhunt_foxhunt-network: - external: true - -volumes: - prometheus-data: - name: foxhunt-prometheus-data - grafana-data: - name: foxhunt-grafana-data - alertmanager-data: - name: foxhunt-alertmanager-data diff --git a/monitoring/grafana/api_gateway_dashboard.json b/monitoring/grafana/api_gateway_dashboard.json deleted file mode 100644 index 894eb6d37..000000000 --- a/monitoring/grafana/api_gateway_dashboard.json +++ /dev/null @@ -1,421 +0,0 @@ -{ - "dashboard": { - "title": "API Gateway - Authentication & Performance", - "tags": ["api-gateway", "authentication", "hft"], - "timezone": "browser", - "schemaVersion": 16, - "version": 1, - "refresh": "5s", - "panels": [ - { - "id": 1, - "title": "Authentication Overview", - "type": "row", - "gridPos": { "x": 0, "y": 0, "w": 24, "h": 1 } - }, - { - "id": 2, - "title": "Auth Requests (Total vs Success vs Failure)", - "type": "graph", - "gridPos": { "x": 0, "y": 1, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "rate(api_gateway_auth_requests_total[1m])", - "legendFormat": "Total Requests/s", - "refId": "A" - }, - { - "expr": "rate(api_gateway_auth_requests_success[1m])", - "legendFormat": "Success/s", - "refId": "B" - }, - { - "expr": "rate(api_gateway_auth_requests_failure[1m])", - "legendFormat": "Failures/s", - "refId": "C" - } - ], - "yaxes": [ - { "format": "reqps", "label": "Requests/s" }, - { "format": "short" } - ] - }, - { - "id": 3, - "title": "Auth Success Rate (%)", - "type": "singlestat", - "gridPos": { "x": 12, "y": 1, "w": 6, "h": 4 }, - "targets": [ - { - "expr": "100 * rate(api_gateway_auth_requests_success[5m]) / rate(api_gateway_auth_requests_total[5m])", - "refId": "A" - } - ], - "format": "percent", - "thresholds": "90,95", - "colors": ["#d44a3a", "#e0b400", "#299c46"] - }, - { - "id": 4, - "title": "Auth SLA Compliance (<10ΞΌs)", - "type": "singlestat", - "gridPos": { "x": 18, "y": 1, "w": 6, "h": 4 }, - "targets": [ - { - "expr": "100 * rate(api_gateway_auth_sla_met[5m]) / (rate(api_gateway_auth_sla_met[5m]) + rate(api_gateway_auth_sla_exceeded[5m]))", - "refId": "A" - } - ], - "format": "percent", - "thresholds": "95,99", - "colors": ["#d44a3a", "#e0b400", "#299c46"] - }, - { - "id": 5, - "title": "Authentication Layer Latencies (ΞΌs)", - "type": "graph", - "gridPos": { "x": 0, "y": 9, "w": 24, "h": 8 }, - "targets": [ - { - "expr": "histogram_quantile(0.99, rate(api_gateway_jwt_extraction_duration_microseconds_bucket[1m]))", - "legendFormat": "JWT Extraction p99", - "refId": "A" - }, - { - "expr": "histogram_quantile(0.99, rate(api_gateway_jwt_validation_duration_microseconds_bucket[1m]))", - "legendFormat": "JWT Validation p99", - "refId": "B" - }, - { - "expr": "histogram_quantile(0.99, rate(api_gateway_revocation_check_duration_microseconds_bucket[1m]))", - "legendFormat": "Revocation Check p99", - "refId": "C" - }, - { - "expr": "histogram_quantile(0.99, rate(api_gateway_rbac_check_duration_microseconds_bucket[1m]))", - "legendFormat": "RBAC Check p99", - "refId": "D" - }, - { - "expr": "histogram_quantile(0.99, rate(api_gateway_rate_limit_check_duration_microseconds_bucket[1m]))", - "legendFormat": "Rate Limit Check p99", - "refId": "E" - }, - { - "expr": "histogram_quantile(0.99, rate(api_gateway_auth_total_duration_microseconds_bucket[1m]))", - "legendFormat": "Total Auth p99", - "refId": "F" - } - ], - "yaxes": [ - { "format": "Β΅s", "label": "Latency (ΞΌs)" }, - { "format": "short" } - ], - "alert": { - "name": "Auth Latency SLA Violation", - "conditions": [ - { - "evaluator": { "params": [10], "type": "gt" }, - "query": { "params": ["F", "5m", "now"] }, - "type": "query" - } - ], - "message": "Authentication latency exceeded 10ΞΌs SLA" - } - }, - { - "id": 6, - "title": "Authentication Errors by Type", - "type": "graph", - "gridPos": { "x": 0, "y": 17, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "rate(api_gateway_auth_errors_missing_jwt[1m])", - "legendFormat": "Missing JWT", - "refId": "A" - }, - { - "expr": "rate(api_gateway_auth_errors_invalid_jwt[1m])", - "legendFormat": "Invalid JWT", - "refId": "B" - }, - { - "expr": "rate(api_gateway_auth_errors_expired_jwt[1m])", - "legendFormat": "Expired JWT", - "refId": "C" - }, - { - "expr": "rate(api_gateway_auth_errors_revoked_jwt[1m])", - "legendFormat": "Revoked JWT", - "refId": "D" - }, - { - "expr": "rate(api_gateway_auth_errors_permission_denied[1m])", - "legendFormat": "Permission Denied", - "refId": "E" - }, - { - "expr": "rate(api_gateway_auth_errors_rate_limited[1m])", - "legendFormat": "Rate Limited", - "refId": "F" - } - ], - "yaxes": [ - { "format": "reqps", "label": "Errors/s" }, - { "format": "short" } - ] - }, - { - "id": 7, - "title": "Cache Performance", - "type": "graph", - "gridPos": { "x": 12, "y": 17, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "100 * rate(api_gateway_jwt_cache_hits[1m]) / (rate(api_gateway_jwt_cache_hits[1m]) + rate(api_gateway_jwt_cache_misses[1m]))", - "legendFormat": "JWT Cache Hit Rate %", - "refId": "A" - }, - { - "expr": "100 * rate(api_gateway_rbac_cache_hits[1m]) / (rate(api_gateway_rbac_cache_hits[1m]) + rate(api_gateway_rbac_cache_misses[1m]))", - "legendFormat": "RBAC Cache Hit Rate %", - "refId": "B" - } - ], - "yaxes": [ - { "format": "percent", "label": "Hit Rate %", "min": 0, "max": 100 }, - { "format": "short" } - ] - }, - { - "id": 8, - "title": "Backend Services", - "type": "row", - "gridPos": { "x": 0, "y": 25, "w": 24, "h": 1 } - }, - { - "id": 9, - "title": "Backend Request Latency by Service (p99)", - "type": "graph", - "gridPos": { "x": 0, "y": 26, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "histogram_quantile(0.99, rate(api_gateway_backend_request_duration_milliseconds_bucket{service=\"trading\"}[1m]))", - "legendFormat": "Trading Service p99", - "refId": "A" - }, - { - "expr": "histogram_quantile(0.99, rate(api_gateway_backend_request_duration_milliseconds_bucket{service=\"backtesting\"}[1m]))", - "legendFormat": "Backtesting Service p99", - "refId": "B" - }, - { - "expr": "histogram_quantile(0.99, rate(api_gateway_backend_request_duration_milliseconds_bucket{service=\"ml_training\"}[1m]))", - "legendFormat": "ML Training Service p99", - "refId": "C" - } - ], - "yaxes": [ - { "format": "ms", "label": "Latency (ms)" }, - { "format": "short" } - ] - }, - { - "id": 10, - "title": "Circuit Breaker States", - "type": "graph", - "gridPos": { "x": 12, "y": 26, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "api_gateway_circuit_breaker_state{service=\"trading\"}", - "legendFormat": "Trading (0=closed, 2=open)", - "refId": "A" - }, - { - "expr": "api_gateway_circuit_breaker_state{service=\"backtesting\"}", - "legendFormat": "Backtesting", - "refId": "B" - }, - { - "expr": "api_gateway_circuit_breaker_state{service=\"ml_training\"}", - "legendFormat": "ML Training", - "refId": "C" - } - ], - "yaxes": [ - { "format": "short", "label": "State", "min": 0, "max": 2 }, - { "format": "short" } - ], - "alert": { - "name": "Circuit Breaker Open", - "conditions": [ - { - "evaluator": { "params": [1.5], "type": "gt" }, - "query": { "params": ["A", "1m", "now"] }, - "type": "query" - } - ], - "message": "Circuit breaker opened for backend service" - } - }, - { - "id": 11, - "title": "Backend Health Status", - "type": "table", - "gridPos": { "x": 0, "y": 34, "w": 12, "h": 6 }, - "targets": [ - { - "expr": "api_gateway_health_status", - "format": "table", - "instant": true, - "refId": "A" - } - ], - "styles": [ - { - "pattern": "Value", - "type": "string", - "mappingType": 1, - "valueMaps": [ - { "value": "0", "text": "Unhealthy" }, - { "value": "1", "text": "Healthy" } - ] - } - ] - }, - { - "id": 12, - "title": "Connection Pool Utilization", - "type": "graph", - "gridPos": { "x": 12, "y": 34, "w": 12, "h": 6 }, - "targets": [ - { - "expr": "100 * api_gateway_connection_pool_active / api_gateway_connection_pool_max", - "legendFormat": "{{service}} Pool Utilization %", - "refId": "A" - } - ], - "yaxes": [ - { "format": "percent", "label": "Pool Utilization %", "min": 0, "max": 100 }, - { "format": "short" } - ] - }, - { - "id": 13, - "title": "Configuration & Hot-Reload", - "type": "row", - "gridPos": { "x": 0, "y": 40, "w": 24, "h": 1 } - }, - { - "id": 14, - "title": "Configuration Reload Events", - "type": "graph", - "gridPos": { "x": 0, "y": 41, "w": 12, "h": 6 }, - "targets": [ - { - "expr": "rate(api_gateway_config_updates_auth[5m])", - "legendFormat": "Auth Config Updates", - "refId": "A" - }, - { - "expr": "rate(api_gateway_config_updates_routing[5m])", - "legendFormat": "Routing Config Updates", - "refId": "B" - }, - { - "expr": "rate(api_gateway_config_updates_rate_limit[5m])", - "legendFormat": "Rate Limit Config Updates", - "refId": "C" - }, - { - "expr": "rate(api_gateway_config_updates_backend[5m])", - "legendFormat": "Backend Config Updates", - "refId": "D" - } - ], - "yaxes": [ - { "format": "ops", "label": "Updates/s" }, - { "format": "short" } - ] - }, - { - "id": 15, - "title": "Hot-Reload Latency (p95)", - "type": "graph", - "gridPos": { "x": 12, "y": 41, "w": 12, "h": 6 }, - "targets": [ - { - "expr": "histogram_quantile(0.95, rate(api_gateway_config_reload_duration_milliseconds_bucket[1m]))", - "legendFormat": "Config Reload p95", - "refId": "A" - }, - { - "expr": "histogram_quantile(0.95, rate(api_gateway_config_fetch_duration_milliseconds_bucket[1m]))", - "legendFormat": "Config Fetch p95", - "refId": "B" - } - ], - "yaxes": [ - { "format": "ms", "label": "Latency (ms)" }, - { "format": "short" } - ] - }, - { - "id": 16, - "title": "NOTIFY Listener Status", - "type": "singlestat", - "gridPos": { "x": 0, "y": 47, "w": 6, "h": 4 }, - "targets": [ - { - "expr": "api_gateway_notify_listener_connected", - "refId": "A" - } - ], - "valueName": "current", - "valueMaps": [ - { "value": "0", "text": "Disconnected" }, - { "value": "1", "text": "Connected" } - ], - "thresholds": "0.5,1", - "colors": ["#d44a3a", "#e0b400", "#299c46"] - }, - { - "id": 17, - "title": "Rate Limiting", - "type": "row", - "gridPos": { "x": 0, "y": 51, "w": 24, "h": 1 } - }, - { - "id": 18, - "title": "Rate Limit Hits by User (Top 10)", - "type": "graph", - "gridPos": { "x": 0, "y": 52, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "topk(10, rate(api_gateway_rate_limits_by_user[1m]))", - "legendFormat": "{{user_id}}", - "refId": "A" - } - ], - "yaxes": [ - { "format": "reqps", "label": "Rate Limit Hits/s" }, - { "format": "short" } - ] - }, - { - "id": 19, - "title": "Active Rate Limiter Entries", - "type": "singlestat", - "gridPos": { "x": 12, "y": 52, "w": 6, "h": 4 }, - "targets": [ - { - "expr": "api_gateway_rate_limiter_entries", - "refId": "A" - } - ], - "format": "short", - "valueName": "current" - } - ] - } -} diff --git a/monitoring/grafana/ensemble_ml_production.json b/monitoring/grafana/ensemble_ml_production.json deleted file mode 100644 index 0e5843c2d..000000000 --- a/monitoring/grafana/ensemble_ml_production.json +++ /dev/null @@ -1,977 +0,0 @@ -{ - "annotations": { - "list": [ - { - "builtIn": 1, - "datasource": { - "type": "grafana", - "uid": "-- Grafana --" - }, - "enable": true, - "hide": true, - "iconColor": "rgba(0, 211, 255, 1)", - "name": "Annotations & Alerts", - "type": "dashboard" - } - ] - }, - "editable": true, - "fiscalYearStartMonth": 0, - "graphTooltip": 0, - "id": null, - "links": [], - "panels": [ - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisBorderShow": false, - "axisCenteredZero": false, - "axisColorMode": "text", - "axisLabel": "", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "tooltip": false, - "viz": false, - "legend": false - }, - "lineInterpolation": "linear", - "lineWidth": 2, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "never", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "max": 1, - "min": 0, - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "yellow", - "value": 0.5 - }, - { - "color": "red", - "value": 0.7 - } - ] - }, - "unit": "percentunit" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 0, - "y": 0 - }, - "id": 1, - "options": { - "legend": { - "calcs": [ - "mean", - "lastNotNull", - "max" - ], - "displayMode": "table", - "placement": "bottom", - "showLegend": true - }, - "tooltip": { - "mode": "single", - "sort": "none" - } - }, - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "ensemble_confidence_score{symbol=~\"$symbol\"}", - "legendFormat": "Confidence - {{symbol}}", - "range": true, - "refId": "A" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "ensemble_disagreement_rate{symbol=~\"$symbol\"}", - "legendFormat": "Disagreement - {{symbol}}", - "range": true, - "refId": "B" - } - ], - "title": "Panel 1: Ensemble Confidence & Disagreement", - "type": "timeseries", - "alert": { - "alertRuleTags": {}, - "conditions": [ - { - "evaluator": { - "params": [0.5], - "type": "gt" - }, - "operator": { - "type": "and" - }, - "query": { - "params": ["B", "5m", "now"] - }, - "reducer": { - "params": [], - "type": "last" - }, - "type": "query" - } - ], - "executionErrorState": "alerting", - "frequency": "1m", - "handler": 1, - "name": "High Disagreement Alert", - "noDataState": "no_data", - "notifications": [] - } - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisBorderShow": false, - "axisCenteredZero": false, - "axisColorMode": "text", - "axisLabel": "", - "axisPlacement": "auto", - "fillOpacity": 80, - "gradientMode": "none", - "hideFrom": { - "tooltip": false, - "viz": false, - "legend": false - }, - "lineWidth": 1 - }, - "mappings": [], - "max": 1, - "min": 0, - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - } - ] - }, - "unit": "percentunit" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 12, - "y": 0 - }, - "id": 2, - "options": { - "legend": { - "calcs": [ - "mean", - "lastNotNull" - ], - "displayMode": "table", - "placement": "bottom", - "showLegend": true - }, - "tooltip": { - "mode": "single", - "sort": "none" - } - }, - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "ensemble_model_weight{symbol=~\"$symbol\"}", - "legendFormat": "{{model_id}} - {{symbol}}", - "range": true, - "refId": "A" - } - ], - "title": "Panel 2: Model Weights (Dynamic Contribution)", - "type": "timeseries", - "options": { - "legend": { - "calcs": [ - "mean", - "lastNotNull" - ], - "displayMode": "table", - "placement": "bottom", - "showLegend": true - }, - "tooltip": { - "mode": "single", - "sort": "none" - } - } - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "fieldConfig": { - "defaults": { - "color": { - "mode": "thresholds" - }, - "custom": { - "align": "auto", - "cellOptions": { - "type": "color-background" - }, - "inspect": false - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "red", - "value": null - }, - { - "color": "yellow", - "value": 0 - }, - { - "color": "green", - "value": 100 - } - ] - }, - "unit": "currencyUSD" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 0, - "y": 8 - }, - "id": 3, - "options": { - "cellHeight": "sm", - "footer": { - "countRows": false, - "fields": "", - "reducer": [ - "sum" - ], - "show": true - }, - "showHeader": true - }, - "pluginVersion": "10.0.0", - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "sum(rate(ensemble_model_pnl_contribution_dollars_sum{symbol=~\"$symbol\"}[$__rate_interval])) by (model_id, symbol)", - "format": "table", - "legendFormat": "__auto", - "range": true, - "refId": "A" - } - ], - "title": "Panel 3: Per-Model P&L Attribution", - "transformations": [ - { - "id": "organize", - "options": { - "excludeByName": { - "Time": true - }, - "indexByName": { - "model_id": 0, - "symbol": 1, - "Value": 2 - }, - "renameByName": { - "Value": "P&L ($/sec)", - "model_id": "Model", - "symbol": "Symbol" - } - } - } - ], - "type": "table" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisBorderShow": false, - "axisCenteredZero": false, - "axisColorMode": "text", - "axisLabel": "Latency (ΞΌs)", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "tooltip": false, - "viz": false, - "legend": false - }, - "lineInterpolation": "linear", - "lineWidth": 2, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "never", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "line" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "yellow", - "value": 25 - }, - { - "color": "red", - "value": 50 - } - ] - }, - "unit": "Β΅s" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 12, - "y": 8 - }, - "id": 4, - "options": { - "legend": { - "calcs": [ - "mean", - "lastNotNull", - "max" - ], - "displayMode": "table", - "placement": "bottom", - "showLegend": true - }, - "tooltip": { - "mode": "single", - "sort": "none" - } - }, - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "histogram_quantile(0.50, rate(ensemble_aggregation_latency_microseconds_bucket{aggregation_method=~\"$aggregation_method\"}[$__rate_interval]))", - "legendFormat": "P50 - {{aggregation_method}}", - "range": true, - "refId": "A" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "histogram_quantile(0.95, rate(ensemble_aggregation_latency_microseconds_bucket{aggregation_method=~\"$aggregation_method\"}[$__rate_interval]))", - "legendFormat": "P95 - {{aggregation_method}}", - "range": true, - "refId": "B" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "histogram_quantile(0.99, rate(ensemble_aggregation_latency_microseconds_bucket{aggregation_method=~\"$aggregation_method\"}[$__rate_interval]))", - "legendFormat": "P99 - {{aggregation_method}}", - "range": true, - "refId": "C" - } - ], - "title": "Panel 4: Aggregation Latency (Target: P99 < 25ΞΌs)", - "type": "timeseries" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisBorderShow": false, - "axisCenteredZero": false, - "axisColorMode": "text", - "axisLabel": "Events/Hour", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "bars", - "fillOpacity": 80, - "gradientMode": "none", - "hideFrom": { - "tooltip": false, - "viz": false, - "legend": false - }, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "never", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "normal" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - } - ] - }, - "unit": "short" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 0, - "y": 16 - }, - "id": 5, - "options": { - "legend": { - "calcs": [ - "sum" - ], - "displayMode": "table", - "placement": "bottom", - "showLegend": true - }, - "tooltip": { - "mode": "single", - "sort": "none" - } - }, - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "rate(ensemble_high_disagreement_total{symbol=~\"$symbol\"}[1h]) * 3600", - "legendFormat": "{{symbol}} - {{threshold}}", - "range": true, - "refId": "A" - } - ], - "title": "Panel 5: High Disagreement Events (Market Regime Shifts)", - "type": "timeseries" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisBorderShow": false, - "axisCenteredZero": false, - "axisColorMode": "text", - "axisLabel": "Count", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "tooltip": false, - "viz": false, - "legend": false - }, - "lineInterpolation": "linear", - "lineWidth": 2, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "auto", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "yellow", - "value": 0.05 - }, - { - "color": "red", - "value": 0.1 - } - ] - }, - "unit": "short" - }, - "overrides": [ - { - "matcher": { - "id": "byName", - "options": "Rollback Rate" - }, - "properties": [ - { - "id": "color", - "value": { - "fixedColor": "red", - "mode": "fixed" - } - } - ] - } - ] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 12, - "y": 16 - }, - "id": 6, - "options": { - "legend": { - "calcs": [ - "sum", - "lastNotNull" - ], - "displayMode": "table", - "placement": "bottom", - "showLegend": true - }, - "tooltip": { - "mode": "single", - "sort": "none" - } - }, - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "checkpoint_swaps_total{status=\"success\"}", - "legendFormat": "Success - {{model_id}}", - "range": true, - "refId": "A" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "checkpoint_swaps_total{status=\"rollback\"}", - "legendFormat": "Rollback - {{model_id}}", - "range": true, - "refId": "B" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "sum(checkpoint_swaps_total{status=\"rollback\"}) / sum(checkpoint_swaps_total)", - "legendFormat": "Rollback Rate", - "range": true, - "refId": "C" - } - ], - "title": "Panel 6: Checkpoint Swap Health (Alert: Rollback > 10%)", - "type": "timeseries" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "fieldConfig": { - "defaults": { - "color": { - "mode": "thresholds" - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "red", - "value": null - }, - { - "color": "yellow", - "value": 0 - }, - { - "color": "green", - "value": 0.1 - } - ] - }, - "unit": "percentunit" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 0, - "y": 24 - }, - "id": 7, - "options": { - "minVizHeight": 75, - "minVizWidth": 75, - "orientation": "auto", - "reduceOptions": { - "values": false, - "calcs": [ - "lastNotNull" - ], - "fields": "" - }, - "showThresholdLabels": false, - "showThresholdMarkers": true, - "sizing": "auto" - }, - "pluginVersion": "10.0.0", - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "ab_test_metric_difference{test_id=~\"$test_id\", metric=\"sharpe_ratio\"}", - "legendFormat": "Sharpe Ratio Lift", - "range": true, - "refId": "A" - } - ], - "title": "Panel 7: A/B Test Progress - Sharpe Ratio Lift (Treatment vs Control)", - "type": "gauge" - }, - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "hideFrom": { - "tooltip": false, - "viz": false, - "legend": false - } - }, - "mappings": [] - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 12, - "y": 24 - }, - "id": 8, - "options": { - "displayLabels": [ - "percent" - ], - "legend": { - "displayMode": "table", - "placement": "bottom", - "showLegend": true, - "values": [ - "value" - ] - }, - "pieType": "pie", - "reduceOptions": { - "values": false, - "calcs": [ - "lastNotNull" - ], - "fields": "" - }, - "tooltip": { - "mode": "single", - "sort": "none" - } - }, - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "editorMode": "code", - "expr": "ab_test_assignments_total{test_id=~\"$test_id\"}", - "legendFormat": "{{group}}", - "range": true, - "refId": "A" - } - ], - "title": "A/B Test Group Assignments (Should be 50/50)", - "type": "piechart" - } - ], - "refresh": "5s", - "schemaVersion": 38, - "tags": [ - "ensemble", - "ml", - "production", - "trading" - ], - "templating": { - "list": [ - { - "current": { - "selected": false, - "text": "Prometheus", - "value": "Prometheus" - }, - "hide": 0, - "includeAll": false, - "label": "Data Source", - "multi": false, - "name": "DS_PROMETHEUS", - "options": [], - "query": "prometheus", - "refresh": 1, - "regex": "", - "skipUrlSync": false, - "type": "datasource" - }, - { - "allValue": ".*", - "current": { - "selected": true, - "text": "All", - "value": "$__all" - }, - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "definition": "label_values(ensemble_confidence_score, symbol)", - "hide": 0, - "includeAll": true, - "label": "Symbol", - "multi": true, - "name": "symbol", - "options": [], - "query": { - "query": "label_values(ensemble_confidence_score, symbol)", - "refId": "PrometheusVariableQueryEditor-VariableQuery" - }, - "refresh": 1, - "regex": "", - "skipUrlSync": false, - "sort": 1, - "type": "query" - }, - { - "allValue": ".*", - "current": { - "selected": true, - "text": "All", - "value": "$__all" - }, - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "definition": "label_values(ensemble_aggregation_latency_microseconds_bucket, aggregation_method)", - "hide": 0, - "includeAll": true, - "label": "Aggregation Method", - "multi": true, - "name": "aggregation_method", - "options": [], - "query": { - "query": "label_values(ensemble_aggregation_latency_microseconds_bucket, aggregation_method)", - "refId": "PrometheusVariableQueryEditor-VariableQuery" - }, - "refresh": 1, - "regex": "", - "skipUrlSync": false, - "sort": 1, - "type": "query" - }, - { - "allValue": ".*", - "current": { - "selected": true, - "text": "All", - "value": "$__all" - }, - "datasource": { - "type": "prometheus", - "uid": "${DS_PROMETHEUS}" - }, - "definition": "label_values(ab_test_assignments_total, test_id)", - "hide": 0, - "includeAll": true, - "label": "A/B Test ID", - "multi": true, - "name": "test_id", - "options": [], - "query": { - "query": "label_values(ab_test_assignments_total, test_id)", - "refId": "PrometheusVariableQueryEditor-VariableQuery" - }, - "refresh": 1, - "regex": "", - "skipUrlSync": false, - "sort": 1, - "type": "query" - } - ] - }, - "time": { - "from": "now-1h", - "to": "now" - }, - "timepicker": {}, - "timezone": "", - "title": "Ensemble ML Production Monitoring", - "uid": "ensemble-ml-prod", - "version": 1, - "weekStart": "" -} diff --git a/monitoring/grafana/ml_trading_dashboard.json b/monitoring/grafana/ml_trading_dashboard.json deleted file mode 100644 index 229667f9f..000000000 --- a/monitoring/grafana/ml_trading_dashboard.json +++ /dev/null @@ -1,1354 +0,0 @@ -{ - "dashboard": { - "title": "ML Trading Monitoring", - "uid": "ml-trading", - "description": "Real-time monitoring of ML trading models, ensemble decisions, and order execution", - "timezone": "browser", - "editable": true, - "graphTooltip": 1, - "time": { - "from": "now-1h", - "to": "now" - }, - "refresh": "10s", - "schemaVersion": 39, - "tags": ["ml", "trading", "production"], - - "templating": { - "list": [ - { - "name": "model_filter", - "label": "Model", - "type": "custom", - "multi": true, - "includeAll": true, - "allValue": ".*", - "query": "DQN,MAMBA2,PPO,TFT,TLOB,Liquid", - "options": [ - { "text": "All", "value": "$__all" }, - { "text": "DQN", "value": "DQN" }, - { "text": "MAMBA2", "value": "MAMBA2" }, - { "text": "PPO", "value": "PPO" }, - { "text": "TFT", "value": "TFT" }, - { "text": "TLOB", "value": "TLOB" }, - { "text": "Liquid", "value": "Liquid" } - ], - "current": { - "text": "All", - "value": "$__all" - } - }, - { - "name": "symbol_filter", - "label": "Symbol", - "type": "custom", - "multi": true, - "includeAll": true, - "allValue": ".*", - "query": "ES.FUT,NQ.FUT,ZN.FUT,6E.FUT,CL.FUT", - "options": [ - { "text": "All", "value": "$__all" }, - { "text": "ES.FUT", "value": "ES.FUT" }, - { "text": "NQ.FUT", "value": "NQ.FUT" }, - { "text": "ZN.FUT", "value": "ZN.FUT" }, - { "text": "6E.FUT", "value": "6E.FUT" }, - { "text": "CL.FUT", "value": "CL.FUT" } - ], - "current": { - "text": "All", - "value": "$__all" - } - }, - { - "name": "datasource", - "label": "Data Source", - "type": "datasource", - "query": "prometheus", - "current": { - "text": "Prometheus", - "value": "prometheus" - } - } - ] - }, - - "annotations": { - "list": [ - { - "name": "ML Alerts", - "datasource": "prometheus", - "enable": true, - "iconColor": "red", - "expr": "ALERTS{component=\"ml\",alertstate=\"firing\"}", - "titleFormat": "{{ alertname }}", - "textFormat": "{{ annotations.description }}", - "tagKeys": "severity,model" - }, - { - "name": "Model Deployments", - "datasource": "prometheus", - "enable": true, - "iconColor": "green", - "expr": "ml_model_deployment_timestamp", - "titleFormat": "Model Deployed", - "textFormat": "{{ model }} version {{ version }}" - } - ] - }, - - "panels": [ - { - "id": 1, - "type": "row", - "title": "Model Performance Overview", - "gridPos": { "x": 0, "y": 0, "w": 24, "h": 1 }, - "collapsed": false - }, - - { - "id": 2, - "type": "gauge", - "title": "ML Model Win Rate", - "datasource": "${datasource}", - "gridPos": { "x": 0, "y": 1, "w": 6, "h": 6 }, - "targets": [ - { - "expr": "100 * ml_model_win_rate{model=~\"$model_filter\",symbol=~\"$symbol_filter\"}", - "legendFormat": "{{ model }} - {{ symbol }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "percent", - "min": 0, - "max": 100, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "red" }, - { "value": 55, "color": "yellow" }, - { "value": 65, "color": "green" } - ] - } - } - }, - "options": { - "orientation": "auto", - "showThresholdLabels": true, - "showThresholdMarkers": true, - "text": { - "titleSize": 14, - "valueSize": 28 - } - }, - "description": "Win rate threshold: <55% (red), 55-65% (yellow), >65% (green)" - }, - - { - "id": 3, - "type": "barchart", - "title": "Sharpe Ratios by Model", - "datasource": "${datasource}", - "gridPos": { "x": 6, "y": 1, "w": 9, "h": 6 }, - "targets": [ - { - "expr": "ml_model_sharpe_ratio{model=~\"$model_filter\",symbol=~\"$symbol_filter\"}", - "legendFormat": "{{ model }} - {{ symbol }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "short", - "decimals": 2, - "color": { - "mode": "thresholds" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": -999, "color": "red" }, - { "value": 0, "color": "orange" }, - { "value": 1.0, "color": "yellow" }, - { "value": 1.5, "color": "green" }, - { "value": 2.0, "color": "blue" } - ] - } - }, - "overrides": [ - { - "matcher": { "id": "byName", "options": "Target" }, - "properties": [ - { - "id": "custom.lineStyle", - "value": { "fill": "dash" } - }, - { - "id": "color", - "value": { "fixedColor": "white", "mode": "fixed" } - } - ] - } - ] - }, - "options": { - "orientation": "horizontal", - "showValue": "always", - "groupWidth": 0.7, - "barWidth": 0.8 - }, - "description": "Sharpe ratio measures risk-adjusted returns. Target: >1.0 (industry standard), >1.5 (excellent)" - }, - - { - "id": 4, - "type": "histogram", - "title": "Prediction Confidence Distribution", - "datasource": "${datasource}", - "gridPos": { "x": 15, "y": 1, "w": 9, "h": 6 }, - "targets": [ - { - "expr": "ml_predictions_confidence{model=~\"$model_filter\",symbol=~\"$symbol_filter\"}", - "legendFormat": "{{ model }}", - "refId": "A", - "format": "time_series" - } - ], - "fieldConfig": { - "defaults": { - "unit": "short", - "custom": { - "fillOpacity": 70 - } - } - }, - "options": { - "bucketOffset": 0, - "bucketSize": 0.05 - }, - "description": "Distribution of model prediction confidence scores (0-1). Higher confidence = more certain predictions." - }, - - { - "id": 5, - "type": "row", - "title": "Order Activity", - "gridPos": { "x": 0, "y": 7, "w": 24, "h": 1 }, - "collapsed": false - }, - - { - "id": 6, - "type": "timeseries", - "title": "ML Orders by Model (5m rate)", - "datasource": "${datasource}", - "gridPos": { "x": 0, "y": 8, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "rate(ml_orders_submitted_total{model=~\"$model_filter\",symbol=~\"$symbol_filter\"}[5m]) * 60", - "legendFormat": "{{ model }} - {{ symbol }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "ops", - "custom": { - "drawStyle": "line", - "lineInterpolation": "smooth", - "lineWidth": 2, - "fillOpacity": 20, - "gradientMode": "opacity", - "spanNulls": false, - "showPoints": "never", - "pointSize": 5, - "stacking": { - "mode": "normal", - "group": "A" - }, - "axisPlacement": "auto", - "axisLabel": "Orders/min", - "scaleDistribution": { - "type": "linear" - } - } - } - }, - "options": { - "tooltip": { - "mode": "multi", - "sort": "desc" - }, - "legend": { - "displayMode": "table", - "placement": "bottom", - "showLegend": true, - "calcs": ["last", "mean", "max"] - } - }, - "description": "Order submission rate by model (stacked). Shows trading activity intensity." - }, - - { - "id": 7, - "type": "stat", - "title": "Fill Rate by Model", - "datasource": "${datasource}", - "gridPos": { "x": 12, "y": 8, "w": 6, "h": 8 }, - "targets": [ - { - "expr": "100 * sum by(model) (ml_orders_filled_total{model=~\"$model_filter\"}) / sum by(model) (ml_orders_submitted_total{model=~\"$model_filter\"})", - "legendFormat": "{{ model }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "percent", - "decimals": 1, - "color": { - "mode": "thresholds" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "red" }, - { "value": 70, "color": "orange" }, - { "value": 85, "color": "yellow" }, - { "value": 95, "color": "green" } - ] - } - } - }, - "options": { - "graphMode": "area", - "colorMode": "background", - "justifyMode": "center", - "textMode": "value_and_name", - "reduceOptions": { - "values": false, - "calcs": ["lastNotNull"] - }, - "orientation": "vertical" - }, - "description": "Percentage of submitted orders successfully filled. Low fill rate may indicate liquidity issues or aggressive pricing." - }, - - { - "id": 8, - "type": "piechart", - "title": "Order Rejection Reasons", - "datasource": "${datasource}", - "gridPos": { "x": 18, "y": 8, "w": 6, "h": 8 }, - "targets": [ - { - "expr": "sum by(reason) (ml_orders_rejected_total{model=~\"$model_filter\"})", - "legendFormat": "{{ reason }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "short" - } - }, - "options": { - "legend": { - "displayMode": "table", - "placement": "right", - "showLegend": true, - "values": ["value", "percent"] - }, - "pieType": "pie", - "tooltip": { - "mode": "single" - }, - "displayLabels": ["name", "percent"] - }, - "description": "Breakdown of why orders were rejected (e.g., risk limits, invalid price, insufficient margin)" - }, - - { - "id": 9, - "type": "row", - "title": "Ensemble Monitoring", - "gridPos": { "x": 0, "y": 16, "w": 24, "h": 1 }, - "collapsed": false - }, - - { - "id": 10, - "type": "gauge", - "title": "Ensemble Agreement Rate", - "datasource": "${datasource}", - "gridPos": { "x": 0, "y": 17, "w": 8, "h": 6 }, - "targets": [ - { - "expr": "100 * ml_ensemble_agreement_rate{symbol=~\"$symbol_filter\"}", - "legendFormat": "{{ symbol }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "percent", - "min": 0, - "max": 100, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "red" }, - { "value": 70, "color": "yellow" }, - { "value": 85, "color": "green" } - ] - } - } - }, - "options": { - "orientation": "auto", - "showThresholdLabels": true, - "showThresholdMarkers": true, - "text": { - "titleSize": 14, - "valueSize": 28 - } - }, - "description": "Percentage of predictions where all models agree. <70% triggers alert (high disagreement = uncertain market conditions)" - }, - - { - "id": 11, - "type": "timeseries", - "title": "Model Disagreement Events", - "datasource": "${datasource}", - "gridPos": { "x": 8, "y": 17, "w": 8, "h": 6 }, - "targets": [ - { - "expr": "rate(ml_ensemble_disagreement_events{symbol=~\"$symbol_filter\"}[1m]) * 60", - "legendFormat": "{{ symbol }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "events/min", - "custom": { - "drawStyle": "line", - "lineInterpolation": "smooth", - "lineWidth": 2, - "fillOpacity": 30, - "gradientMode": "hue", - "showPoints": "auto" - } - } - }, - "options": { - "tooltip": { - "mode": "multi" - }, - "legend": { - "displayMode": "list", - "placement": "bottom", - "showLegend": true - } - }, - "description": "Rate of significant model disagreements. Spikes indicate market regime changes or model divergence." - }, - - { - "id": 12, - "type": "stat", - "title": "Active Models (5m window)", - "datasource": "${datasource}", - "gridPos": { "x": 16, "y": 17, "w": 8, "h": 6 }, - "targets": [ - { - "expr": "count(count_over_time(ml_predictions_total{model=~\"$model_filter\"}[5m]) > 0)", - "legendFormat": "Active Models", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "short", - "decimals": 0, - "color": { - "mode": "thresholds" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "red" }, - { "value": 1, "color": "orange" }, - { "value": 3, "color": "yellow" }, - { "value": 4, "color": "green" } - ] - } - } - }, - "options": { - "graphMode": "area", - "colorMode": "background", - "justifyMode": "center", - "textMode": "value", - "reduceOptions": { - "values": false, - "calcs": ["lastNotNull"] - } - }, - "description": "Number of models generating predictions in the last 5 minutes. Expected: 4-6 models (DQN, MAMBA2, PPO, TFT, TLOB, Liquid)" - }, - - { - "id": 13, - "type": "row", - "title": "Performance Metrics", - "gridPos": { "x": 0, "y": 23, "w": 24, "h": 1 }, - "collapsed": false - }, - - { - "id": 14, - "type": "table", - "title": "Model Performance Summary", - "datasource": "${datasource}", - "gridPos": { "x": 0, "y": 24, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "100 * ml_model_accuracy{model=~\"$model_filter\"}", - "legendFormat": "__auto", - "refId": "Accuracy", - "format": "table", - "instant": true - }, - { - "expr": "ml_model_sharpe_ratio{model=~\"$model_filter\"}", - "legendFormat": "__auto", - "refId": "Sharpe", - "format": "table", - "instant": true - }, - { - "expr": "ml_model_avg_return_pct{model=~\"$model_filter\"}", - "legendFormat": "__auto", - "refId": "AvgReturn", - "format": "table", - "instant": true - }, - { - "expr": "ml_model_max_drawdown_pct{model=~\"$model_filter\"}", - "legendFormat": "__auto", - "refId": "MaxDrawdown", - "format": "table", - "instant": true - }, - { - "expr": "ml_model_total_trades{model=~\"$model_filter\"}", - "legendFormat": "__auto", - "refId": "TotalTrades", - "format": "table", - "instant": true - } - ], - "transformations": [ - { - "id": "merge", - "options": {} - }, - { - "id": "organize", - "options": { - "excludeByName": { - "Time": true, - "__name__": true, - "job": true, - "instance": true - }, - "indexByName": { - "model": 0, - "symbol": 1, - "Value #Accuracy": 2, - "Value #Sharpe": 3, - "Value #AvgReturn": 4, - "Value #MaxDrawdown": 5, - "Value #TotalTrades": 6 - }, - "renameByName": { - "model": "Model", - "symbol": "Symbol", - "Value #Accuracy": "Accuracy (%)", - "Value #Sharpe": "Sharpe Ratio", - "Value #AvgReturn": "Avg Return (%)", - "Value #MaxDrawdown": "Max Drawdown (%)", - "Value #TotalTrades": "Total Trades" - } - } - } - ], - "fieldConfig": { - "defaults": { - "custom": { - "align": "center", - "displayMode": "auto" - } - }, - "overrides": [ - { - "matcher": { "id": "byName", "options": "Accuracy (%)" }, - "properties": [ - { - "id": "custom.displayMode", - "value": "gradient-gauge" - }, - { - "id": "unit", - "value": "percent" - }, - { - "id": "decimals", - "value": 1 - }, - { - "id": "thresholds", - "value": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "red" }, - { "value": 55, "color": "yellow" }, - { "value": 65, "color": "green" } - ] - } - } - ] - }, - { - "matcher": { "id": "byName", "options": "Sharpe Ratio" }, - "properties": [ - { - "id": "custom.displayMode", - "value": "gradient-gauge" - }, - { - "id": "decimals", - "value": 2 - }, - { - "id": "thresholds", - "value": { - "mode": "absolute", - "steps": [ - { "value": -999, "color": "red" }, - { "value": 0, "color": "orange" }, - { "value": 1.0, "color": "yellow" }, - { "value": 1.5, "color": "green" } - ] - } - } - ] - }, - { - "matcher": { "id": "byName", "options": "Max Drawdown (%)" }, - "properties": [ - { - "id": "custom.displayMode", - "value": "gradient-gauge" - }, - { - "id": "unit", - "value": "percent" - }, - { - "id": "decimals", - "value": 2 - }, - { - "id": "thresholds", - "value": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "green" }, - { "value": 10, "color": "yellow" }, - { "value": 20, "color": "red" } - ] - } - } - ] - } - ] - }, - "options": { - "showHeader": true, - "sortBy": [ - { - "displayName": "Sharpe Ratio", - "desc": true - } - ] - }, - "description": "Comprehensive model performance metrics. Sorted by Sharpe ratio (risk-adjusted returns)." - }, - - { - "id": 15, - "type": "timeseries", - "title": "Inference Latency P99 (ms)", - "datasource": "${datasource}", - "gridPos": { "x": 12, "y": 24, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "histogram_quantile(0.99, rate(ml_model_inference_latency_bucket{model=~\"$model_filter\"}[1m]))", - "legendFormat": "{{ model }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "ms", - "custom": { - "drawStyle": "line", - "lineInterpolation": "smooth", - "lineWidth": 2, - "fillOpacity": 10, - "showPoints": "auto" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "green" }, - { "value": 50, "color": "yellow" }, - { "value": 100, "color": "red" } - ] - } - }, - "overrides": [ - { - "matcher": { "id": "byName", "options": "Target (100ms)" }, - "properties": [ - { - "id": "custom.lineStyle", - "value": { "fill": "dash", "dash": [10, 10] } - }, - { - "id": "color", - "value": { "fixedColor": "red", "mode": "fixed" } - }, - { - "id": "custom.lineWidth", - "value": 1 - } - ] - } - ] - }, - "options": { - "tooltip": { - "mode": "multi" - }, - "legend": { - "displayMode": "table", - "placement": "bottom", - "showLegend": true, - "calcs": ["last", "mean", "max"] - } - }, - "description": "99th percentile inference latency by model. Target: <100ms. High latency degrades signal quality." - }, - - { - "id": 16, - "type": "row", - "title": "Trading Volume & Risk", - "gridPos": { "x": 0, "y": 32, "w": 24, "h": 1 }, - "collapsed": false - }, - - { - "id": 17, - "type": "stat", - "title": "Total Predictions (24h)", - "datasource": "${datasource}", - "gridPos": { "x": 0, "y": 33, "w": 6, "h": 4 }, - "targets": [ - { - "expr": "sum(increase(ml_predictions_total{model=~\"$model_filter\"}[24h]))", - "legendFormat": "24h Predictions", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "short", - "decimals": 0, - "color": { - "mode": "thresholds" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "red" }, - { "value": 1000, "color": "yellow" }, - { "value": 10000, "color": "green" } - ] - } - } - }, - "options": { - "graphMode": "area", - "colorMode": "background", - "justifyMode": "center", - "textMode": "value", - "reduceOptions": { - "values": false, - "calcs": ["lastNotNull"] - } - }, - "description": "Total model predictions generated in the last 24 hours across all models" - }, - - { - "id": 18, - "type": "timeseries", - "title": "Prediction Volume by Model", - "datasource": "${datasource}", - "gridPos": { "x": 6, "y": 33, "w": 9, "h": 8 }, - "targets": [ - { - "expr": "rate(ml_predictions_total{model=~\"$model_filter\",symbol=~\"$symbol_filter\"}[5m]) * 60", - "legendFormat": "{{ model }} - {{ symbol }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "predictions/min", - "custom": { - "drawStyle": "line", - "lineInterpolation": "smooth", - "lineWidth": 2, - "fillOpacity": 20, - "gradientMode": "opacity", - "showPoints": "never", - "stacking": { - "mode": "normal", - "group": "A" - } - } - } - }, - "options": { - "tooltip": { - "mode": "multi", - "sort": "desc" - }, - "legend": { - "displayMode": "table", - "placement": "bottom", - "showLegend": true, - "calcs": ["last", "mean", "max"] - } - }, - "description": "Real-time prediction generation rate (stacked by model)" - }, - - { - "id": 19, - "type": "stat", - "title": "ML Order Flow (5m)", - "datasource": "${datasource}", - "gridPos": { "x": 15, "y": 33, "w": 9, "h": 4 }, - "targets": [ - { - "expr": "sum(rate(ml_orders_submitted_total{model=~\"$model_filter\"}[5m])) * 300", - "legendFormat": "Submitted", - "refId": "A" - }, - { - "expr": "sum(rate(ml_orders_filled_total{model=~\"$model_filter\"}[5m])) * 300", - "legendFormat": "Filled", - "refId": "B" - }, - { - "expr": "sum(rate(ml_orders_rejected_total{model=~\"$model_filter\"}[5m])) * 300", - "legendFormat": "Rejected", - "refId": "C" - } - ], - "fieldConfig": { - "defaults": { - "unit": "short", - "decimals": 0 - }, - "overrides": [ - { - "matcher": { "id": "byName", "options": "Submitted" }, - "properties": [ - { "id": "color", "value": { "fixedColor": "blue", "mode": "fixed" } } - ] - }, - { - "matcher": { "id": "byName", "options": "Filled" }, - "properties": [ - { "id": "color", "value": { "fixedColor": "green", "mode": "fixed" } } - ] - }, - { - "matcher": { "id": "byName", "options": "Rejected" }, - "properties": [ - { "id": "color", "value": { "fixedColor": "red", "mode": "fixed" } } - ] - } - ] - }, - "options": { - "graphMode": "area", - "colorMode": "background", - "justifyMode": "center", - "textMode": "value_and_name", - "reduceOptions": { - "values": false, - "calcs": ["lastNotNull"] - }, - "orientation": "horizontal" - }, - "description": "Order flow summary: submitted, filled, and rejected orders in the last 5 minutes" - }, - - { - "id": 20, - "type": "timeseries", - "title": "Model Error Rate", - "datasource": "${datasource}", - "gridPos": { "x": 15, "y": 37, "w": 9, "h": 4 }, - "targets": [ - { - "expr": "100 * rate(ml_prediction_errors_total{model=~\"$model_filter\"}[5m]) / rate(ml_predictions_total{model=~\"$model_filter\"}[5m])", - "legendFormat": "{{ model }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "percent", - "decimals": 2, - "custom": { - "drawStyle": "line", - "lineInterpolation": "smooth", - "lineWidth": 2, - "fillOpacity": 20, - "showPoints": "auto" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "green" }, - { "value": 2, "color": "yellow" }, - { "value": 5, "color": "red" } - ] - } - } - }, - "options": { - "tooltip": { - "mode": "multi" - }, - "legend": { - "displayMode": "list", - "placement": "bottom", - "showLegend": true - } - }, - "description": "Percentage of predictions that resulted in errors. Target: <2%, Alert: >5%" - }, - - { - "id": 21, - "type": "stat", - "title": "Risk Rejections (1h)", - "datasource": "${datasource}", - "gridPos": { "x": 0, "y": 37, "w": 6, "h": 4 }, - "targets": [ - { - "expr": "sum(increase(ml_orders_rejected_total{reason=\"risk_limit\",model=~\"$model_filter\"}[1h]))", - "legendFormat": "Risk Rejections", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "short", - "decimals": 0, - "color": { - "mode": "thresholds" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "green" }, - { "value": 10, "color": "yellow" }, - { "value": 50, "color": "red" } - ] - } - } - }, - "options": { - "graphMode": "area", - "colorMode": "background", - "justifyMode": "center", - "textMode": "value", - "reduceOptions": { - "values": false, - "calcs": ["lastNotNull"] - } - }, - "description": "Number of orders rejected due to risk limits in the last hour. High count indicates aggressive trading or insufficient risk limits." - }, - - { - "id": 22, - "type": "row", - "title": "System Health & Alerts", - "gridPos": { "x": 0, "y": 41, "w": 24, "h": 1 }, - "collapsed": false - }, - - { - "id": 23, - "type": "alertlist", - "title": "Active ML Alerts", - "datasource": "${datasource}", - "gridPos": { "x": 0, "y": 42, "w": 12, "h": 8 }, - "options": { - "showOptions": "current", - "maxItems": 20, - "sortOrder": 1, - "dashboardAlerts": false, - "alertName": "", - "dashboardTitle": "", - "tags": [], - "stateFilter": { - "firing": true, - "pending": true, - "noData": false, - "normal": false - }, - "alertInstanceLabelFilter": "{component=\"ml\"}" - }, - "description": "Currently firing ML-related alerts from Prometheus. Linked to alerts defined in ml_training_alerts.yml" - }, - - { - "id": 24, - "type": "table", - "title": "Recent Model Deployments", - "datasource": "${datasource}", - "gridPos": { "x": 12, "y": 42, "w": 12, "h": 8 }, - "targets": [ - { - "expr": "ml_model_deployment_timestamp{model=~\"$model_filter\"}", - "legendFormat": "__auto", - "refId": "A", - "format": "table", - "instant": true - }, - { - "expr": "ml_model_deployment_version{model=~\"$model_filter\"}", - "legendFormat": "__auto", - "refId": "B", - "format": "table", - "instant": true - } - ], - "transformations": [ - { - "id": "merge", - "options": {} - }, - { - "id": "organize", - "options": { - "excludeByName": { - "__name__": true, - "job": true, - "instance": true - }, - "indexByName": { - "model": 0, - "Time": 1, - "Value #B": 2 - }, - "renameByName": { - "model": "Model", - "Time": "Deployed At", - "Value #B": "Version" - } - } - } - ], - "fieldConfig": { - "defaults": { - "custom": { - "align": "left" - } - }, - "overrides": [ - { - "matcher": { "id": "byName", "options": "Deployed At" }, - "properties": [ - { - "id": "unit", - "value": "dateTimeFromNow" - } - ] - } - ] - }, - "options": { - "showHeader": true, - "sortBy": [ - { - "displayName": "Deployed At", - "desc": true - } - ] - }, - "description": "Recent model deployments with timestamps and version information" - }, - - { - "id": 25, - "type": "row", - "title": "GPU & Infrastructure", - "gridPos": { "x": 0, "y": 50, "w": 24, "h": 1 }, - "collapsed": false - }, - - { - "id": 26, - "type": "timeseries", - "title": "GPU Utilization", - "datasource": "${datasource}", - "gridPos": { "x": 0, "y": 51, "w": 12, "h": 6 }, - "targets": [ - { - "expr": "ml_gpu_utilization_percent{gpu_id=~\".*\"}", - "legendFormat": "GPU {{ gpu_id }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "percent", - "min": 0, - "max": 100, - "custom": { - "drawStyle": "line", - "lineInterpolation": "smooth", - "lineWidth": 2, - "fillOpacity": 30, - "showPoints": "auto" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "blue" }, - { "value": 30, "color": "green" }, - { "value": 80, "color": "yellow" }, - { "value": 95, "color": "red" } - ] - } - } - }, - "options": { - "tooltip": { - "mode": "multi" - }, - "legend": { - "displayMode": "list", - "placement": "bottom", - "showLegend": true - } - }, - "description": "GPU compute utilization. Target: 50-80% (efficient), <30% (underutilized), >95% (bottleneck)" - }, - - { - "id": 27, - "type": "timeseries", - "title": "GPU Memory Usage", - "datasource": "${datasource}", - "gridPos": { "x": 12, "y": 51, "w": 12, "h": 6 }, - "targets": [ - { - "expr": "100 * ml_gpu_memory_used_bytes{gpu_id=~\".*\"} / ml_gpu_memory_total_bytes{gpu_id=~\".*\"}", - "legendFormat": "GPU {{ gpu_id }}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "percent", - "min": 0, - "max": 100, - "custom": { - "drawStyle": "line", - "lineInterpolation": "smooth", - "lineWidth": 2, - "fillOpacity": 30, - "showPoints": "auto" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "green" }, - { "value": 70, "color": "yellow" }, - { "value": 90, "color": "red" } - ] - } - } - }, - "options": { - "tooltip": { - "mode": "multi" - }, - "legend": { - "displayMode": "list", - "placement": "bottom", - "showLegend": true - } - }, - "description": "GPU memory utilization. Target: <70%, Warning: >90%, Critical: >95% (OOM risk)" - }, - - { - "id": 28, - "type": "stat", - "title": "ML Service Status", - "datasource": "${datasource}", - "gridPos": { "x": 0, "y": 57, "w": 6, "h": 4 }, - "targets": [ - { - "expr": "up{job=\"ml_training_service\"}", - "legendFormat": "ML Training Service", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "mappings": [ - { - "type": "value", - "options": { - "0": { - "text": "DOWN", - "color": "red" - }, - "1": { - "text": "UP", - "color": "green" - } - } - } - ], - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "red" }, - { "value": 1, "color": "green" } - ] - } - } - }, - "options": { - "graphMode": "none", - "colorMode": "background", - "justifyMode": "center", - "textMode": "value", - "reduceOptions": { - "values": false, - "calcs": ["lastNotNull"] - } - }, - "description": "ML Training Service health status from Prometheus scrape" - }, - - { - "id": 29, - "type": "timeseries", - "title": "Feature Extraction Latency P95", - "datasource": "${datasource}", - "gridPos": { "x": 6, "y": 57, "w": 9, "h": 4 }, - "targets": [ - { - "expr": "histogram_quantile(0.95, rate(ml_feature_extraction_seconds_bucket[5m]))", - "legendFormat": "p95", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "s", - "custom": { - "drawStyle": "line", - "lineInterpolation": "smooth", - "lineWidth": 2, - "fillOpacity": 20 - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "green" }, - { "value": 3, "color": "yellow" }, - { "value": 5, "color": "red" } - ] - } - } - }, - "description": "Feature extraction latency (p95). Target: <3s, Alert: >5s" - }, - - { - "id": 30, - "type": "stat", - "title": "Model Cache Hit Rate", - "datasource": "${datasource}", - "gridPos": { "x": 15, "y": 57, "w": 9, "h": 4 }, - "targets": [ - { - "expr": "100 * rate(ml_model_cache_hits[5m]) / (rate(ml_model_cache_hits[5m]) + rate(ml_model_cache_misses[5m]))", - "legendFormat": "Cache Hit Rate", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "percent", - "decimals": 1, - "color": { - "mode": "thresholds" - }, - "thresholds": { - "mode": "absolute", - "steps": [ - { "value": 0, "color": "red" }, - { "value": 70, "color": "yellow" }, - { "value": 85, "color": "green" } - ] - } - } - }, - "options": { - "graphMode": "area", - "colorMode": "background", - "justifyMode": "center", - "textMode": "value", - "reduceOptions": { - "values": false, - "calcs": ["lastNotNull"] - } - }, - "description": "Model cache efficiency. Target: >85%, Low: <70% (increased loading latency)" - } - ] - } -} diff --git a/monitoring/grafana/ml_training_dashboard.json b/monitoring/grafana/ml_training_dashboard.json deleted file mode 100644 index dbe4a973d..000000000 --- a/monitoring/grafana/ml_training_dashboard.json +++ /dev/null @@ -1,613 +0,0 @@ -{ - "dashboard": { - "id": null, - "uid": "ml-training-monitoring", - "title": "ML Training Service - Production Monitoring", - "tags": ["ml", "training", "gpu", "monitoring"], - "timezone": "browser", - "schemaVersion": 16, - "version": 0, - "refresh": "30s", - "time": { - "from": "now-6h", - "to": "now" - }, - "panels": [ - { - "id": 1, - "title": "Training Jobs by Status", - "type": "stat", - "gridPos": { - "x": 0, - "y": 0, - "w": 6, - "h": 4 - }, - "targets": [ - { - "expr": "ml_training_jobs_by_status", - "legendFormat": "{{status}}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - } - ] - } - } - } - }, - { - "id": 2, - "title": "GPU Memory Usage", - "type": "gauge", - "gridPos": { - "x": 6, - "y": 0, - "w": 6, - "h": 4 - }, - "targets": [ - { - "expr": "100 * ml_gpu_memory_used_bytes / ml_gpu_memory_total_bytes", - "legendFormat": "GPU {{gpu_id}}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "min": 0, - "max": 100, - "unit": "percent", - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": 0 - }, - { - "color": "yellow", - "value": 80 - }, - { - "color": "red", - "value": 90 - } - ] - } - } - } - }, - { - "id": 3, - "title": "GPU Temperature", - "type": "gauge", - "gridPos": { - "x": 12, - "y": 0, - "w": 6, - "h": 4 - }, - "targets": [ - { - "expr": "ml_gpu_temperature_celsius", - "legendFormat": "GPU {{gpu_id}}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "min": 0, - "max": 100, - "unit": "celsius", - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": 0 - }, - { - "color": "yellow", - "value": 75 - }, - { - "color": "red", - "value": 85 - } - ] - } - } - } - }, - { - "id": 4, - "title": "GPU Utilization", - "type": "gauge", - "gridPos": { - "x": 18, - "y": 0, - "w": 6, - "h": 4 - }, - "targets": [ - { - "expr": "ml_gpu_utilization_percent", - "legendFormat": "GPU {{gpu_id}}", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "min": 0, - "max": 100, - "unit": "percent", - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "red", - "value": 0 - }, - { - "color": "yellow", - "value": 30 - }, - { - "color": "green", - "value": 60 - } - ] - } - } - } - }, - { - "id": 5, - "title": "Training Loss (All Models)", - "type": "graph", - "gridPos": { - "x": 0, - "y": 4, - "w": 12, - "h": 8 - }, - "targets": [ - { - "expr": "ml_training_loss", - "legendFormat": "{{model_type}} - {{job_id}}", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "short", - "logBase": 1, - "show": true - } - ], - "xaxis": { - "show": true, - "mode": "time" - } - }, - { - "id": 6, - "title": "Validation Loss (All Models)", - "type": "graph", - "gridPos": { - "x": 12, - "y": 4, - "w": 12, - "h": 8 - }, - "targets": [ - { - "expr": "ml_training_validation_loss", - "legendFormat": "{{model_type}} - {{job_id}}", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "short", - "logBase": 1, - "show": true - } - ], - "xaxis": { - "show": true, - "mode": "time" - } - }, - { - "id": 7, - "title": "Training Speed (Epochs/sec)", - "type": "graph", - "gridPos": { - "x": 0, - "y": 12, - "w": 12, - "h": 6 - }, - "targets": [ - { - "expr": "ml_training_epochs_per_second", - "legendFormat": "{{model_type}} - {{job_id}}", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "short", - "logBase": 1, - "show": true, - "min": 0 - } - ], - "alert": { - "conditions": [ - { - "evaluator": { - "params": [0.1], - "type": "lt" - }, - "operator": { - "type": "and" - }, - "query": { - "params": ["A", "5m", "now"] - }, - "reducer": { - "params": [], - "type": "avg" - }, - "type": "query" - } - ], - "executionErrorState": "alerting", - "for": "5m", - "frequency": "1m", - "name": "Training Speed Degraded", - "noDataState": "no_data", - "notifications": [] - } - }, - { - "id": 8, - "title": "Training Progress (%)", - "type": "graph", - "gridPos": { - "x": 12, - "y": 12, - "w": 12, - "h": 6 - }, - "targets": [ - { - "expr": "ml_training_progress_percent", - "legendFormat": "{{model_type}} - {{job_id}}", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "percent", - "logBase": 1, - "show": true, - "min": 0, - "max": 100 - } - ] - }, - { - "id": 9, - "title": "Checkpoint Save Duration (P95)", - "type": "graph", - "gridPos": { - "x": 0, - "y": 18, - "w": 12, - "h": 6 - }, - "targets": [ - { - "expr": "histogram_quantile(0.95, rate(ml_checkpoint_save_duration_seconds_bucket[5m]))", - "legendFormat": "{{model_type}} - P95", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "s", - "logBase": 1, - "show": true - } - ] - }, - { - "id": 10, - "title": "NaN Detection Events", - "type": "graph", - "gridPos": { - "x": 12, - "y": 18, - "w": 12, - "h": 6 - }, - "targets": [ - { - "expr": "rate(ml_training_nan_count[5m])", - "legendFormat": "{{model_type}} - {{tensor_type}}", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "short", - "logBase": 1, - "show": true, - "min": 0 - } - ] - }, - { - "id": 11, - "title": "Model Accuracy (Validation)", - "type": "graph", - "gridPos": { - "x": 0, - "y": 24, - "w": 12, - "h": 6 - }, - "targets": [ - { - "expr": "ml_model_accuracy", - "legendFormat": "{{model_type}}", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "percentunit", - "logBase": 1, - "show": true, - "min": 0, - "max": 1 - } - ] - }, - { - "id": 12, - "title": "Data Loading Duration (P95)", - "type": "graph", - "gridPos": { - "x": 12, - "y": 24, - "w": 12, - "h": 6 - }, - "targets": [ - { - "expr": "histogram_quantile(0.95, rate(ml_training_data_loading_seconds_bucket[5m]))", - "legendFormat": "{{model_type}} - P95", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "s", - "logBase": 1, - "show": true - } - ] - }, - { - "id": 13, - "title": "Training Failures by Type", - "type": "piechart", - "gridPos": { - "x": 0, - "y": 30, - "w": 8, - "h": 6 - }, - "targets": [ - { - "expr": "sum by (error_type) (rate(ml_training_failures_total[1h]))", - "legendFormat": "{{error_type}}", - "refId": "A" - } - ] - }, - { - "id": 14, - "title": "S3 Request Errors", - "type": "stat", - "gridPos": { - "x": 8, - "y": 30, - "w": 8, - "h": 6 - }, - "targets": [ - { - "expr": "rate(ml_s3_request_errors_total[5m])", - "legendFormat": "Errors/sec", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "yellow", - "value": 0.5 - }, - { - "color": "red", - "value": 1 - } - ] - } - } - } - }, - { - "id": 15, - "title": "Model Storage Usage", - "type": "graph", - "gridPos": { - "x": 16, - "y": 30, - "w": 8, - "h": 6 - }, - "targets": [ - { - "expr": "100 * ml_model_storage_used_bytes / ml_model_storage_limit_bytes", - "legendFormat": "Storage Usage %", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "percent", - "logBase": 1, - "show": true, - "min": 0, - "max": 100 - } - ] - }, - { - "id": 16, - "title": "Data Drift Score (All Features)", - "type": "graph", - "gridPos": { - "x": 0, - "y": 36, - "w": 12, - "h": 6 - }, - "targets": [ - { - "expr": "ml_model_drift_score", - "legendFormat": "{{feature}}", - "refId": "A" - } - ], - "yaxes": [ - { - "format": "short", - "logBase": 1, - "show": true, - "min": 0, - "max": 1 - } - ], - "alert": { - "conditions": [ - { - "evaluator": { - "params": [0.15], - "type": "gt" - }, - "operator": { - "type": "and" - }, - "query": { - "params": ["A", "5m", "now"] - }, - "reducer": { - "params": [], - "type": "avg" - }, - "type": "query" - } - ], - "executionErrorState": "alerting", - "for": "5m", - "frequency": "1m", - "name": "Data Drift Detected", - "noDataState": "no_data", - "notifications": [] - } - }, - { - "id": 17, - "title": "Cost Tracking (Monthly Projection)", - "type": "stat", - "gridPos": { - "x": 12, - "y": 36, - "w": 12, - "h": 6 - }, - "targets": [ - { - "expr": "ml_monthly_cost_projection_dollars", - "legendFormat": "Projected Monthly Cost", - "refId": "A" - } - ], - "fieldConfig": { - "defaults": { - "unit": "currencyUSD", - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "yellow", - "value": 800 - }, - { - "color": "red", - "value": 1000 - } - ] - } - } - } - } - ], - "annotations": { - "list": [ - { - "datasource": "Prometheus", - "enable": true, - "expr": "ALERTS{alertname=~\".*Training.*|.*GPU.*\"}", - "iconColor": "red", - "name": "Training Alerts", - "step": "60s", - "tagKeys": "alertname,severity", - "titleFormat": "Alert: {{alertname}}", - "type": "tags" - } - ] - } - } -} diff --git a/monitoring/latency_tracker.rs b/monitoring/latency_tracker.rs deleted file mode 100644 index 439105553..000000000 --- a/monitoring/latency_tracker.rs +++ /dev/null @@ -1,416 +0,0 @@ -#![warn(missing_docs)] -//! Comprehensive P99 latency tracking for critical HFT operations -//! -//! This module provides high-precision latency measurement and percentile tracking -//! for all critical trading operations in the Foxhunt HFT system. - -use hdrhistogram::Histogram; -use std::sync::{Arc, RwLock}; -use std::time::Instant; -use std::collections::HashMap; - -/// High-precision latency tracker using HDR histogram for accurate percentile calculations -pub struct LatencyTracker { - histogram: RwLock>, - operation_name: String, -} - -impl LatencyTracker { - /// Create a new latency tracker for a specific operation - pub fn new(operation_name: &str) -> Self { - Self { - histogram: RwLock::new( - Histogram::new_with_bounds(1, 60_000_000_000, 3).unwrap() // 1ns to 60s - ), - operation_name: operation_name.to_string(), - } - } - - /// Record a latency measurement in nanoseconds - pub fn record(&self, nanos: u64) { - if let Ok(mut hist) = self.histogram.write() { - let _ = hist.record(nanos); - } - } - - /// Get the 99th percentile latency in nanoseconds - pub fn get_p99(&self) -> u64 { - self.histogram - .read() - .map(|hist| hist.value_at_percentile(99.0)) - .unwrap_or(0) - } - - /// Get the 50th percentile (median) latency in nanoseconds - pub fn get_p50(&self) -> u64 { - self.histogram - .read() - .map(|hist| hist.value_at_percentile(50.0)) - .unwrap_or(0) - } - - /// Get the 95th percentile latency in nanoseconds - pub fn get_p95(&self) -> u64 { - self.histogram - .read() - .map(|hist| hist.value_at_percentile(95.0)) - .unwrap_or(0) - } - - /// Get the 99.9th percentile latency in nanoseconds - pub fn get_p999(&self) -> u64 { - self.histogram - .read() - .map(|hist| hist.value_at_percentile(99.9)) - .unwrap_or(0) - } - - /// Get the maximum recorded latency in nanoseconds - pub fn get_max(&self) -> u64 { - self.histogram - .read() - .map(|hist| hist.max()) - .unwrap_or(0) - } - - /// Get the minimum recorded latency in nanoseconds - pub fn get_min(&self) -> u64 { - self.histogram - .read() - .map(|hist| hist.min()) - .unwrap_or(0) - } - - /// Get the mean latency in nanoseconds - pub fn get_mean(&self) -> f64 { - self.histogram - .read() - .map(|hist| hist.mean()) - .unwrap_or(0.0) - } - - /// Get the total number of recorded samples - pub fn get_count(&self) -> u64 { - self.histogram - .read() - .map(|hist| hist.len()) - .unwrap_or(0) - } - - /// Reset all recorded latencies - pub fn reset(&self) { - if let Ok(mut hist) = self.histogram.write() { - hist.reset(); - } - } - - /// Get comprehensive latency statistics - pub fn get_stats(&self) -> LatencyStats { - if let Ok(hist) = self.histogram.read() { - LatencyStats { - operation: self.operation_name.clone(), - count: hist.len(), - min: hist.min(), - max: hist.max(), - mean: hist.mean(), - p50: hist.value_at_percentile(50.0), - p95: hist.value_at_percentile(95.0), - p99: hist.value_at_percentile(99.0), - p999: hist.value_at_percentile(99.9), - } - } else { - LatencyStats::default_for_operation(&self.operation_name) - } - } -} - -/// Comprehensive latency statistics for an operation -#[derive(Debug, Clone)] -pub struct LatencyStats { - /// Operation name - pub operation: String, - /// Total number of samples - pub count: u64, - /// Minimum latency (nanoseconds) - pub min: u64, - /// Maximum latency (nanoseconds) - pub max: u64, - /// Mean latency (nanoseconds) - pub mean: f64, - /// 50th percentile latency (nanoseconds) - pub p50: u64, - /// 95th percentile latency (nanoseconds) - pub p95: u64, - /// 99th percentile latency (nanoseconds) - pub p99: u64, - /// 99.9th percentile latency (nanoseconds) - pub p999: u64, -} - -impl LatencyStats { - fn default_for_operation(operation: &str) -> Self { - Self { - operation: operation.to_string(), - count: 0, - min: 0, - max: 0, - mean: 0.0, - p50: 0, - p95: 0, - p99: 0, - p999: 0, - } - } - - /// Convert nanoseconds to microseconds - pub fn p99_micros(&self) -> f64 { - self.p99 as f64 / 1_000.0 - } - - /// Convert nanoseconds to milliseconds - pub fn p99_millis(&self) -> f64 { - self.p99 as f64 / 1_000_000.0 - } - - /// Check if P99 latency exceeds threshold (in nanoseconds) - pub fn exceeds_p99_threshold(&self, threshold_nanos: u64) -> bool { - self.p99 > threshold_nanos - } - - /// Format latency for human-readable output - pub fn format_p99(&self) -> String { - if self.p99 < 1_000 { - format!("{}ns", self.p99) - } else if self.p99 < 1_000_000 { - format!("{:.1}ΞΌs", self.p99 as f64 / 1_000.0) - } else if self.p99 < 1_000_000_000 { - format!("{:.1}ms", self.p99 as f64 / 1_000_000.0) - } else { - format!("{:.1}s", self.p99 as f64 / 1_000_000_000.0) - } - } -} - -/// Global registry for all latency trackers in the system -pub struct LatencyRegistry { - trackers: RwLock>>, -} - -impl LatencyRegistry { - /// Create a new latency registry - pub fn new() -> Self { - Self { - trackers: RwLock::new(HashMap::new()), - } - } - - /// Get or create a latency tracker for an operation - pub fn get_tracker(&self, operation: &str) -> Arc { - { - let trackers = self.trackers.read().unwrap(); - if let Some(tracker) = trackers.get(operation) { - return Arc::clone(tracker); - } - } - - let mut trackers = self.trackers.write().unwrap(); - let tracker = Arc::new(LatencyTracker::new(operation)); - trackers.insert(operation.to_string(), Arc::clone(&tracker)); - tracker - } - - /// Get all registered tracker statistics - pub fn get_all_stats(&self) -> Vec { - let trackers = self.trackers.read().unwrap(); - trackers - .values() - .map(|tracker| tracker.get_stats()) - .collect() - } - - /// Reset all trackers - pub fn reset_all(&self) { - let trackers = self.trackers.read().unwrap(); - for tracker in trackers.values() { - tracker.reset(); - } - } -} - -impl Default for LatencyRegistry { - fn default() -> Self { - Self::new() - } -} - -/// RAII timer for automatic latency measurement -pub struct LatencyTimer { - tracker: Arc, - start: Instant, -} - -impl LatencyTimer { - /// Start timing an operation - pub fn start(tracker: Arc) -> Self { - Self { - tracker, - start: Instant::now(), - } - } - - /// Manually record the elapsed time (useful for early recording) - pub fn record_now(&self) { - let elapsed = self.start.elapsed().as_nanos() as u64; - self.tracker.record(elapsed); - } -} - -impl Drop for LatencyTimer { - fn drop(&mut self) { - let elapsed = self.start.elapsed().as_nanos() as u64; - self.tracker.record(elapsed); - } -} - -/// Global latency registry instance -static GLOBAL_REGISTRY: std::sync::OnceLock = std::sync::OnceLock::new(); - -/// Get the global latency registry -pub fn global_registry() -> &'static LatencyRegistry { - GLOBAL_REGISTRY.get_or_init(|| LatencyRegistry::new()) -} - -/// Convenience macro for timing operations -#[macro_export] -macro_rules! time_operation { - ($operation:expr, $code:block) => {{ - let tracker = $crate::monitoring::latency_tracker::global_registry() - .get_tracker($operation); - let _timer = $crate::monitoring::latency_tracker::LatencyTimer::start(tracker); - $code - }}; -} - -/// Convenience function to record a single latency measurement -pub fn record_latency(operation: &str, nanos: u64) { - let tracker = global_registry().get_tracker(operation); - tracker.record(nanos); -} - -/// Critical HFT operation names for consistent tracking -pub mod operations { - /// Order placement latency - pub const ORDER_PLACEMENT: &str = "order_placement"; - /// Order cancellation latency - pub const ORDER_CANCELLATION: &str = "order_cancellation"; - /// Market data processing latency - pub const MARKET_DATA_PROCESSING: &str = "market_data_processing"; - /// Risk check latency - pub const RISK_CHECK: &str = "risk_check"; - /// Position update latency - pub const POSITION_UPDATE: &str = "position_update"; - /// Trade execution latency - pub const TRADE_EXECUTION: &str = "trade_execution"; - /// Signal generation latency - pub const SIGNAL_GENERATION: &str = "signal_generation"; - /// Portfolio rebalancing latency - pub const PORTFOLIO_REBALANCING: &str = "portfolio_rebalancing"; - /// Database write latency - pub const DATABASE_WRITE: &str = "database_write"; - /// Database read latency - pub const DATABASE_READ: &str = "database_read"; - /// Message queue publish latency - pub const MESSAGE_PUBLISH: &str = "message_publish"; - /// Message queue consume latency - pub const MESSAGE_CONSUME: &str = "message_consume"; - /// Broker API call latency - pub const BROKER_API_CALL: &str = "broker_api_call"; - /// AI model inference latency - pub const AI_MODEL_INFERENCE: &str = "ai_model_inference"; - /// End-to-end trade latency - pub const END_TO_END_TRADE: &str = "end_to_end_trade"; -} - -#[cfg(test)] -mod tests { - use super::*; - use std::thread; - use std::time::Duration; - - #[test] - fn test_latency_tracker_basic() { - let tracker = LatencyTracker::new("test_operation"); - - // Record some test latencies - tracker.record(1000); // 1ΞΌs - tracker.record(2000); // 2ΞΌs - tracker.record(5000); // 5ΞΌs - tracker.record(10000); // 10ΞΌs - - assert_eq!(tracker.get_count(), 4); - assert!(tracker.get_p50() > 0); - assert!(tracker.get_p99() > 0); - assert!(tracker.get_max() >= 10000); - assert!(tracker.get_min() <= 1000); - } - - #[test] - fn test_latency_timer() { - let tracker = Arc::new(LatencyTracker::new("timer_test")); - let tracker_clone = Arc::clone(&tracker); - - { - let _timer = LatencyTimer::start(tracker_clone); - thread::sleep(Duration::from_micros(100)); - } - - assert_eq!(tracker.get_count(), 1); - assert!(tracker.get_p99() > 50_000); // Should be > 50ΞΌs - } - - #[test] - fn test_latency_registry() { - let registry = LatencyRegistry::new(); - - let tracker1 = registry.get_tracker("operation1"); - let tracker2 = registry.get_tracker("operation2"); - let tracker1_again = registry.get_tracker("operation1"); - - // Should return the same tracker for the same operation - assert!(Arc::ptr_eq(&tracker1, &tracker1_again)); - - tracker1.record(1000); - tracker2.record(2000); - - let stats = registry.get_all_stats(); - assert_eq!(stats.len(), 2); - } - - #[test] - fn test_latency_stats_formatting() { - let tracker = LatencyTracker::new("format_test"); - - tracker.record(500); // 500ns - tracker.record(1500); // 1.5ΞΌs - tracker.record(1_500_000); // 1.5ms - - let stats = tracker.get_stats(); - let formatted = stats.format_p99(); - - // Should format appropriately based on magnitude - assert!(!formatted.is_empty()); - } - - #[test] - fn test_time_operation_macro() { - let result = time_operation!("macro_test", { - thread::sleep(Duration::from_micros(10)); - 42 - }); - - assert_eq!(result, 42); - - let tracker = global_registry().get_tracker("macro_test"); - assert_eq!(tracker.get_count(), 1); - } -} diff --git a/monitoring/metrics.rs b/monitoring/metrics.rs deleted file mode 100644 index feec5d716..000000000 --- a/monitoring/metrics.rs +++ /dev/null @@ -1,482 +0,0 @@ -use prometheus::{ - Counter, Histogram, Gauge, IntCounter, IntGauge, - register_counter, register_histogram, register_gauge, - register_int_counter, register_int_gauge, - Opts, HistogramOpts, Registry, Encoder, TextEncoder -}; -use std::collections::HashMap; -use std::sync::Arc; -use lazy_static::lazy_static; -use tracing::{error, info, warn}; - -/// Core HFT trading metrics for Foxhunt system -/// -/// Optimized for high-frequency data collection with minimal latency impact - -lazy_static! { - // Order processing metrics - static ref ORDER_COUNTER: Counter = register_counter!( - "foxhunt_orders_total", - "Total orders processed by the trading system" - ).expect("Failed to register orders counter"); - - static ref ORDER_FILL_COUNTER: Counter = register_counter!( - "foxhunt_order_fills_total", - "Total order fills executed" - ).expect("Failed to register order fills counter"); - - static ref ORDER_REJECTION_COUNTER: Counter = register_counter!( - "foxhunt_order_rejections_total", - "Total order rejections" - ).expect("Failed to register order rejections counter"); - - // Latency metrics - critical for HFT performance - static ref LATENCY_HISTOGRAM: Histogram = register_histogram!( - HistogramOpts::new( - "foxhunt_latency_microseconds", - "Latency distribution in microseconds" - ).buckets(vec![1.0, 5.0, 10.0, 25.0, 50.0, 100.0, 250.0, 500.0, 1000.0]) - ).expect("Failed to register latency histogram"); - - static ref ORDER_PROCESSING_LATENCY: Histogram = register_histogram!( - HistogramOpts::new( - "foxhunt_order_processing_latency_microseconds", - "Order processing latency from receipt to exchange submission" - ).buckets(vec![1.0, 5.0, 10.0, 25.0, 50.0, 100.0, 250.0, 500.0]) - ).expect("Failed to register order processing latency histogram"); - - static ref MARKET_DATA_LATENCY: Histogram = register_histogram!( - HistogramOpts::new( - "foxhunt_market_data_latency_microseconds", - "Market data processing latency" - ).buckets(vec![0.1, 0.5, 1.0, 5.0, 10.0, 25.0, 50.0, 100.0]) - ).expect("Failed to register market data latency histogram"); - - // Risk metrics - static ref RISK_BREACH_COUNTER: Counter = register_counter!( - "foxhunt_risk_breaches_total", - "Total risk limit breaches" - ).expect("Failed to register risk breaches counter"); - - static ref POSITION_VALUE_GAUGE: Gauge = register_gauge!( - "foxhunt_position_value_usd", - "Current total position value in USD" - ).expect("Failed to register position value gauge"); - - static ref VAR_GAUGE: Gauge = register_gauge!( - "foxhunt_var_usd", - "Current Value at Risk in USD" - ).expect("Failed to register VaR gauge"); - - // Trading engine metrics - static ref ACTIVE_ORDERS_GAUGE: IntGauge = register_int_gauge!( - "foxhunt_active_orders", - "Number of currently active orders" - ).expect("Failed to register active orders gauge"); - - static ref TRADING_SESSIONS_GAUGE: IntGauge = register_int_gauge!( - "foxhunt_trading_sessions_active", - "Number of active trading sessions" - ).expect("Failed to register trading sessions gauge"); - - // Market data metrics - static ref MARKET_DATA_MESSAGES_COUNTER: Counter = register_counter!( - "foxhunt_market_data_messages_total", - "Total market data messages received" - ).expect("Failed to register market data messages counter"); - - static ref MARKET_DATA_DROPS_COUNTER: Counter = register_counter!( - "foxhunt_market_data_drops_total", - "Total market data messages dropped" - ).expect("Failed to register market data drops counter"); - - // AI/ML metrics - static ref ML_PREDICTIONS_COUNTER: Counter = register_counter!( - "foxhunt_ml_predictions_total", - "Total ML model predictions generated" - ).expect("Failed to register ML predictions counter"); - - static ref ML_MODEL_ACCURACY_GAUGE: Gauge = register_gauge!( - "foxhunt_ml_model_accuracy", - "Current ML model accuracy percentage" - ).expect("Failed to register ML model accuracy gauge"); - - // System health metrics - static ref CPU_USAGE_GAUGE: Gauge = register_gauge!( - "foxhunt_cpu_usage_percent", - "Current CPU usage percentage" - ).expect("Failed to register CPU usage gauge"); - - static ref MEMORY_USAGE_GAUGE: Gauge = register_gauge!( - "foxhunt_memory_usage_bytes", - "Current memory usage in bytes" - ).expect("Failed to register memory usage gauge"); - - // Broker connectivity metrics - static ref BROKER_CONNECTIONS_GAUGE: IntGauge = register_int_gauge!( - "foxhunt_broker_connections", - "Number of active broker connections" - ).expect("Failed to register broker connections gauge"); - - static ref BROKER_DISCONNECTS_COUNTER: Counter = register_counter!( - "foxhunt_broker_disconnects_total", - "Total broker disconnection events" - ).expect("Failed to register broker disconnects counter"); - - // Performance metrics - static ref THROUGHPUT_GAUGE: Gauge = register_gauge!( - "foxhunt_throughput_ops_per_second", - "Current system throughput in operations per second" - ).expect("Failed to register throughput gauge"); -} - -/// Metrics collector for the Foxhunt HFT system -#[derive(Debug)] -pub struct FoxhuntMetrics { - registry: Arc, - custom_counters: HashMap, - custom_histograms: HashMap, - custom_gauges: HashMap, -} - -impl Default for FoxhuntMetrics { - fn default() -> Self { - Self::new() - } -} - -impl FoxhuntMetrics { - /// Create a new metrics collector instance - pub fn new() -> Self { - Self { - registry: Arc::new(Registry::new()), - custom_counters: HashMap::new(), - custom_histograms: HashMap::new(), - custom_gauges: HashMap::new(), - } - } - - /// Record an order being processed - #[inline(always)] - pub fn record_order() { - ORDER_COUNTER.inc(); - } - - /// Record an order fill - #[inline(always)] - pub fn record_order_fill() { - ORDER_FILL_COUNTER.inc(); - } - - /// Record an order rejection - #[inline(always)] - pub fn record_order_rejection() { - ORDER_REJECTION_COUNTER.inc(); - } - - /// Record latency measurement in microseconds - #[inline(always)] - pub fn record_latency(latency_us: f64) { - LATENCY_HISTOGRAM.observe(latency_us); - } - - /// Record order processing latency in microseconds - #[inline(always)] - pub fn record_order_processing_latency(latency_us: f64) { - ORDER_PROCESSING_LATENCY.observe(latency_us); - } - - /// Record market data latency in microseconds - #[inline(always)] - pub fn record_market_data_latency(latency_us: f64) { - MARKET_DATA_LATENCY.observe(latency_us); - } - - /// Record a risk breach event - #[inline(always)] - pub fn record_risk_breach() { - RISK_BREACH_COUNTER.inc(); - warn!("Risk breach recorded in metrics"); - } - - /// Update current position value - #[inline(always)] - pub fn update_position_value(value_usd: f64) { - POSITION_VALUE_GAUGE.set(value_usd); - } - - /// Update Value at Risk - #[inline(always)] - pub fn update_var(var_usd: f64) { - VAR_GAUGE.set(var_usd); - } - - /// Update active orders count - #[inline(always)] - pub fn update_active_orders(count: i64) { - ACTIVE_ORDERS_GAUGE.set(count); - } - - /// Update trading sessions count - #[inline(always)] - pub fn update_trading_sessions(count: i64) { - TRADING_SESSIONS_GAUGE.set(count); - } - - /// Record market data message received - #[inline(always)] - pub fn record_market_data_message() { - MARKET_DATA_MESSAGES_COUNTER.inc(); - } - - /// Record market data message dropped - #[inline(always)] - pub fn record_market_data_drop() { - MARKET_DATA_DROPS_COUNTER.inc(); - } - - /// Record ML prediction generated - #[inline(always)] - pub fn record_ml_prediction() { - ML_PREDICTIONS_COUNTER.inc(); - } - - /// Update ML model accuracy - #[inline(always)] - pub fn update_ml_accuracy(accuracy: f64) { - ML_MODEL_ACCURACY_GAUGE.set(accuracy); - } - - /// Update CPU usage percentage - #[inline(always)] - pub fn update_cpu_usage(percentage: f64) { - CPU_USAGE_GAUGE.set(percentage); - } - - /// Update memory usage in bytes - #[inline(always)] - pub fn update_memory_usage(bytes: f64) { - MEMORY_USAGE_GAUGE.set(bytes); - } - - /// Update broker connections count - #[inline(always)] - pub fn update_broker_connections(count: i64) { - BROKER_CONNECTIONS_GAUGE.set(count); - } - - /// Record broker disconnection - #[inline(always)] - pub fn record_broker_disconnect() { - BROKER_DISCONNECTS_COUNTER.inc(); - } - - /// Update system throughput - #[inline(always)] - pub fn update_throughput(ops_per_second: f64) { - THROUGHPUT_GAUGE.set(ops_per_second); - } - - /// Get metrics in Prometheus text format - pub fn export_metrics() -> Result> { - let encoder = TextEncoder::new(); - let metric_families = prometheus::gather(); - let mut buffer = Vec::new(); - encoder.encode(&metric_families, &mut buffer)?; - Ok(String::from_utf8(buffer)?) - } - - /// Create a custom counter metric - pub fn create_custom_counter(&mut self, name: &str, help: &str) -> Result<(), Box> { - let counter = Counter::new(name, help)?; - self.registry.register(Box::new(counter.clone()))?; - self.custom_counters.insert(name.to_string(), counter); - Ok(()) - } - - /// Increment a custom counter - pub fn increment_custom_counter(&self, name: &str) { - if let Some(counter) = self.custom_counters.get(name) { - counter.inc(); - } else { - error!("Custom counter '{}' not found", name); - } - } - - /// Reset all metrics (for testing purposes) - #[cfg(test)] - pub fn reset_all() { - // Reset all static metrics to zero - ORDER_COUNTER.reset(); - ORDER_FILL_COUNTER.reset(); - ORDER_REJECTION_COUNTER.reset(); - RISK_BREACH_COUNTER.reset(); - MARKET_DATA_MESSAGES_COUNTER.reset(); - MARKET_DATA_DROPS_COUNTER.reset(); - ML_PREDICTIONS_COUNTER.reset(); - BROKER_DISCONNECTS_COUNTER.reset(); - - // Reset gauges to zero - POSITION_VALUE_GAUGE.set(0.0); - VAR_GAUGE.set(0.0); - ACTIVE_ORDERS_GAUGE.set(0); - TRADING_SESSIONS_GAUGE.set(0); - ML_MODEL_ACCURACY_GAUGE.set(0.0); - CPU_USAGE_GAUGE.set(0.0); - MEMORY_USAGE_GAUGE.set(0.0); - BROKER_CONNECTIONS_GAUGE.set(0); - THROUGHPUT_GAUGE.set(0.0); - } - - /// Log current metrics summary - pub fn log_metrics_summary() { - info!( - "Metrics Summary - Orders: {}, Fills: {}, Rejections: {}, Active Orders: {}, Risk Breaches: {}", - ORDER_COUNTER.get(), - ORDER_FILL_COUNTER.get(), - ORDER_REJECTION_COUNTER.get(), - ACTIVE_ORDERS_GAUGE.get(), - RISK_BREACH_COUNTER.get() - ); - } -} - -/// Convenience functions for direct metric recording -/// -/// These are optimized for hot path usage with minimal overhead - -/// Record order with timing -#[inline(always)] -pub fn record_order() { - FoxhuntMetrics::record_order(); -} - -/// Record order fill -#[inline(always)] -pub fn record_order_fill() { - FoxhuntMetrics::record_order_fill(); -} - -/// Record order rejection -#[inline(always)] -pub fn record_order_rejection() { - FoxhuntMetrics::record_order_rejection(); -} - -/// Record latency measurement -#[inline(always)] -pub fn record_latency(latency_us: f64) { - FoxhuntMetrics::record_latency(latency_us); -} - -/// Record order processing latency -#[inline(always)] -pub fn record_order_processing_latency(latency_us: f64) { - FoxhuntMetrics::record_order_processing_latency(latency_us); -} - -/// Record market data latency -#[inline(always)] -pub fn record_market_data_latency(latency_us: f64) { - FoxhuntMetrics::record_market_data_latency(latency_us); -} - -/// Record risk breach -#[inline(always)] -pub fn record_risk_breach() { - FoxhuntMetrics::record_risk_breach(); -} - -/// Update position value -#[inline(always)] -pub fn update_position_value(value_usd: f64) { - FoxhuntMetrics::update_position_value(value_usd); -} - -/// Update VaR -#[inline(always)] -pub fn update_var(var_usd: f64) { - FoxhuntMetrics::update_var(var_usd); -} - -/// Update active orders count -#[inline(always)] -pub fn update_active_orders(count: i64) { - FoxhuntMetrics::update_active_orders(count); -} - -/// Record market data message -#[inline(always)] -pub fn record_market_data_message() { - FoxhuntMetrics::record_market_data_message(); -} - -/// Record ML prediction -#[inline(always)] -pub fn record_ml_prediction() { - FoxhuntMetrics::record_ml_prediction(); -} - -/// Update system throughput -#[inline(always)] -pub fn update_throughput(ops_per_second: f64) { - FoxhuntMetrics::update_throughput(ops_per_second); -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn test_order_metrics() { - FoxhuntMetrics::reset_all(); - - record_order(); - record_order(); - record_order_fill(); - - assert_eq!(ORDER_COUNTER.get(), 2.0); - assert_eq!(ORDER_FILL_COUNTER.get(), 1.0); - } - - #[test] - fn test_latency_metrics() { - record_latency(25.5); - record_order_processing_latency(15.2); - record_market_data_latency(2.1); - - // Histograms don't have direct getters, but we can verify they accept values - // In a real environment, these would be scraped by Prometheus - } - - #[test] - fn test_risk_metrics() { - FoxhuntMetrics::reset_all(); - - record_risk_breach(); - update_position_value(150000.0); - update_var(5000.0); - - assert_eq!(RISK_BREACH_COUNTER.get(), 1.0); - assert_eq!(POSITION_VALUE_GAUGE.get(), 150000.0); - assert_eq!(VAR_GAUGE.get(), 5000.0); - } - - #[test] - fn test_custom_metrics() { - let mut metrics = FoxhuntMetrics::new(); - - metrics.create_custom_counter("test_counter", "Test counter").unwrap(); - metrics.increment_custom_counter("test_counter"); - - // Custom counter incremented successfully - } - - #[test] - fn test_metrics_export() { - record_order(); - - let exported = FoxhuntMetrics::export_metrics().unwrap(); - assert!(exported.contains("foxhunt_orders_total")); - } -} \ No newline at end of file diff --git a/monitoring/mod.rs b/monitoring/mod.rs deleted file mode 100644 index 4a2cee15a..000000000 --- a/monitoring/mod.rs +++ /dev/null @@ -1,32 +0,0 @@ -//! Foxhunt Monitoring Module -//! -//! Provides comprehensive Prometheus metrics collection for the HFT trading system. -//! Optimized for minimal latency impact in critical trading paths. - -pub mod metrics; -pub mod server; - -// DO NOT RE-EXPORT - Use explicit imports at usage sites - FoxhuntMetrics, - record_order, - record_order_fill, - record_order_rejection, - record_latency, - record_order_processing_latency, - record_market_data_latency, - record_risk_breach, - update_position_value, - update_var, - update_active_orders, - record_market_data_message, - record_ml_prediction, - update_throughput, -}; - -// DO NOT RE-EXPORT - Use explicit imports at usage sites - MetricsServer, - MetricsServerConfig, - MetricsError, - start_metrics_server, - start_metrics_server_with_config, -}; \ No newline at end of file diff --git a/monitoring/prometheus/alerts/api_gateway_alerts.yml b/monitoring/prometheus/alerts/api_gateway_alerts.yml deleted file mode 100644 index 6b6c7d2c8..000000000 --- a/monitoring/prometheus/alerts/api_gateway_alerts.yml +++ /dev/null @@ -1,162 +0,0 @@ -# Prometheus Alert Rules for API Gateway -# -# Critical alerts for authentication, proxy, and configuration - -groups: - - name: api_gateway_auth - interval: 10s - rules: - # Auth SLA Violation: >10ΞΌs latency - - alert: AuthLatencySLAViolation - expr: histogram_quantile(0.99, rate(api_gateway_auth_total_duration_microseconds_bucket[1m])) > 10 - for: 1m - labels: - severity: critical - component: auth - annotations: - summary: "API Gateway auth latency exceeded 10ΞΌs SLA" - description: "p99 auth latency is {{ $value }}ΞΌs (target: <10ΞΌs)" - - # High auth failure rate - - alert: HighAuthFailureRate - expr: 100 * rate(api_gateway_auth_requests_failure[5m]) / rate(api_gateway_auth_requests_total[5m]) > 10 - for: 2m - labels: - severity: warning - component: auth - annotations: - summary: "High authentication failure rate" - description: "Auth failure rate is {{ $value }}% (threshold: 10%)" - - # Redis connection failure - - alert: RedisConnectionFailure - expr: rate(api_gateway_auth_errors_redis_failure[1m]) > 0 - for: 1m - labels: - severity: critical - component: auth - annotations: - summary: "JWT revocation Redis connection failed" - description: "Redis errors detected: {{ $value }}/s" - - # JWT revocation cache size explosion - - alert: RevocationCacheSizeExplosion - expr: api_gateway_revoked_tokens_cached > 100000 - for: 5m - labels: - severity: warning - component: auth - annotations: - summary: "JWT revocation cache size excessive" - description: "Revoked tokens cached: {{ $value }} (threshold: 100k)" - - # Cache hit rate too low - - alert: LowCacheHitRate - expr: | - 100 * rate(api_gateway_rbac_cache_hits[5m]) / - (rate(api_gateway_rbac_cache_hits[5m]) + rate(api_gateway_rbac_cache_misses[5m])) < 90 - for: 5m - labels: - severity: warning - component: auth - annotations: - summary: "RBAC cache hit rate below 90%" - description: "Cache hit rate is {{ $value }}% (target: >90%)" - - - name: api_gateway_proxy - interval: 10s - rules: - # Circuit breaker open - - alert: CircuitBreakerOpen - expr: api_gateway_circuit_breaker_state > 1.5 - for: 1m - labels: - severity: critical - component: proxy - annotations: - summary: "Circuit breaker open for {{ $labels.service }}" - description: "Backend service {{ $labels.service }} circuit breaker is open" - - # Backend service unhealthy - - alert: BackendServiceUnhealthy - expr: api_gateway_health_status == 0 - for: 2m - labels: - severity: critical - component: proxy - annotations: - summary: "Backend service {{ $labels.service }} unhealthy" - description: "Health checks failing for {{ $labels.service }}" - - # High backend latency - - alert: HighBackendLatency - expr: histogram_quantile(0.99, rate(api_gateway_backend_request_duration_milliseconds_bucket[1m])) > 100 - for: 3m - labels: - severity: warning - component: proxy - annotations: - summary: "High latency to {{ $labels.service }}" - description: "p99 latency to {{ $labels.service }} is {{ $value }}ms (threshold: 100ms)" - - # Connection pool exhaustion - - alert: ConnectionPoolExhaustion - expr: | - 100 * api_gateway_connection_pool_active / api_gateway_connection_pool_max > 90 - for: 5m - labels: - severity: warning - component: proxy - annotations: - summary: "Connection pool nearly exhausted for {{ $labels.service }}" - description: "Pool utilization: {{ $value }}% (threshold: 90%)" - - - name: api_gateway_config - interval: 10s - rules: - # NOTIFY listener disconnected - - alert: NotifyListenerDisconnected - expr: api_gateway_notify_listener_connected == 0 - for: 1m - labels: - severity: critical - component: config - annotations: - summary: "PostgreSQL NOTIFY listener disconnected" - description: "Hot-reload capability lost - configuration changes will not propagate" - - # High config reload latency - - alert: HighConfigReloadLatency - expr: histogram_quantile(0.95, rate(api_gateway_config_reload_duration_milliseconds_bucket[1m])) > 100 - for: 5m - labels: - severity: warning - component: config - annotations: - summary: "Slow configuration reload" - description: "p95 config reload latency is {{ $value }}ms (threshold: 100ms)" - - # Config validation failures - - alert: ConfigValidationFailures - expr: rate(api_gateway_config_validation_failure[5m]) > 0 - for: 2m - labels: - severity: warning - component: config - annotations: - summary: "Configuration validation failures detected" - description: "Invalid config updates: {{ $value }}/s" - - - name: api_gateway_rate_limiting - interval: 10s - rules: - # Excessive rate limiting - - alert: ExcessiveRateLimiting - expr: rate(api_gateway_auth_errors_rate_limited[1m]) > 10 - for: 5m - labels: - severity: warning - component: rate_limiting - annotations: - summary: "High rate limit rejection rate" - description: "Rate limit rejections: {{ $value }}/s (may indicate DDoS or misconfiguration)" diff --git a/monitoring/prometheus/alerts/backtesting_alerts.yml b/monitoring/prometheus/alerts/backtesting_alerts.yml deleted file mode 100644 index c3083bf1f..000000000 --- a/monitoring/prometheus/alerts/backtesting_alerts.yml +++ /dev/null @@ -1,311 +0,0 @@ -# Prometheus Alert Rules for Backtesting Service -# -# Alerts for strategy testing, performance analytics, and data replay - -groups: - - name: backtesting_performance - interval: 10s - rules: - # Backtest execution time excessive - - alert: BacktestExecutionTimeSlow - expr: histogram_quantile(0.95, rate(backtesting_execution_duration_seconds_bucket[10m])) > 300 - for: 5m - labels: - severity: warning - component: backtesting - annotations: - summary: "Backtest execution time excessive" - description: "p95 execution time {{ $value }}s (threshold: 300s)" - impact: "Strategy testing taking too long" - - # Data replay latency high - - alert: DataReplayLatencyHigh - expr: histogram_quantile(0.99, rate(backtesting_data_replay_microseconds_bucket[5m])) > 1000 - for: 3m - labels: - severity: warning - component: backtesting - annotations: - summary: "Data replay latency high" - description: "p99 replay latency {{ $value }}ΞΌs (threshold: 1ms)" - impact: "Backtest simulation slower than expected" - - # Backtest throughput low - - alert: BacktestThroughputLow - expr: rate(backtesting_events_processed_total[5m]) < 1000 - for: 5m - labels: - severity: warning - component: backtesting - annotations: - summary: "Backtest event processing throughput low" - description: "Processing {{ $value }} events/sec (threshold: 1000)" - - # Strategy simulation errors - - alert: StrategySimulationErrors - expr: rate(backtesting_simulation_errors_total[5m]) > 0.5 - for: 2m - labels: - severity: warning - component: backtesting - annotations: - summary: "Strategy simulation errors detected" - description: "{{ $value }} simulation errors/sec" - impact: "Backtest results may be unreliable" - - - name: backtesting_availability - interval: 10s - rules: - # Backtesting service down - - alert: BacktestingServiceDown - expr: up{job="backtesting_service"} == 0 - for: 30s - labels: - severity: high - component: backtesting - annotations: - summary: "Backtesting service is DOWN" - description: "Service unreachable for >30 seconds" - impact: "Strategy testing unavailable" - runbook_url: "https://docs.foxhunt.io/runbooks/backtesting-service-down" - - # Active backtest count high - - alert: ActiveBacktestCountHigh - expr: backtesting_active_backtests > 10 - for: 10m - labels: - severity: warning - component: backtesting - annotations: - summary: "High number of concurrent backtests" - description: "{{ $value }} active backtests (threshold: 10)" - impact: "System resources may be constrained" - - # Backtest queue depth high - - alert: BacktestQueueDepthHigh - expr: backtesting_queue_depth > 50 - for: 5m - labels: - severity: warning - component: backtesting - annotations: - summary: "Backtest queue depth high" - description: "{{ $value }} backtests queued (threshold: 50)" - impact: "Backtest execution delays likely" - - - name: backtesting_data_quality - interval: 15s - rules: - # Parquet file read errors - - alert: ParquetFileReadErrors - expr: rate(backtesting_parquet_read_errors_total[5m]) > 0.1 - for: 2m - labels: - severity: warning - component: backtesting - annotations: - summary: "Parquet file read errors detected" - description: "{{ $value }} read errors/sec" - impact: "Historical data replay failing" - action: "1. Check file integrity 2. Verify storage access 3. Review file formats" - - # Missing historical data - - alert: MissingHistoricalData - expr: backtesting_missing_data_points > 100 - for: 5m - labels: - severity: warning - component: backtesting - annotations: - summary: "Missing historical data points detected" - description: "{{ $value }} missing data points in backtest" - impact: "Backtest results may have gaps" - - # Data timestamp inconsistency - - alert: DataTimestampInconsistency - expr: rate(backtesting_timestamp_errors_total[5m]) > 0 - for: 2m - labels: - severity: warning - component: backtesting - annotations: - summary: "Data timestamp inconsistencies detected" - description: "{{ $value }} timestamp errors/sec" - impact: "Backtest timeline accuracy compromised" - - - name: backtesting_analytics - interval: 15s - rules: - # Strategy Sharpe ratio low - - alert: StrategyPerformancePoor - expr: backtesting_strategy_sharpe_ratio < 1.0 - for: 0s - labels: - severity: info - component: backtesting - annotations: - summary: "Strategy Sharpe ratio below target" - description: "Strategy {{ $labels.strategy }} Sharpe ratio {{ $value }} (target: >1.0)" - impact: "Strategy may not be profitable enough" - - # Excessive drawdown in backtest - - alert: BacktestDrawdownExcessive - expr: backtesting_max_drawdown_pct > 25 - for: 0s - labels: - severity: warning - component: backtesting - annotations: - summary: "Backtest shows excessive drawdown" - description: "Strategy {{ $labels.strategy }} max drawdown {{ $value }}% (threshold: 25%)" - impact: "Strategy risk profile unacceptable" - - # Win rate too low - - alert: BacktestWinRateLow - expr: backtesting_win_rate_pct < 45 - for: 0s - labels: - severity: info - component: backtesting - annotations: - summary: "Backtest win rate low" - description: "Strategy {{ $labels.strategy }} win rate {{ $value }}% (threshold: 45%)" - - # PnL volatility high - - alert: BacktestPnLVolatilityHigh - expr: backtesting_pnl_volatility > 0.10 - for: 0s - labels: - severity: info - component: backtesting - annotations: - summary: "Backtest PnL volatility high" - description: "Strategy {{ $labels.strategy }} PnL volatility {{ $value }} (threshold: 0.10)" - impact: "Strategy may be too risky" - - - name: backtesting_model_loader - interval: 15s - rules: - # Model loading failures in backtest - - alert: BacktestModelLoadingFailures - expr: rate(backtesting_model_load_failures_total[5m]) > 0.1 - for: 2m - labels: - severity: critical - component: backtesting - annotations: - summary: "ML model loading failures in backtesting" - description: "{{ $value }} model load failures/sec" - impact: "Cannot test ML-based strategies" - action: "1. Check model files 2. Verify S3 access 3. Check model compatibility" - - # Model cache misses - - alert: BacktestModelCacheMissesHigh - expr: | - 100 * rate(backtesting_model_cache_misses[5m]) / - (rate(backtesting_model_cache_hits[5m]) + rate(backtesting_model_cache_misses[5m])) > 30 - for: 5m - labels: - severity: warning - component: backtesting - annotations: - summary: "Model cache miss rate high in backtesting" - description: "Cache miss rate {{ $value }}% (threshold: 30%)" - impact: "Increased backtest execution time" - - # Model inference errors during backtest - - alert: BacktestModelInferenceErrors - expr: rate(backtesting_model_inference_errors_total[5m]) > 1 - for: 2m - labels: - severity: warning - component: backtesting - annotations: - summary: "ML inference errors during backtest" - description: "{{ $value }} inference errors/sec" - impact: "Strategy simulation results unreliable" - - - name: backtesting_resources - interval: 15s - rules: - # High memory usage - - alert: BacktestingServiceMemoryHigh - expr: | - 100 * process_resident_memory_bytes{job="backtesting_service"} / - node_memory_MemTotal_bytes > 85 - for: 5m - labels: - severity: warning - component: backtesting - annotations: - summary: "Backtesting service memory usage high" - description: "Memory usage {{ $value }}% (threshold: 85%)" - impact: "Risk of OOM during large backtests" - - # High CPU usage - - alert: BacktestingServiceCPUHigh - expr: | - 100 * rate(process_cpu_seconds_total{job="backtesting_service"}[1m]) > 90 - for: 3m - labels: - severity: warning - component: backtesting - annotations: - summary: "Backtesting service CPU usage high" - description: "CPU usage {{ $value }}% (threshold: 90%)" - impact: "Backtest execution may slow down" - - # Storage space low for results - - alert: BacktestResultStorageLow - expr: | - 100 * backtesting_results_storage_used_bytes / - backtesting_results_storage_limit_bytes > 85 - for: 10m - labels: - severity: warning - component: backtesting - annotations: - summary: "Backtest result storage usage high" - description: "Storage {{ $value }}% full (threshold: 85%)" - impact: "Cannot save new backtest results" - action: "1. Archive old results 2. Clean temporary files 3. Increase storage" - - - name: backtesting_database - interval: 15s - rules: - # Database connection pool exhaustion - - alert: BacktestingDBPoolExhaustion - expr: | - 100 * backtesting_db_connections_active / - backtesting_db_connections_max > 80 - for: 2m - labels: - severity: warning - component: backtesting - annotations: - summary: "Backtesting database connection pool nearly exhausted" - description: "Pool utilization {{ $value }}% (threshold: 80%)" - - # Slow database queries - - alert: BacktestingDBQueriesSlow - expr: histogram_quantile(0.95, rate(backtesting_db_query_duration_seconds_bucket[5m])) > 1 - for: 3m - labels: - severity: warning - component: backtesting - annotations: - summary: "Backtesting database queries slow" - description: "p95 query time {{ $value }}s (threshold: 1s)" - impact: "Backtest result persistence delayed" - - # Database write errors - - alert: BacktestingDBWriteErrors - expr: rate(backtesting_db_write_errors_total[5m]) > 0.1 - for: 2m - labels: - severity: warning - component: backtesting - annotations: - summary: "Database write errors in backtesting" - description: "{{ $value }} write errors/sec" - impact: "Backtest results may not be saved" diff --git a/monitoring/prometheus/alerts/ensemble_ml_alerts.yml b/monitoring/prometheus/alerts/ensemble_ml_alerts.yml deleted file mode 100644 index 8180f7f2a..000000000 --- a/monitoring/prometheus/alerts/ensemble_ml_alerts.yml +++ /dev/null @@ -1,601 +0,0 @@ -# Prometheus Alert Rules for Ensemble ML Production Monitoring -# -# Critical alerts for 6-model ensemble (DQN, PPO, MAMBA-2, TFT, TLOB) -# Monitors: Sharpe ratio, disagreement, latency, model failures, cascades -# -# See: ENSEMBLE_METRICS_QUICK_REFERENCE.md -# Runbook: ENSEMBLE_RUNBOOK.md - -groups: - # ============================================================================= - # GROUP 1: PERFORMANCE DEGRADATION ALERTS - # ============================================================================= - - name: ensemble_performance_degradation - interval: 15s - rules: - # CRITICAL: Sharpe ratio drops >50% (15min window) - - alert: EnsembleSharpeRatioDropCritical - expr: | - ( - ( - sum(rate(ensemble_model_pnl_contribution_dollars_sum[15m])) / - sqrt(sum(rate(ensemble_model_pnl_contribution_dollars_count[15m]))) - ) / - ( - sum(rate(ensemble_model_pnl_contribution_dollars_sum[15m] offset 1h)) / - sqrt(sum(rate(ensemble_model_pnl_contribution_dollars_count[15m] offset 1h))) - ) - ) < 0.5 - for: 15m - labels: - severity: critical - component: ensemble - alert_type: performance - annotations: - summary: "Ensemble Sharpe ratio dropped >50% in 15 minutes" - description: "Current Sharpe ratio is {{ $value | humanizePercentage }} of baseline (threshold: 50%)" - impact: "Strategy profitability severely degraded - potential regime shift or model drift" - action: | - 1. Check Grafana dashboard: http://localhost:3000/d/ensemble-ml-production - 2. Review model disagreement rates - 3. Check for market regime changes - 4. Consider reducing position sizes or switching to defensive mode - 5. Review model weights - any single model dominating? - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-sharpe-drop" - - # WARNING: Sharpe ratio drops 25-50% (15min window) - - alert: EnsembleSharpeRatioDropWarning - expr: | - ( - ( - sum(rate(ensemble_model_pnl_contribution_dollars_sum[15m])) / - sqrt(sum(rate(ensemble_model_pnl_contribution_dollars_count[15m]))) - ) / - ( - sum(rate(ensemble_model_pnl_contribution_dollars_sum[15m] offset 1h)) / - sqrt(sum(rate(ensemble_model_pnl_contribution_dollars_count[15m] offset 1h))) - ) - ) < 0.75 - for: 15m - labels: - severity: warning - component: ensemble - alert_type: performance - annotations: - summary: "Ensemble Sharpe ratio degraded by 25-50%" - description: "Sharpe ratio {{ $value | humanizePercentage }} of baseline (threshold: 75%)" - impact: "Strategy performance degrading - early warning" - action: "Monitor closely - review model weights and disagreement patterns" - - # CRITICAL: Win rate collapse (<40%) - - alert: EnsembleWinRateCollapse - expr: | - 100 * ( - sum(rate(ensemble_predictions_total{action="buy"}[15m])) + - sum(rate(ensemble_predictions_total{action="sell"}[15m])) - ) / - sum(rate(ensemble_predictions_total[15m])) < 40 - for: 15m - labels: - severity: critical - component: ensemble - alert_type: performance - annotations: - summary: "Ensemble win rate collapsed below 40%" - description: "Win rate: {{ $value }}% (threshold: 40%)" - impact: "Strategy losing money - immediate intervention required" - action: | - 1. STOP trading immediately if drawdown >10% - 2. Review prediction confidence levels - 3. Check model weights - any model contributing negative P&L? - 4. Switch to single best-performing model - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-win-rate-collapse" - - # CRITICAL: Negative P&L trending over 30min - - alert: EnsembleNegativePnLTrend - expr: | - sum(rate(ensemble_model_pnl_contribution_dollars_sum[30m])) < -1000 - for: 30m - labels: - severity: critical - component: ensemble - alert_type: performance - annotations: - summary: "Ensemble generating consistent negative P&L" - description: "30-min P&L trend: ${{ $value | humanize }} (threshold: -$1000)" - impact: "Active capital loss - trading system hemorrhaging money" - action: | - 1. IMMEDIATE: Halt automated trading - 2. Review all active positions - 3. Identify losing model(s) via P&L attribution - 4. Switch to manual trading or single-model mode - 5. Escalate to ML team for model investigation - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-negative-pnl" - - # ============================================================================= - # GROUP 2: MODEL DISAGREEMENT & UNCERTAINTY - # ============================================================================= - - name: ensemble_disagreement_alerts - interval: 5s - rules: - # CRITICAL: High disagreement >70% sustained for 5min - - alert: EnsembleHighDisagreementCritical - expr: ensemble_disagreement_rate > 0.7 - for: 5m - labels: - severity: critical - component: ensemble - alert_type: disagreement - annotations: - summary: "Critical model disagreement >70% for 5 minutes" - description: "Disagreement rate {{ $value | humanizePercentage }} on {{ $labels.symbol }} (threshold: 70%)" - impact: "Models fundamentally disagree - prediction reliability compromised" - action: | - 1. REDUCE position sizes by 50% immediately - 2. Check for market regime shift (Grafana regime detection panel) - 3. Review individual model predictions for {{ $labels.symbol }} - 4. Consider pausing trading for {{ $labels.symbol }} if disagreement persists >15min - 5. Check recent news/events for {{ $labels.symbol }} - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-high-disagreement" - - # WARNING: Moderate disagreement 50-70% sustained for 5min - - alert: EnsembleHighDisagreementWarning - expr: ensemble_disagreement_rate > 0.5 and ensemble_disagreement_rate <= 0.7 - for: 5m - labels: - severity: warning - component: ensemble - alert_type: disagreement - annotations: - summary: "Elevated model disagreement 50-70%" - description: "Disagreement {{ $value | humanizePercentage }} on {{ $labels.symbol }} (threshold: 50%)" - impact: "Models showing divergence - increased prediction uncertainty" - action: | - 1. Monitor closely - review model weights - 2. Check ensemble confidence scores - 3. Consider tightening stop-losses for {{ $labels.symbol }} - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-disagreement" - - # CRITICAL: Disagreement spike rate (rapid increase) - - alert: EnsembleDisagreementSpikeRate - expr: | - rate(ensemble_high_disagreement_total{threshold="0.7"}[5m]) > 10 - for: 5m - labels: - severity: critical - component: ensemble - alert_type: disagreement - annotations: - summary: "Rapid increase in high-disagreement predictions" - description: "{{ $value }} high-disagreement events/sec (threshold: 10)" - impact: "Market conditions rapidly changing - model consensus breaking down" - action: | - 1. PAUSE new position entries immediately - 2. Close positions with low confidence (<0.6) - 3. Review market volatility indicators - 4. Check for flash crash or news events - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-disagreement-spike" - - # CRITICAL: Low confidence + high disagreement (worst case) - - alert: EnsembleLowConfidenceHighDisagreement - expr: | - (ensemble_confidence_score < 0.6) and (ensemble_disagreement_rate > 0.7) - for: 2m - labels: - severity: critical - component: ensemble - alert_type: uncertainty - annotations: - summary: "Low confidence AND high disagreement detected" - description: "Confidence {{ $labels.confidence }} / Disagreement {{ $labels.disagreement }} on {{ $labels.symbol }}" - impact: "Models uncertain AND disagreeing - prediction reliability at minimum" - action: | - 1. HALT all trading for {{ $labels.symbol }} immediately - 2. Switch to observation mode (log predictions without execution) - 3. Manually review market conditions - 4. Do NOT resume automated trading until confidence >0.7 AND disagreement <0.5 - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-uncertainty" - - # ============================================================================= - # GROUP 3: LATENCY & PERFORMANCE - # ============================================================================= - - name: ensemble_latency_alerts - interval: 15s - rules: - # CRITICAL: Aggregation latency P99 >50ΞΌs for 1min - - alert: EnsembleAggregationLatencyP99High - expr: | - histogram_quantile(0.99, rate(ensemble_aggregation_latency_microseconds_bucket[1m])) > 50 - for: 1m - labels: - severity: critical - component: ensemble - alert_type: latency - annotations: - summary: "Ensemble aggregation P99 latency >50ΞΌs SLA violation" - description: "P99 latency {{ $value }}ΞΌs for {{ $labels.aggregation_method }} (threshold: 50ΞΌs)" - impact: "HFT execution delays - potential alpha decay and slippage" - action: | - 1. Check CPU utilization on trading service - 2. Review concurrent prediction load - 3. Check for contention in model inference - 4. Consider scaling to additional instances - 5. Review Grafana latency breakdown by model - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-latency-high" - - # WARNING: Aggregation latency P99 25-50ΞΌs for 1min - - alert: EnsembleAggregationLatencyP99Warning - expr: | - histogram_quantile(0.99, rate(ensemble_aggregation_latency_microseconds_bucket[1m])) > 25 - for: 1m - labels: - severity: warning - component: ensemble - alert_type: latency - annotations: - summary: "Ensemble aggregation P99 latency elevated" - description: "P99 latency {{ $value }}ΞΌs (target: <25ΞΌs)" - impact: "Performance degrading - monitor for further increases" - - # WARNING: Aggregation latency P95 >25ΞΌs for 2min - - alert: EnsembleAggregationLatencyP95High - expr: | - histogram_quantile(0.95, rate(ensemble_aggregation_latency_microseconds_bucket[2m])) > 25 - for: 2m - labels: - severity: warning - component: ensemble - alert_type: latency - annotations: - summary: "Ensemble aggregation P95 latency elevated" - description: "P95 latency {{ $value }}ΞΌs (target: <10ΞΌs)" - - # CRITICAL: Prediction throughput collapse - - alert: EnsemblePredictionThroughputCollapse - expr: rate(ensemble_predictions_total[1m]) < 10 - for: 2m - labels: - severity: critical - component: ensemble - alert_type: performance - annotations: - summary: "Ensemble prediction throughput collapsed" - description: "{{ $value }} predictions/sec (expected: >100)" - impact: "Ensemble not generating predictions - trading halted" - action: | - 1. Check trading service health status - 2. Review model loading status (any models failed?) - 3. Check inference queue depth - 4. Review logs for errors - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-throughput-collapse" - - # ============================================================================= - # GROUP 4: MODEL FAILURES & CASCADE DETECTION - # ============================================================================= - - name: ensemble_model_failures - interval: 5s - rules: - # CRITICAL: Any model failed to load or predict (immediate) - - alert: EnsembleModelFailureDetected - expr: | - ( - checkpoint_swaps_total{status="failed"} > 0 - ) or ( - sum(rate(ensemble_predictions_total[1m])) by (model_id) == 0 - ) - for: 0s - labels: - severity: critical - component: ensemble - alert_type: model_failure - annotations: - summary: "Model failure detected in ensemble" - description: "Model {{ $labels.model_id }} failed (checkpoint swap or prediction)" - impact: "Ensemble operating with reduced capacity - prediction quality degraded" - action: | - 1. Identify failed model: {{ $labels.model_id }} - 2. Check trading service logs for error details - 3. Verify checkpoint file integrity in MinIO - 4. Attempt manual checkpoint reload - 5. If reload fails, remove model from ensemble temporarily - 6. Escalate to ML team if issue persists - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-model-failure" - - # CRITICAL: Multiple models failed (cascade failure) - - alert: EnsembleCascadeFailureDetected - expr: | - count( - sum(rate(ensemble_predictions_total[1m])) by (model_id) == 0 - ) >= 2 - for: 30s - labels: - severity: critical - component: ensemble - alert_type: cascade_failure - annotations: - summary: "CASCADE FAILURE: Multiple models failed simultaneously" - description: "{{ $value }} models not producing predictions" - impact: "CRITICAL: Ensemble integrity compromised - predictions unreliable" - action: | - 1. IMMEDIATE: Halt all automated trading - 2. Switch to manual trading mode - 3. Review trading service logs for common failure cause - 4. Check infrastructure: GPU, memory, disk, network - 5. Verify MinIO checkpoint storage accessibility - 6. Page on-call ML engineer immediately - 7. Do NOT resume automated trading until >=4 models operational - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-cascade-failure" - - # CRITICAL: High checkpoint rollback rate (>10% over 10min) - - alert: EnsembleCheckpointRollbackRateHigh - expr: | - 100 * ( - sum(rate(checkpoint_swaps_total{status="rollback"}[10m])) / - sum(rate(checkpoint_swaps_total[10m])) - ) > 10 - for: 10m - labels: - severity: critical - component: ensemble - alert_type: checkpoint_quality - annotations: - summary: "High checkpoint rollback rate detected" - description: "{{ $value }}% of checkpoint swaps rolled back (threshold: 10%)" - impact: "New checkpoints failing canary validation - model quality issues" - action: | - 1. Review recent checkpoint training quality metrics - 2. Check canary validation failure reasons in logs - 3. Investigate training data quality issues - 4. Review hyperparameter tuning results - 5. Consider pausing automated checkpoint updates - 6. Escalate to ML training team - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-checkpoint-rollback" - - # WARNING: Single model checkpoint swap failed - - alert: EnsembleCheckpointSwapFailed - expr: | - rate(checkpoint_swaps_total{status="failed"}[5m]) > 0 - for: 5m - labels: - severity: warning - component: ensemble - alert_type: checkpoint - annotations: - summary: "Checkpoint swap failed for model {{ $labels.model_id }}" - description: "Failed to load new checkpoint - using previous version" - impact: "Model {{ $labels.model_id }} not updated - operating on stale checkpoint" - action: | - 1. Check MinIO checkpoint availability - 2. Verify checkpoint file integrity - 3. Review disk space and memory availability - 4. Attempt manual checkpoint load - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-checkpoint-swap-failed" - - # ============================================================================= - # GROUP 5: MODEL WEIGHT & CONTRIBUTION ANOMALIES - # ============================================================================= - - name: ensemble_weight_anomalies - interval: 30s - rules: - # CRITICAL: Single model dominating (>70% weight) - - alert: EnsembleSingleModelDominance - expr: | - max(ensemble_model_weight) by (symbol) > 0.7 - for: 10m - labels: - severity: critical - component: ensemble - alert_type: weight_imbalance - annotations: - summary: "Single model dominating ensemble decisions" - description: "Model {{ $labels.model_id }} has {{ $value | humanizePercentage }} weight on {{ $labels.symbol }} (threshold: 70%)" - impact: "Ensemble diversity lost - effectively operating as single model" - action: | - 1. Review why other models have low weights - 2. Check P&L attribution - is dominant model actually best performer? - 3. Verify other models are producing valid predictions - 4. Consider manual weight rebalancing if issue persists - 5. May indicate other models underperforming or failing silently - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-model-dominance" - - # WARNING: Model weights not summing to 1.0 (coordination issue) - - alert: EnsembleWeightSumAnomalous - expr: | - abs(sum(ensemble_model_weight) by (symbol) - 1.0) > 0.1 - for: 5m - labels: - severity: warning - component: ensemble - alert_type: weight_coordination - annotations: - summary: "Ensemble model weights not summing to 1.0" - description: "Weight sum {{ $value }} on {{ $labels.symbol }} (expected: 1.0 Β± 0.1)" - impact: "Weight normalization issue - ensemble coordination broken" - action: | - 1. Check ensemble coordinator logic - 2. Review weight update algorithm - 3. Restart trading service if issue persists - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-weight-sum-anomaly" - - # CRITICAL: Model contributing consistently negative P&L - - alert: EnsembleModelNegativePnLContribution - expr: | - sum(rate(ensemble_model_pnl_contribution_dollars_sum[1h])) by (model_id, symbol) < -500 - for: 1h - labels: - severity: critical - component: ensemble - alert_type: model_performance - annotations: - summary: "Model {{ $labels.model_id }} contributing negative P&L" - description: "1-hour P&L: ${{ $value | humanize }} on {{ $labels.symbol }} (threshold: -$500)" - impact: "Model actively losing money - dragging down ensemble performance" - action: | - 1. Reduce weight for {{ $labels.model_id }} to 0 temporarily - 2. Review model training quality - 3. Check for model drift or data distribution shift - 4. Consider removing model from ensemble until retrained - 5. Analyze prediction patterns for {{ $labels.model_id }} - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-model-negative-pnl" - - # WARNING: Model stopped contributing (weight=0 for >15min) - - alert: EnsembleModelZeroWeight - expr: | - ensemble_model_weight == 0 - for: 15m - labels: - severity: warning - component: ensemble - alert_type: model_weight - annotations: - summary: "Model {{ $labels.model_id }} has zero weight" - description: "Model not contributing to ensemble decisions on {{ $labels.symbol }}" - impact: "Reduced ensemble diversity - operating with fewer models" - action: | - 1. Check if model is producing predictions - 2. Review recent P&L attribution for this model - 3. Verify model not marked as failed - 4. Check weight update algorithm logic - - # ============================================================================= - # GROUP 6: A/B TESTING & EXPERIMENTAL MODELS - # ============================================================================= - - name: ensemble_ab_testing_alerts - interval: 30s - rules: - # WARNING: A/B test assignment imbalance (>55/45 split) - - alert: EnsembleABTestImbalance - expr: | - abs( - ( - sum(rate(ab_test_assignments_total{group="treatment"}[10m])) / - sum(rate(ab_test_assignments_total[10m])) - ) - 0.5 - ) > 0.05 - for: 10m - labels: - severity: warning - component: ensemble - alert_type: ab_test - annotations: - summary: "A/B test assignment imbalance detected" - description: "Treatment group {{ $value | humanizePercentage }} of assignments (expected: 50% Β± 5%)" - impact: "A/B test results may be biased - statistical validity compromised" - action: | - 1. Review user assignment hashing algorithm - 2. Check for bias in traffic sources - 3. Verify random seed is properly set - 4. Consider restarting A/B test with corrected assignment - - # INFO: A/B test showing significant lift - - alert: EnsembleABTestSignificantLift - expr: | - ab_test_metric_difference{metric="sharpe_ratio"} > 0.2 - for: 1h - labels: - severity: info - component: ensemble - alert_type: ab_test - annotations: - summary: "A/B test showing significant Sharpe ratio lift" - description: "Treatment group {{ $value | humanizePercentage }} better than control" - impact: "Positive signal - treatment variant may be ready for full rollout" - action: | - 1. Review statistical significance (p-value <0.05) - 2. Check sample size (N >1000 per group) - 3. Verify lift is consistent across all symbols - 4. Consider graduating treatment to production - - # CRITICAL: A/B test showing significant negative impact - - alert: EnsembleABTestNegativeImpact - expr: | - ab_test_metric_difference{metric="sharpe_ratio"} < -0.15 - for: 30m - labels: - severity: critical - component: ensemble - alert_type: ab_test - annotations: - summary: "A/B test treatment performing significantly worse" - description: "Treatment group {{ $value | humanizePercentage }} worse than control" - impact: "Experimental model/configuration actively degrading performance" - action: | - 1. STOP A/B test immediately - 2. Revert all treatment group users to control - 3. Review what changed in treatment configuration - 4. Do NOT roll out treatment variant - 5. Investigate root cause before retrying - - # ============================================================================= - # GROUP 7: SYSTEM HEALTH & INFRASTRUCTURE - # ============================================================================= - - name: ensemble_system_health - interval: 15s - rules: - # CRITICAL: Trading service memory usage high (>85%) - - alert: EnsembleServiceMemoryHigh - expr: | - 100 * process_resident_memory_bytes{job="trading_service"} / - node_memory_MemTotal_bytes > 85 - for: 2m - labels: - severity: critical - component: ensemble - alert_type: resources - annotations: - summary: "Trading service memory usage critical" - description: "Memory usage {{ $value }}% (threshold: 85%)" - impact: "Risk of OOM kill - service instability" - action: | - 1. Check for memory leaks in ensemble coordinator - 2. Review loaded model sizes (MAMBA-2: 150-500MB, TFT: 1.5-2.5GB) - 3. Consider restarting service during low-traffic window - 4. Scale to larger instance if issue persists - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-memory-high" - - # CRITICAL: GPU inference failing (if GPU-accelerated) - - alert: EnsembleGPUInferenceFailure - expr: | - rate(ml_inference_gpu_errors_total[5m]) > 1 - for: 5m - labels: - severity: critical - component: ensemble - alert_type: gpu - annotations: - summary: "GPU inference errors detected" - description: "{{ $value }} GPU errors/sec" - impact: "GPU-accelerated models (MAMBA-2, TFT) may fail - falling back to CPU" - action: | - 1. Check GPU health: nvidia-smi - 2. Review CUDA error messages in logs - 3. Verify GPU memory not exhausted - 4. Restart service to reset GPU state - 5. Check for driver issues - - # WARNING: Ensemble coordinator errors - - alert: EnsembleCoordinatorErrors - expr: | - rate(ensemble_coordinator_errors_total[5m]) > 1 - for: 5m - labels: - severity: warning - component: ensemble - alert_type: errors - annotations: - summary: "Ensemble coordinator errors detected" - description: "{{ $value }} errors/sec in ensemble coordination logic" - impact: "Ensemble predictions may be incomplete or degraded" - action: | - 1. Review trading service error logs - 2. Check for pattern in error types - 3. Verify all models loaded correctly - 4. Consider restarting service if errors persist - -# ============================================================================= -# INHIBITION RULES (defined in alertmanager.yml) -# ============================================================================= -# - If EnsembleCascadeFailureDetected fires, suppress EnsembleModelFailureDetected -# - If EnsembleSharpeRatioDropCritical fires, suppress EnsembleSharpeRatioDropWarning -# - If EnsembleServiceMemoryHigh fires, suppress EnsembleAggregationLatencyP99High -# - If EnsembleCheckpointRollbackRateHigh fires, suppress EnsembleCheckpointSwapFailed diff --git a/monitoring/prometheus/alerts/ml_trading_alerts.yml b/monitoring/prometheus/alerts/ml_trading_alerts.yml deleted file mode 100644 index d8fc93a50..000000000 --- a/monitoring/prometheus/alerts/ml_trading_alerts.yml +++ /dev/null @@ -1,393 +0,0 @@ -# Prometheus Alert Rules for ML Trading Operations -# -# Alerts for ML model predictions, order execution, and ensemble behavior. -# Coordinates with ml_training_alerts.yml (training) and ensemble_ml_alerts.yml (aggregation). - -groups: - - name: ml_trading_accuracy - interval: 60s - rules: - # Model accuracy below production threshold - - alert: MLModelLowAccuracy - expr: ml_prediction_accuracy{model_id!="test"} < 55 - for: 1h - labels: - severity: warning - component: ml_trading - alert_type: performance - annotations: - summary: "ML model {{ $labels.model_id }} accuracy below 55%" - description: "Model {{ $labels.model_id }} prediction accuracy is {{ $value }}% (threshold: 55%)" - impact: "Model performance degraded - trading signals less reliable" - action: | - 1. Check model drift metrics - 2. Review recent data quality - 3. Compare with other models - 4. Consider retraining or disabling model - runbook_url: "https://docs.foxhunt.io/runbooks/ml-low-accuracy" - - # Win rate below profitability threshold - - alert: MLModelLowWinRate - expr: ml_model_win_rate{model_id!="test"} < 0.55 - for: 2h - labels: - severity: warning - component: ml_trading - alert_type: performance - annotations: - summary: "Model {{ $labels.model_id }} win rate below 55%" - description: "Win rate is {{ $value | humanizePercentage }} (threshold: 55%)" - impact: "Unprofitable trading - capital at risk" - action: | - 1. Review trade history for patterns - 2. Check market conditions (regime shift?) - 3. Analyze model confidence scores - 4. Consider reducing position sizes or disabling - dashboard_url: "https://grafana.foxhunt.io/d/ml-trading/model-performance" - - # Sharpe ratio below target - - alert: MLModelLowSharpeRatio - expr: ml_model_sharpe_ratio{model_id!="test"} < 1.5 - for: 6h - labels: - severity: info - component: ml_trading - alert_type: performance - annotations: - summary: "Model {{ $labels.model_id }} Sharpe ratio below 1.5" - description: "Sharpe ratio is {{ $value }} (target: >1.5)" - impact: "Risk-adjusted returns suboptimal" - action: | - 1. Compare with benchmark models - 2. Analyze volatility patterns - 3. Review position sizing strategy - 4. Monitor over longer timeframe - - - name: ml_trading_predictions - interval: 10s - rules: - # Model stopped making predictions (stale) - - alert: MLModelStalePredictions - expr: (time() - ml_model_last_prediction_time{model_id!="test"}) > 3600 - for: 5m - labels: - severity: warning - component: ml_trading - alert_type: availability - annotations: - summary: "Model {{ $labels.model_id }} has not made predictions in >1h" - description: "Last prediction was {{ $value | humanizeDuration }} ago (threshold: 1h)" - impact: "Model not generating signals - missing trading opportunities" - action: | - 1. Check model health status - 2. Verify data pipeline connectivity - 3. Review inference service logs - 4. Restart model if necessary - runbook_url: "https://docs.foxhunt.io/runbooks/stale-predictions" - - # Low prediction confidence (median) - - alert: MLPredictionConfidenceLow - expr: | - histogram_quantile(0.50, sum by (model_id, le) (rate(ml_predictions_confidence_bucket[5m]))) < 0.70 - for: 10m - labels: - severity: info - component: ml_trading - alert_type: quality - annotations: - summary: "Model {{ $labels.model_id }} median confidence < 0.70" - description: "Median prediction confidence is {{ $value }} (threshold: 0.70)" - impact: "High model uncertainty - signals less reliable" - action: | - 1. Check if market regime changed - 2. Review feature distribution - 3. Compare with historical confidence - 4. Consider reducing position sizes - - # Prediction rate anomaly (too low) - - alert: MLPredictionRateLow - expr: | - sum by (model_id) (rate(ml_predictions_total[5m])) < 0.01 - for: 15m - labels: - severity: warning - component: ml_trading - alert_type: availability - annotations: - summary: "Model {{ $labels.model_id }} prediction rate abnormally low" - description: "Prediction rate is {{ $value }}/sec (threshold: 0.01/sec)" - impact: "Model not actively generating signals" - action: | - 1. Check inference service health - 2. Verify data feed connectivity - 3. Review model load errors - 4. Check CPU/GPU utilization - - - name: ml_trading_orders - interval: 10s - rules: - # High order rejection rate - - alert: MLOrderRejectionRateHigh - expr: | - 100 * ( - sum by (model_id, symbol) (rate(ml_orders_rejected_total[5m])) - / - sum by (model_id, symbol) (rate(ml_orders_submitted_total[5m])) - ) > 10 - for: 5m - labels: - severity: warning - component: ml_trading - alert_type: execution - annotations: - summary: "High order rejection rate for {{ $labels.model_id }} on {{ $labels.symbol }}" - description: "Rejection rate is {{ $value }}% (threshold: 10%)" - impact: "Many orders being rejected - missing trades" - action: | - 1. Check rejection reasons (see ml_orders_rejected_total labels) - 2. Review risk limits for {{ $labels.symbol }} - 3. Verify margin availability - 4. Check market hours and trading halts - 5. Analyze price validity checks - dashboard_url: "https://grafana.foxhunt.io/d/ml-trading/order-execution" - - # Low order fill rate - - alert: MLOrderFillRateLow - expr: | - 100 * ( - sum by (model_id, symbol) (rate(ml_orders_filled_total[5m])) - / - sum by (model_id, symbol) (rate(ml_orders_submitted_total[5m])) - ) < 80 - for: 10m - labels: - severity: info - component: ml_trading - alert_type: execution - annotations: - summary: "Low fill rate for {{ $labels.model_id }} on {{ $labels.symbol }}" - description: "Fill rate is {{ $value }}% (threshold: 80%)" - impact: "Poor order execution - slippage may be high" - action: | - 1. Review order types (market vs limit) - 2. Check liquidity on {{ $labels.symbol }} - 3. Analyze order size relative to market depth - 4. Consider adjusting execution strategy - - # Risk limit rejections spike - - alert: MLOrderRiskLimitRejections - expr: | - rate(ml_orders_rejected_total{reason="risk_limit"}[5m]) > 0.1 - for: 2m - labels: - severity: critical - component: ml_trading - alert_type: risk - annotations: - summary: "Risk limit rejections for {{ $labels.model_id }} on {{ $labels.symbol }}" - description: "{{ $value }}/sec orders rejected due to risk limits" - impact: "Model hitting risk constraints - capital protection active" - action: | - 1. Review current exposure for {{ $labels.symbol }} - 2. Check if position limits are appropriate - 3. Verify VaR calculations - 4. Consider temporary model disablement if excessive - priority: high - - - name: ml_trading_ensemble - interval: 30s - rules: - # High ensemble disagreement rate - - alert: MLEnsembleHighDisagreement - expr: ml_ensemble_agreement_rate{symbol!=""} < 0.5 - for: 10m - labels: - severity: info - component: ml_trading - alert_type: ensemble - annotations: - summary: "High ensemble disagreement on {{ $labels.symbol }}" - description: "Model agreement rate is {{ $value | humanizePercentage }} (threshold: 50%)" - impact: "Models disagree significantly - market uncertainty or data issues" - action: | - 1. Check if market regime changed - 2. Review data quality metrics - 3. Compare individual model predictions - 4. Consider reducing position sizes - 5. Monitor for regime shift - dashboard_url: "https://grafana.foxhunt.io/d/ensemble/disagreement-analysis" - - # Frequent high disagreement events - - alert: MLEnsembleDisagreementEventsFrequent - expr: | - rate(ml_ensemble_disagreement_events{threshold="0.7"}[5m]) > 0.3 - for: 10m - labels: - severity: warning - component: ml_trading - alert_type: ensemble - annotations: - summary: "Frequent high disagreement events on {{ $labels.symbol }}" - description: "{{ $value }}/sec disagreement events (threshold: 0.3/sec)" - impact: "Models frequently conflicting - possible regime shift" - action: | - 1. Investigate market conditions - 2. Check for data anomalies - 3. Review model drift scores - 4. Consider ensemble rebalancing - 5. Alert trading desk - - # Ensemble voting stopped - - alert: MLEnsembleVotingStopped - expr: | - rate(ml_ensemble_votes_total[5m]) == 0 - for: 15m - labels: - severity: critical - component: ml_trading - alert_type: availability - annotations: - summary: "Ensemble voting stopped for {{ $labels.symbol }}" - description: "No ensemble votes in last 15 minutes" - impact: "Ensemble system not generating signals - trading halted" - action: | - 1. Check ensemble coordinator service - 2. Verify individual model health - 3. Review aggregation service logs - 4. Restart ensemble if necessary - priority: high - runbook_url: "https://docs.foxhunt.io/runbooks/ensemble-down" - - - name: ml_trading_latency - interval: 10s - rules: - # Inference latency P99 high - - alert: MLInferenceLatencyHigh - expr: | - histogram_quantile(0.99, sum by (model_id, le) (rate(ml_model_inference_latency_bucket[5m]))) > 1000 - for: 5m - labels: - severity: warning - component: ml_trading - alert_type: performance - annotations: - summary: "Model {{ $labels.model_id }} P99 inference latency > 1ms" - description: "P99 latency is {{ $value }}ΞΌs (threshold: 1000ΞΌs)" - impact: "Slow inference - trading signal delays" - action: | - 1. Check GPU utilization - 2. Review model load (batch size) - 3. Profile inference bottlenecks - 4. Consider model optimization - dashboard_url: "https://grafana.foxhunt.io/d/ml-trading/inference-latency" - - # Inference latency P99 critical - - alert: MLInferenceLatencyCritical - expr: | - histogram_quantile(0.99, sum by (model_id, le) (rate(ml_model_inference_latency_bucket[5m]))) > 5000 - for: 2m - labels: - severity: critical - component: ml_trading - alert_type: performance - annotations: - summary: "Model {{ $labels.model_id }} P99 inference latency > 5ms" - description: "P99 latency is {{ $value }}ΞΌs (threshold: 5000ΞΌs)" - impact: "Critical inference delays - real-time trading compromised" - action: | - 1. IMMEDIATE: Check system resources - 2. Reduce model load if possible - 3. Consider failover to faster model - 4. Alert on-call engineer - priority: high - - - name: ml_trading_risk - interval: 30s - rules: - # Large drawdown detected - - alert: MLModelLargeDrawdown - expr: ml_model_max_drawdown{model_id!="test"} < -5000 - for: 1h - labels: - severity: critical - component: ml_trading - alert_type: risk - annotations: - summary: "Model {{ $labels.model_id }} experiencing large drawdown" - description: "Maximum drawdown is ${{ $value }} (threshold: -$5000)" - impact: "Significant capital loss - risk controls may need adjustment" - action: | - 1. IMMEDIATE: Review current positions - 2. Consider reducing model allocation - 3. Analyze losing trades for patterns - 4. Check if stop-losses are working - 5. Alert risk management team - priority: high - dashboard_url: "https://grafana.foxhunt.io/d/ml-trading/risk-metrics" - - # Negative cumulative PnL - - alert: MLModelNegativePnL - expr: ml_model_cumulative_pnl{model_id!="test"} < 0 - for: 24h - labels: - severity: warning - component: ml_trading - alert_type: risk - annotations: - summary: "Model {{ $labels.model_id }} cumulative PnL negative" - description: "Cumulative PnL is ${{ $value }}" - impact: "Model unprofitable - consider disabling" - action: | - 1. Review 24h trading history - 2. Compare with other models - 3. Check market conditions - 4. Consider model retraining - 5. Evaluate continued deployment - - - name: ml_trading_healthcheck - interval: 30s - rules: - # No predictions from any model - - alert: MLTradingSystemDown - expr: | - sum(rate(ml_predictions_total[5m])) == 0 - for: 10m - labels: - severity: critical - component: ml_trading - alert_type: availability - annotations: - summary: "ML trading system not generating predictions" - description: "No predictions from any model in last 10 minutes" - impact: "Complete trading system outage - no signals generated" - action: | - 1. IMMEDIATE: Check trading service health - 2. Verify all model services running - 3. Check data pipeline connectivity - 4. Review system logs - 5. Alert on-call engineer - 6. Consider manual trading fallback - priority: emergency - runbook_url: "https://docs.foxhunt.io/runbooks/ml-system-down" - - # No orders submitted despite predictions - - alert: MLOrderSubmissionFailure - expr: | - (sum(rate(ml_predictions_total{action!="hold"}[5m])) > 0.01) and - (sum(rate(ml_orders_submitted_total[5m])) == 0) - for: 5m - labels: - severity: critical - component: ml_trading - alert_type: execution - annotations: - summary: "Models generating signals but no orders submitted" - description: "{{ $value }} buy/sell predictions but 0 orders submitted" - impact: "Order execution broken - missing all trades" - action: | - 1. IMMEDIATE: Check order submission service - 2. Verify exchange connectivity - 3. Review risk system status - 4. Check order validation logic - 5. Alert on-call engineer - priority: emergency diff --git a/monitoring/prometheus/alerts/ml_training_alerts.yml b/monitoring/prometheus/alerts/ml_training_alerts.yml deleted file mode 100644 index ad605e0d3..000000000 --- a/monitoring/prometheus/alerts/ml_training_alerts.yml +++ /dev/null @@ -1,446 +0,0 @@ -# Prometheus Alert Rules for ML Training Service -# -# Alerts for model training, GPU utilization, and ML inference performance - -groups: - - name: ml_training_performance - interval: 10s - rules: - # NaN values detected during training (CRITICAL) - - alert: TrainingNaNDetected - expr: ml_training_nan_count > 0 - for: 30s - labels: - severity: critical - component: ml - annotations: - summary: "NaN values detected during training" - description: "Model {{ $labels.model_type }} job {{ $labels.job_id }} detected NaN in {{ $labels.tensor_type }}" - impact: "Training unstable - model will produce invalid results" - action: "1. Stop training immediately 2. Check learning rate 3. Review data normalization 4. Inspect gradient clipping" - - # Training slowdown detected - - alert: TrainingSlowdown - expr: ml_training_epochs_per_second < 0.1 - for: 5m - labels: - severity: warning - component: ml - annotations: - summary: "Training speed degraded" - description: "Model {{ $labels.model_type }} training at {{ $value }} epochs/sec (threshold: 0.1)" - impact: "Training will take much longer than expected" - action: "1. Check GPU utilization 2. Profile data loading 3. Review batch size" - - # ML inference latency high - - alert: MLInferenceLatencyHigh - expr: histogram_quantile(0.99, rate(ml_inference_duration_milliseconds_bucket[1m])) > 100 - for: 2m - labels: - severity: warning - component: ml - annotations: - summary: "ML inference latency high" - description: "p99 inference latency is {{ $value }}ms (threshold: 100ms)" - impact: "Trading signal generation delayed" - - # Model prediction accuracy degraded - - alert: MLModelAccuracyDegraded - expr: ml_model_accuracy < 0.85 - for: 5m - labels: - severity: critical - component: ml - annotations: - summary: "ML model accuracy degraded" - description: "Model {{ $labels.model }} accuracy is {{ $value }} (threshold: 0.85)" - impact: "Trading strategy performance severely degraded" - action: "1. Check training data quality 2. Investigate model drift 3. Consider retraining" - - # Model prediction error rate high - - alert: MLPredictionErrorRateHigh - expr: | - 100 * rate(ml_prediction_errors_total[5m]) / - rate(ml_predictions_total[5m]) > 5 - for: 2m - labels: - severity: warning - component: ml - annotations: - summary: "ML prediction error rate high" - description: "Error rate {{ $value }}% (threshold: 5%)" - - # Training iteration time high - - alert: TrainingIterationTimeSlow - expr: histogram_quantile(0.95, rate(ml_training_iteration_seconds_bucket[5m])) > 60 - for: 10m - labels: - severity: warning - component: ml - annotations: - summary: "Training iterations running slow" - description: "p95 iteration time is {{ $value }}s (threshold: 60s)" - impact: "Model training taking longer than expected" - - # Model convergence stalled - - alert: ModelConvergenceStalled - expr: delta(ml_training_loss[10m]) > -0.001 - for: 10m - labels: - severity: warning - component: ml - annotations: - summary: "Model training convergence stalled" - description: "Training loss not improving: {{ $value }}" - impact: "Model may not be learning properly" - action: "1. Check hyperparameters 2. Verify training data 3. Consider stopping" - - - name: ml_training_availability - interval: 10s - rules: - # ML training service down - - alert: MLTrainingServiceDown - expr: up{job="ml_training_service"} == 0 - for: 30s - labels: - severity: high - component: ml - annotations: - summary: "ML training service is DOWN" - description: "Service unreachable for >30 seconds" - impact: "Model training and predictions unavailable" - runbook_url: "https://docs.foxhunt.io/runbooks/ml-service-down" - - # Model loading failures - - alert: ModelLoadingFailures - expr: rate(ml_model_load_failures_total[5m]) > 0.1 - for: 2m - labels: - severity: critical - component: ml - annotations: - summary: "ML model loading failures" - description: "{{ $value }} model load failures/sec" - impact: "Unable to serve predictions" - action: "1. Check model files 2. Verify S3 connectivity 3. Check disk space" - - # Model cache misses high - - alert: ModelCacheMissesHigh - expr: | - 100 * rate(ml_model_cache_misses[5m]) / - (rate(ml_model_cache_hits[5m]) + rate(ml_model_cache_misses[5m])) > 20 - for: 5m - labels: - severity: warning - component: ml - annotations: - summary: "Model cache miss rate high" - description: "Cache miss rate {{ $value }}% (threshold: 20%)" - impact: "Increased model loading latency" - - - name: ml_gpu_resources - interval: 15s - rules: - # GPU utilization low (waste of resources) - - alert: GPUUtilizationLow - expr: ml_gpu_utilization_percent < 30 - for: 10m - labels: - severity: info - component: ml - annotations: - summary: "Low GPU utilization detected" - description: "GPU {{ $labels.gpu_id }} utilization {{ $value }}% (threshold: 30%)" - impact: "Underutilizing expensive GPU resources" - - # GPU utilization critical (near capacity) - - alert: GPUUtilizationCritical - expr: ml_gpu_utilization_percent > 95 - for: 5m - labels: - severity: warning - component: ml - annotations: - summary: "GPU utilization critically high" - description: "GPU {{ $labels.gpu_id }} utilization {{ $value }}% (threshold: 95%)" - impact: "GPU at capacity - potential bottleneck" - - # GPU memory usage high - - alert: GPUMemoryUsageHigh - expr: | - 100 * ml_gpu_memory_used_bytes / ml_gpu_memory_total_bytes > 90 - for: 5m - labels: - severity: warning - component: ml - annotations: - summary: "GPU memory usage high" - description: "GPU {{ $labels.gpu_id }} memory {{ $value }}% (threshold: 90%)" - impact: "Risk of OOM errors during training" - - # GPU memory exhausted (CRITICAL - immediate action required) - - alert: GPUMemoryExhausted - expr: | - 100 * ml_gpu_memory_used_bytes / ml_gpu_memory_total_bytes > 95 - for: 1m - labels: - severity: critical - component: ml - annotations: - summary: "GPU memory critically exhausted" - description: "GPU {{ $labels.gpu_id }} memory {{ $value }}% (threshold: 95%)" - impact: "Imminent OOM - training will crash" - action: "1. Reduce batch size 2. Enable gradient checkpointing 3. Clear GPU cache 4. Kill training job if necessary" - - # GPU temperature high - - alert: GPUTemperatureHigh - expr: ml_gpu_temperature_celsius > 85 - for: 2m - labels: - severity: critical - component: ml - annotations: - summary: "GPU temperature critically high" - description: "GPU {{ $labels.gpu_id }} temperature {{ $value }}Β°C (threshold: 85Β°C)" - impact: "Risk of thermal throttling and hardware damage" - action: "1. Check cooling 2. Reduce workload 3. Monitor temperature" - - # GPU errors detected - - alert: GPUErrorsDetected - expr: rate(ml_gpu_errors_total[5m]) > 0 - for: 1m - labels: - severity: critical - component: ml - annotations: - summary: "GPU errors detected" - description: "GPU {{ $labels.gpu_id }} reporting errors: {{ $value }}/sec" - impact: "GPU hardware issues - training results unreliable" - action: "1. Stop training 2. Check nvidia-smi 3. Contact hardware support" - - - name: ml_model_quality - interval: 30s - rules: - # Model drift detected - - alert: ModelDriftDetected - expr: ml_model_drift_score > 0.15 - for: 5m - labels: - severity: warning - component: ml - annotations: - summary: "ML model drift detected" - description: "Model {{ $labels.model }} drift score {{ $value }} (threshold: 0.15)" - impact: "Model predictions becoming less accurate" - action: "1. Analyze recent data 2. Consider model retraining" - - # Feature distribution shift - - alert: FeatureDistributionShift - expr: ml_feature_distribution_distance > 0.20 - for: 10m - labels: - severity: warning - component: ml - annotations: - summary: "Feature distribution shift detected" - description: "Feature {{ $labels.feature }} distance {{ $value }} (threshold: 0.20)" - impact: "Input data pattern has changed significantly" - - # Prediction confidence low - - alert: PredictionConfidenceLow - expr: histogram_quantile(0.50, rate(ml_prediction_confidence_bucket[5m])) < 0.70 - for: 10m - labels: - severity: warning - component: ml - annotations: - summary: "ML prediction confidence low" - description: "Median prediction confidence {{ $value }} (threshold: 0.70)" - impact: "Model uncertainty high - signals less reliable" - - - name: ml_data_pipeline - interval: 15s - rules: - # Training data stale - - alert: TrainingDataStale - expr: (time() - ml_training_data_last_updated_timestamp) > 86400 - for: 1h - labels: - severity: warning - component: ml - annotations: - summary: "Training data not updated recently" - description: "Training data is {{ $value | humanizeDuration }} old (threshold: 24h)" - impact: "Models training on outdated data" - - # Feature engineering errors - - alert: FeatureEngineeringErrors - expr: rate(ml_feature_engineering_errors_total[5m]) > 1 - for: 2m - labels: - severity: warning - component: ml - annotations: - summary: "Feature engineering errors detected" - description: "{{ $value }} errors/sec in feature pipeline" - impact: "Training data quality degraded" - - # Feature extraction latency high - - alert: FeatureExtractionLatencyHigh - expr: histogram_quantile(0.95, rate(ml_feature_extraction_seconds_bucket[5m])) > 5 - for: 5m - labels: - severity: warning - component: ml - annotations: - summary: "Feature extraction latency high" - description: "p95 extraction time {{ $value }}s (threshold: 5s)" - - - name: ml_storage - interval: 15s - rules: - # S3 connection errors - - alert: S3ConnectionErrors - expr: rate(ml_s3_request_errors_total[5m]) > 1 - for: 2m - labels: - severity: warning - component: ml - annotations: - summary: "S3 connection errors detected" - description: "{{ $value }} S3 errors/sec" - impact: "Model loading/saving operations failing" - - # Model checkpoint save failures - - alert: CheckpointSaveFailures - expr: rate(ml_checkpoint_save_failures_total[10m]) > 0 - for: 5m - labels: - severity: critical - component: ml - annotations: - summary: "Model checkpoint save failures" - description: "{{ $value }} checkpoint save failures" - impact: "Training progress may be lost" - action: "1. Check disk space 2. Verify S3 connectivity 3. Check permissions" - - # Model artifact storage usage high - - alert: ModelStorageUsageHigh - expr: | - 100 * ml_model_storage_used_bytes / ml_model_storage_limit_bytes > 85 - for: 10m - labels: - severity: warning - component: ml - annotations: - summary: "Model artifact storage usage high" - description: "Storage usage {{ $value }}% (threshold: 85%)" - impact: "Risk of storage exhaustion" - action: "1. Clean old model versions 2. Archive unused models" - - - name: ml_training_resources - interval: 15s - rules: - # High memory usage during training - - alert: MLServiceMemoryHigh - expr: | - 100 * process_resident_memory_bytes{job="ml_training_service"} / - node_memory_MemTotal_bytes > 85 - for: 5m - labels: - severity: warning - component: ml - annotations: - summary: "ML service memory usage high" - description: "Memory usage {{ $value }}% (threshold: 85%)" - impact: "Risk of OOM during training" - - # High CPU usage (non-GPU workload) - - alert: MLServiceCPUHigh - expr: | - 100 * rate(process_cpu_seconds_total{job="ml_training_service"}[1m]) > 90 - for: 3m - labels: - severity: warning - component: ml - annotations: - summary: "ML service CPU usage high" - description: "CPU usage {{ $value }}% (threshold: 90%)" - impact: "Possible CPU-bound operation or insufficient GPU utilization" - - - name: ml_automated_pipeline - interval: 30s - rules: - # Automated training job stuck - - alert: AutomatedTrainingJobStuck - expr: | - time() - ml_training_job_last_update_timestamp{status="running"} > 3600 - for: 10m - labels: - severity: critical - component: ml - alert_type: automation - annotations: - summary: "Automated training job stuck" - description: "Job {{ $labels.job_id }} ({{ $labels.model_type }}) no progress for >1 hour" - impact: "Automated ML pipeline blocked, downstream jobs queued" - action: "1. Check job logs 2. Verify GPU availability 3. Consider killing stuck job 4. Restart automation queue" - runbook_url: "https://docs.foxhunt.io/runbooks/stuck-job" - - # Cost budget exceeded - - alert: MonthlyCostBudgetExceeded - expr: ml_monthly_cost_projection_dollars > ml_monthly_budget_dollars - for: 1h - labels: - severity: high - component: ml - alert_type: cost - annotations: - summary: "Monthly cost budget exceeded" - description: "Projected cost ${{ $value }} exceeds budget ${{ $labels.budget }}" - impact: "Cost overrun - budget constraints violated" - action: "1. Review S3 retention policy 2. Pause non-critical training 3. Optimize GPU usage 4. Request budget increase" - - # S3 storage approaching 1TB limit - - alert: S3StorageApproaching1TB - expr: ml_model_storage_used_bytes > 9e11 - for: 1h - labels: - severity: warning - component: ml - alert_type: storage - annotations: - summary: "S3 storage approaching 1TB limit" - description: "S3 storage {{ $value | humanize1024 }} (threshold: 900GB)" - impact: "Storage costs increasing, potential quota limits" - action: "1. Archive old model versions 2. Clean up unused checkpoints 3. Review retention policy" - - # Automated tuning job failure rate high - - alert: AutomatedTuningFailureRateHigh - expr: | - 100 * rate(ml_tuning_job_failures_total[1h]) / - rate(ml_tuning_jobs_total[1h]) > 20 - for: 30m - labels: - severity: warning - component: ml - alert_type: automation - annotations: - summary: "Automated tuning failure rate high" - description: "Tuning job failure rate {{ $value }}% (threshold: 20%)" - impact: "Hyperparameter optimization unreliable" - action: "1. Check search space configuration 2. Review failure logs 3. Verify GPU stability" - - # Data quality degradation - - alert: TrainingDataQualityDegraded - expr: ml_training_data_quality_score < 0.80 - for: 10m - labels: - severity: warning - component: ml - alert_type: data_quality - annotations: - summary: "Training data quality degraded" - description: "Data quality score {{ $value }} (threshold: 0.80)" - impact: "Model training on low-quality data" - action: "1. Check data pipeline 2. Verify feature engineering 3. Review data sources" diff --git a/monitoring/prometheus/alerts/system_alerts.yml b/monitoring/prometheus/alerts/system_alerts.yml deleted file mode 100644 index f8e42e828..000000000 --- a/monitoring/prometheus/alerts/system_alerts.yml +++ /dev/null @@ -1,370 +0,0 @@ -# Prometheus Alert Rules for System-Level Monitoring -# -# Infrastructure, database, cache, and network alerts - -groups: - - name: system_resources - interval: 15s - rules: - # CPU utilization critical across all services - - alert: SystemCPUUtilizationCritical - expr: | - 100 - (avg by (instance) (rate(node_cpu_seconds_total{mode="idle"}[2m])) * 100) > 90 - for: 2m - labels: - severity: critical - component: system - annotations: - summary: "System CPU utilization critically high" - description: "CPU usage {{ $value }}% on {{ $labels.instance }} (threshold: 90%)" - impact: "All services performance severely degraded" - action: "1. Identify CPU-intensive processes 2. Scale resources 3. Kill non-essential services" - - # Memory utilization critical - - alert: SystemMemoryUtilizationCritical - expr: | - 100 * (node_memory_MemTotal_bytes - node_memory_MemAvailable_bytes) / - node_memory_MemTotal_bytes > 95 - for: 2m - labels: - severity: critical - component: system - annotations: - summary: "System memory utilization critically high" - description: "Memory usage {{ $value }}% on {{ $labels.instance }} (threshold: 95%)" - impact: "Risk of OOM kills and system instability" - action: "1. Identify memory leaks 2. Restart services 3. Scale memory" - - # Memory utilization warning - - alert: SystemMemoryUtilizationHigh - expr: | - 100 * (node_memory_MemTotal_bytes - node_memory_MemAvailable_bytes) / - node_memory_MemTotal_bytes > 85 - for: 5m - labels: - severity: warning - component: system - annotations: - summary: "System memory utilization high" - description: "Memory usage {{ $value }}% on {{ $labels.instance }} (threshold: 85%)" - - # Disk space critical - - alert: DiskSpaceCritical - expr: | - 100 * node_filesystem_avail_bytes{fstype!="tmpfs",mountpoint!~".*/proc|.*/sys|.*/run.*"} / - node_filesystem_size_bytes{fstype!="tmpfs",mountpoint!~".*/proc|.*/sys|.*/run.*"} < 5 - for: 5m - labels: - severity: critical - component: system - annotations: - summary: "Disk space critically low" - description: "Only {{ $value }}% space remaining on {{ $labels.mountpoint }} (threshold: 5%)" - impact: "Risk of service failures due to disk full" - action: "1. Clean logs 2. Archive data 3. Expand disk" - - # Disk space low - - alert: DiskSpaceLow - expr: | - 100 * node_filesystem_avail_bytes{fstype!="tmpfs",mountpoint!~".*/proc|.*/sys|.*/run.*"} / - node_filesystem_size_bytes{fstype!="tmpfs",mountpoint!~".*/proc|.*/sys|.*/run.*"} < 15 - for: 10m - labels: - severity: warning - component: system - annotations: - summary: "Disk space low" - description: "{{ $value }}% space remaining on {{ $labels.mountpoint }} (threshold: 15%)" - - # System load high - - alert: SystemLoadHigh - expr: node_load5 / count(node_cpu_seconds_total{mode="idle"}) without(cpu,mode) > 2 - for: 5m - labels: - severity: warning - component: system - annotations: - summary: "System load high" - description: "5-minute load average {{ $value }} on {{ $labels.instance }}" - impact: "System under heavy load - performance degradation likely" - - # Disk I/O utilization high - - alert: DiskIOUtilizationHigh - expr: | - 100 * rate(node_disk_io_time_seconds_total[2m]) > 80 - for: 5m - labels: - severity: warning - component: system - annotations: - summary: "Disk I/O utilization high" - description: "Disk {{ $labels.device }} I/O {{ $value }}% (threshold: 80%)" - impact: "Disk operations may be slow" - - - name: network_connectivity - interval: 10s - rules: - # Network packet loss high - - alert: NetworkPacketLossHigh - expr: | - 100 * rate(node_network_transmit_drop_total[1m]) / - rate(node_network_transmit_packets_total[1m]) > 1 - for: 2m - labels: - severity: critical - component: network - annotations: - summary: "Network packet loss high" - description: "Packet loss {{ $value }}% on {{ $labels.device }} (threshold: 1%)" - impact: "Network reliability degraded" - - # Network errors detected - - alert: NetworkErrorsDetected - expr: rate(node_network_transmit_errs_total[5m]) > 10 - for: 2m - labels: - severity: warning - component: network - annotations: - summary: "Network transmission errors detected" - description: "{{ $value }} errors/sec on {{ $labels.device }}" - - # Network interface down - - alert: NetworkInterfaceDown - expr: node_network_up{device!~"lo|veth.*"} == 0 - for: 1m - labels: - severity: critical - component: network - annotations: - summary: "Network interface down" - description: "Interface {{ $labels.device }} is down on {{ $labels.instance }}" - impact: "Potential connectivity loss" - - - name: database_postgresql - interval: 10s - rules: - # PostgreSQL down - - alert: PostgreSQLDown - expr: up{job="postgresql"} == 0 - for: 30s - labels: - severity: critical - component: database - annotations: - summary: "PostgreSQL database is DOWN" - description: "Database unreachable for >30 seconds" - impact: "All services unable to access persistent data" - action: "1. Check database service 2. Restart if necessary 3. Verify network" - runbook_url: "https://docs.foxhunt.io/runbooks/postgres-down" - - # PostgreSQL connection pool exhaustion - - alert: PostgreSQLConnectionPoolExhaustion - expr: | - 100 * pg_stat_database_numbackends / pg_settings_max_connections > 80 - for: 2m - labels: - severity: critical - component: database - annotations: - summary: "PostgreSQL connection pool nearly exhausted" - description: "{{ $value }}% of max connections used (threshold: 80%)" - impact: "New database connections may fail" - action: "1. Identify connection leaks 2. Restart services 3. Increase max_connections" - - # PostgreSQL replication lag high - - alert: PostgreSQLReplicationLagHigh - expr: pg_replication_lag_seconds > 30 - for: 2m - labels: - severity: warning - component: database - annotations: - summary: "PostgreSQL replication lag high" - description: "Replication lag {{ $value }}s (threshold: 30s)" - impact: "Replica data may be stale" - - # PostgreSQL slow queries - - alert: PostgreSQLSlowQueries - expr: rate(pg_stat_activity_max_tx_duration[5m]) > 10 - for: 5m - labels: - severity: warning - component: database - annotations: - summary: "PostgreSQL slow queries detected" - description: "Long-running queries detected: {{ $value }}s" - impact: "Database performance degraded" - - # PostgreSQL deadlocks - - alert: PostgreSQLDeadlocks - expr: rate(pg_stat_database_deadlocks[5m]) > 0 - for: 2m - labels: - severity: warning - component: database - annotations: - summary: "PostgreSQL deadlocks detected" - description: "{{ $value }} deadlocks/sec" - impact: "Transaction conflicts occurring" - - # PostgreSQL cache hit rate low - - alert: PostgreSQLCacheHitRateLow - expr: | - 100 * pg_stat_database_blks_hit / - (pg_stat_database_blks_hit + pg_stat_database_blks_read) < 90 - for: 10m - labels: - severity: warning - component: database - annotations: - summary: "PostgreSQL cache hit rate low" - description: "Cache hit rate {{ $value }}% (threshold: 90%)" - impact: "Increased disk I/O and slower queries" - - - name: cache_redis - interval: 10s - rules: - # Redis down - - alert: RedisDown - expr: up{job="redis"} == 0 - for: 30s - labels: - severity: critical - component: cache - annotations: - summary: "Redis cache is DOWN" - description: "Redis unreachable for >30 seconds" - impact: "JWT revocation and caching unavailable" - action: "1. Check Redis service 2. Restart if necessary 3. Verify network" - runbook_url: "https://docs.foxhunt.io/runbooks/redis-down" - - # Redis memory usage critical - - alert: RedisMemoryUsageCritical - expr: | - 100 * redis_memory_used_bytes / redis_memory_max_bytes > 95 - for: 2m - labels: - severity: critical - component: cache - annotations: - summary: "Redis memory usage critically high" - description: "Memory usage {{ $value }}% (threshold: 95%)" - impact: "Risk of evictions or OOM" - action: "1. Clear expired keys 2. Increase memory 3. Review retention policies" - - # Redis memory usage high - - alert: RedisMemoryUsageHigh - expr: | - 100 * redis_memory_used_bytes / redis_memory_max_bytes > 85 - for: 5m - labels: - severity: warning - component: cache - annotations: - summary: "Redis memory usage high" - description: "Memory usage {{ $value }}% (threshold: 85%)" - - # Redis evicted keys high - - alert: RedisEvictedKeysHigh - expr: rate(redis_evicted_keys_total[5m]) > 100 - for: 5m - labels: - severity: warning - component: cache - annotations: - summary: "Redis evicting keys due to memory pressure" - description: "Evicting {{ $value }} keys/sec (threshold: 100)" - impact: "Cache effectiveness reduced" - - # Redis connection count high - - alert: RedisConnectionCountHigh - expr: redis_connected_clients > 1000 - for: 5m - labels: - severity: warning - component: cache - annotations: - summary: "Redis connection count high" - description: "{{ $value }} connected clients (threshold: 1000)" - - # Redis hit rate low - - alert: RedisHitRateLow - expr: | - 100 * rate(redis_keyspace_hits_total[5m]) / - (rate(redis_keyspace_hits_total[5m]) + rate(redis_keyspace_misses_total[5m])) < 80 - for: 10m - labels: - severity: warning - component: cache - annotations: - summary: "Redis cache hit rate low" - description: "Hit rate {{ $value }}% (threshold: 80%)" - impact: "Increased backend load" - - - name: service_health - interval: 5s - rules: - # Any service down - - alert: ServiceDown - expr: up{job=~".*_service"} == 0 - for: 30s - labels: - severity: critical - component: service - annotations: - summary: "Service {{ $labels.job }} is DOWN" - description: "Service unreachable for >30 seconds" - impact: "Service unavailable - operations affected" - runbook_url: "https://docs.foxhunt.io/runbooks/service-down" - - # Service restart detected - - alert: ServiceRestarted - expr: changes(process_start_time_seconds{job=~".*_service"}[5m]) > 0 - for: 0s - labels: - severity: info - component: service - annotations: - summary: "Service {{ $labels.job }} restarted" - description: "Service restarted - monitoring for stability" - - - name: monitoring_infrastructure - interval: 30s - rules: - # Prometheus target down - - alert: PrometheusTargetDown - expr: up == 0 - for: 2m - labels: - severity: warning - component: monitoring - annotations: - summary: "Prometheus target {{ $labels.job }} is down" - description: "Cannot scrape metrics from {{ $labels.instance }}" - impact: "Monitoring visibility reduced" - - # Prometheus scrape duration high - - alert: PrometheusScrapeDurationHigh - expr: scrape_duration_seconds > 10 - for: 5m - labels: - severity: warning - component: monitoring - annotations: - summary: "Prometheus scrape duration high" - description: "Scraping {{ $labels.job }} takes {{ $value }}s (threshold: 10s)" - - # Prometheus storage usage high - - alert: PrometheusStorageUsageHigh - expr: | - 100 * prometheus_tsdb_storage_blocks_bytes / - prometheus_tsdb_retention_limit_bytes > 85 - for: 10m - labels: - severity: warning - component: monitoring - annotations: - summary: "Prometheus storage usage high" - description: "Storage {{ $value }}% full (threshold: 85%)" - action: "1. Review retention policies 2. Archive old data 3. Expand storage" diff --git a/monitoring/prometheus/alerts/trading_service_alerts.yml b/monitoring/prometheus/alerts/trading_service_alerts.yml deleted file mode 100644 index 0e877ab02..000000000 --- a/monitoring/prometheus/alerts/trading_service_alerts.yml +++ /dev/null @@ -1,314 +0,0 @@ -# Prometheus Alert Rules for Trading Service -# -# Critical alerts for order execution, position management, and risk controls - -groups: - - name: trading_service_performance - interval: 5s - rules: - # Ultra-critical: Order latency SLA violation - - alert: OrderLatencySLAViolation - expr: histogram_quantile(0.99, rate(trading_order_processing_microseconds_bucket[30s])) > 100 - for: 10s - labels: - severity: critical - component: trading - sla: latency - annotations: - summary: "Trading service order latency exceeded 100ΞΌs SLA" - description: "p99 order processing latency is {{ $value }}ΞΌs (target: <100ΞΌs)" - impact: "Trading performance degraded - potential profit loss" - runbook_url: "https://docs.foxhunt.io/runbooks/trading-latency-high" - - # Order processing latency spike - - alert: OrderLatencySpike - expr: | - ( - histogram_quantile(0.99, rate(trading_order_processing_microseconds_bucket[1m])) / - histogram_quantile(0.99, rate(trading_order_processing_microseconds_bucket[5m] offset 5m)) - ) > 2.0 - for: 30s - labels: - severity: warning - component: trading - annotations: - summary: "Order latency spike detected" - description: "p99 latency increased 2x from baseline: {{ $value }}ΞΌs" - impact: "Temporary performance degradation" - - # High order rejection rate - - alert: HighOrderRejectionRate - expr: | - 100 * rate(trading_orders_rejected_total[5m]) / - rate(trading_orders_submitted_total[5m]) > 5 - for: 2m - labels: - severity: warning - component: trading - annotations: - summary: "High order rejection rate" - description: "{{ $value }}% of orders being rejected (threshold: 5%)" - impact: "Strategy execution quality degraded" - - # Order fill rate collapse - - alert: OrderFillRateCollapse - expr: | - 100 * rate(trading_orders_filled_total[5m]) / - rate(trading_orders_submitted_total[5m]) < 50 - for: 1m - labels: - severity: critical - component: trading - annotations: - summary: "Order fill rate critically low" - description: "Fill rate dropped to {{ $value }}% (threshold: 50%)" - impact: "Severe execution issues - immediate investigation required" - - # Trading throughput degradation - - alert: TradingThroughputLow - expr: rate(trading_orders_submitted_total[1m]) < 100 - for: 2m - labels: - severity: warning - component: trading - annotations: - summary: "Trading throughput below normal" - description: "Processing {{ $value }} orders/sec (expected: >100)" - - - name: trading_service_availability - interval: 5s - rules: - # Trading service down - - alert: TradingServiceDown - expr: up{job="trading_service"} == 0 - for: 5s - labels: - severity: critical - component: trading - annotations: - summary: "CRITICAL: Trading service is DOWN" - description: "Trading service unreachable for >5 seconds" - impact: "All trading operations halted - immediate revenue impact" - action: "1. Check service health 2. Restart if necessary 3. Verify broker connections" - runbook_url: "https://docs.foxhunt.io/runbooks/trading-service-down" - - # Service restart detected - - alert: TradingServiceRestarted - expr: changes(process_start_time_seconds{job="trading_service"}[5m]) > 0 - for: 0s - labels: - severity: info - component: trading - annotations: - summary: "Trading service restarted" - description: "Service restarted at {{ $value | humanizeTimestamp }}" - - # Slow startup time - - alert: TradingServiceSlowStartup - expr: trading_service_startup_duration_seconds > 30 - for: 0s - labels: - severity: warning - component: trading - annotations: - summary: "Trading service startup time excessive" - description: "Service took {{ $value }} seconds to start (threshold: 30s)" - - - name: trading_service_risk - interval: 5s - rules: - # Position limit approaching - - alert: PositionLimitApproaching - expr: | - 100 * trading_current_position_count / trading_max_position_limit > 90 - for: 1m - labels: - severity: warning - component: risk - annotations: - summary: "Position limit approaching" - description: "Using {{ $value }}% of position limit (threshold: 90%)" - impact: "Risk of position limit breach" - - # Risk exposure high - - alert: RiskExposureHigh - expr: | - 100 * trading_current_risk_exposure / trading_max_risk_limit > 85 - for: 30s - labels: - severity: critical - component: risk - annotations: - summary: "Risk exposure approaching limit" - description: "Current risk exposure {{ $value }}% of maximum (threshold: 85%)" - impact: "Approaching risk limits - potential trading halt" - action: "1. Review position sizes 2. Consider risk reduction" - - # VaR utilization critical - - alert: VaRUtilizationCritical - expr: | - 100 * trading_daily_var_used / trading_daily_var_limit > 95 - for: 10s - labels: - severity: critical - component: risk - annotations: - summary: "VaR utilization exceeding 95%" - description: "Daily VaR utilization is {{ $value }}%" - impact: "Approaching daily risk limits - potential trading halt" - - # Drawdown excessive - - alert: DrawdownExcessive - expr: trading_current_drawdown_pct > 20 - for: 1m - labels: - severity: critical - component: risk - annotations: - summary: "Maximum drawdown exceeded" - description: "Current drawdown {{ $value }}% (threshold: 20%)" - impact: "Strategy performance significantly degraded" - - # Position concentration risk - - alert: PositionConcentrationHigh - expr: max(trading_position_concentration_pct) > 40 - for: 1m - labels: - severity: warning - component: risk - annotations: - summary: "Position concentration risk high" - description: "Single position represents {{ $value }}% of portfolio (threshold: 40%)" - - - name: trading_service_connectivity - interval: 5s - rules: - # Broker connection down - - alert: BrokerConnectionDown - expr: broker_connection_status == 0 - for: 5s - labels: - severity: critical - component: connectivity - annotations: - summary: "Broker connection down" - description: "Connection to {{ $labels.broker }} is down" - impact: "Trading capacity reduced - potential execution issues" - action: "1. Check network connectivity 2. Restart connection 3. Switch to backup" - - # Market data feed down - - alert: MarketDataFeedDown - expr: rate(market_data_messages_total[1m]) == 0 - for: 10s - labels: - severity: critical - component: connectivity - annotations: - summary: "Market data feed interruption" - description: "No market data received for >10 seconds" - impact: "Trading decisions based on stale data" - action: "1. Check data feed connection 2. Verify exchange connectivity" - - # Market data latency high - - alert: MarketDataLatencyHigh - expr: histogram_quantile(0.99, rate(market_data_latency_microseconds_bucket[1m])) > 1000 - for: 1m - labels: - severity: warning - component: connectivity - annotations: - summary: "Market data latency high" - description: "p99 data latency is {{ $value }}ΞΌs (threshold: 1ms)" - - - name: trading_service_resources - interval: 15s - rules: - # High memory usage - - alert: TradingServiceMemoryHigh - expr: | - 100 * process_resident_memory_bytes{job="trading_service"} / - node_memory_MemTotal_bytes > 85 - for: 2m - labels: - severity: warning - component: resources - annotations: - summary: "Trading service memory usage high" - description: "Memory usage {{ $value }}% (threshold: 85%)" - impact: "Risk of OOM and service instability" - - # High CPU usage - - alert: TradingServiceCPUHigh - expr: | - 100 * rate(process_cpu_seconds_total{job="trading_service"}[1m]) > 90 - for: 2m - labels: - severity: warning - component: resources - annotations: - summary: "Trading service CPU usage high" - description: "CPU usage {{ $value }}% (threshold: 90%)" - impact: "Performance degradation possible" - - # Database connection pool exhaustion - - alert: DatabaseConnectionPoolExhaustion - expr: | - 100 * trading_db_connections_active / trading_db_connections_max > 80 - for: 1m - labels: - severity: warning - component: resources - annotations: - summary: "Database connection pool nearly exhausted" - description: "Pool utilization: {{ $value }}% (threshold: 80%)" - - # Queue depth high - - alert: OrderQueueDepthHigh - expr: trading_order_queue_depth > 10000 - for: 30s - labels: - severity: warning - component: resources - annotations: - summary: "Order queue depth high" - description: "Queue depth: {{ $value }} orders (threshold: 10K)" - impact: "Risk of order processing delays" - - - name: trading_service_errors - interval: 10s - rules: - # High error rate - - alert: TradingServiceErrorRateHigh - expr: | - 100 * rate(trading_errors_total[5m]) / - rate(trading_requests_total[5m]) > 1 - for: 1m - labels: - severity: critical - component: errors - annotations: - summary: "Trading service error rate high" - description: "Error rate {{ $value }}% (threshold: 1%)" - impact: "Service stability degraded" - - # Database query errors - - alert: DatabaseQueryErrors - expr: rate(trading_db_query_errors_total[5m]) > 1 - for: 1m - labels: - severity: warning - component: errors - annotations: - summary: "Database query errors detected" - description: "{{ $value }} errors/sec" - - # Order validation failures - - alert: OrderValidationFailuresHigh - expr: rate(trading_order_validation_failures_total[5m]) > 10 - for: 2m - labels: - severity: warning - component: errors - annotations: - summary: "High order validation failure rate" - description: "{{ $value }} validation failures/sec" diff --git a/monitoring/prometheus/prometheus.yml b/monitoring/prometheus/prometheus.yml deleted file mode 100644 index f3827da4b..000000000 --- a/monitoring/prometheus/prometheus.yml +++ /dev/null @@ -1,88 +0,0 @@ -# Prometheus Configuration for Foxhunt API Gateway -# -# This configuration scrapes metrics from: -# - API Gateway (authentication, proxy, config) -# - Trading Service -# - Backtesting Service -# - ML Training Service - -global: - scrape_interval: 5s - evaluation_interval: 5s - external_labels: - cluster: 'foxhunt-hft' - env: 'production' - -# Alertmanager configuration -alerting: - alertmanagers: - - static_configs: - - targets: - - 'alertmanager:9093' - -# Load alert rules -rule_files: - - 'alerts/api_gateway_alerts.yml' - - 'alerts/trading_service_alerts.yml' - - 'alerts/ml_training_alerts.yml' - - 'alerts/backtesting_alerts.yml' - - 'alerts/system_alerts.yml' - - 'alerts/ensemble_ml_alerts.yml' - -# Scrape configurations -scrape_configs: - # API Gateway metrics - - job_name: 'api_gateway' - static_configs: - - targets: ['api-gateway:9090'] - metric_relabel_configs: - # Keep only API Gateway metrics - - source_labels: [__name__] - regex: 'api_gateway_.*' - action: keep - - # Trading Service metrics - - job_name: 'trading_service' - static_configs: - - targets: ['trading-service:9091'] - metric_relabel_configs: - - source_labels: [__name__] - regex: 'trading_.*' - action: keep - - # Backtesting Service metrics - - job_name: 'backtesting_service' - static_configs: - - targets: ['backtesting-service:9092'] - metric_relabel_configs: - - source_labels: [__name__] - regex: 'backtesting_.*' - action: keep - - # ML Training Service metrics - - job_name: 'ml_training_service' - static_configs: - - targets: ['ml-training-service:9093'] - metric_relabel_configs: - - source_labels: [__name__] - regex: 'ml_training_.*' - action: keep - - # PostgreSQL exporter (for NOTIFY/config events) - - job_name: 'postgresql' - static_configs: - - targets: ['postgres-exporter:9187'] - - # Redis exporter (for JWT revocation) - - job_name: 'redis' - static_configs: - - targets: ['redis-exporter:9121'] - - # Node exporter (system metrics) - - job_name: 'node' - static_configs: - - targets: - - 'api-gateway-node:9100' - - 'trading-service-node:9100' - - 'backtesting-service-node:9100' - - 'ml-training-service-node:9100' diff --git a/monitoring/prometheus/trading_service_metrics.yml b/monitoring/prometheus/trading_service_metrics.yml deleted file mode 100644 index 50542f7df..000000000 --- a/monitoring/prometheus/trading_service_metrics.yml +++ /dev/null @@ -1,339 +0,0 @@ -# Prometheus Metric Definitions for Trading Service ML Operations -# -# This file documents all ML trading metrics exposed by the trading service. -# These metrics complement existing metrics in ml_training_alerts.yml and -# ensemble_ml_alerts.yml for comprehensive ML production monitoring. - -groups: - - name: ml_trading_prediction_metrics - interval: 10s - rules: - # ML Predictions Total Counter - # Tracks volume of predictions by model, symbol, and action - # Labels: model_id, symbol, action - - record: ml_predictions_total - expr: ml_predictions_total - labels: - component: ml_trading - metric_type: counter - annotations: - description: "Total ML predictions by model, symbol, and action type" - usage: "Track prediction volume and action distribution" - - # ML Prediction Confidence Histogram - # Distribution of confidence scores (0.0-1.0) - # Labels: model_id - # Buckets: 0.1, 0.3, 0.5, 0.7, 0.8, 0.9, 0.95, 1.0 - - record: ml_predictions_confidence - expr: ml_predictions_confidence - labels: - component: ml_trading - metric_type: histogram - annotations: - description: "Distribution of ML model prediction confidence scores" - usage: "Monitor model uncertainty and confidence patterns" - alert_threshold: "P50 < 0.7 indicates low confidence" - - # ML Prediction Accuracy Gauge - # Percentage accuracy by model (0-100) - # Labels: model_id - # Updated: Hourly from PostgreSQL - - record: ml_prediction_accuracy - expr: ml_prediction_accuracy - labels: - component: ml_trading - metric_type: gauge - annotations: - description: "ML model prediction accuracy percentage (updated hourly)" - usage: "Track model performance over time" - target: ">55% for production deployment" - - # Ensemble Votes Total Counter - # Number of ensemble voting events - # Labels: symbol - - record: ml_ensemble_votes_total - expr: ml_ensemble_votes_total - labels: - component: ml_trading - metric_type: counter - annotations: - description: "Total ensemble voting events by symbol" - usage: "Track ensemble decision frequency" - - # Model Last Prediction Timestamp - # Unix epoch seconds of last prediction - # Labels: model_id - - record: ml_model_last_prediction_time - expr: ml_model_last_prediction_time - labels: - component: ml_trading - metric_type: gauge - annotations: - description: "Unix timestamp of last prediction (staleness detection)" - usage: "Alert if time() - ml_model_last_prediction_time > 3600" - - - name: ml_trading_order_metrics - interval: 10s - rules: - # ML Orders Submitted Counter - # Total orders submitted to exchange - # Labels: model_id, symbol - - record: ml_orders_submitted_total - expr: ml_orders_submitted_total - labels: - component: ml_trading - metric_type: counter - annotations: - description: "Total ML-generated orders submitted to exchange" - usage: "Track order submission volume by model" - - # ML Orders Filled Counter - # Successfully filled orders - # Labels: model_id, symbol - - record: ml_orders_filled_total - expr: ml_orders_filled_total - labels: - component: ml_trading - metric_type: counter - annotations: - description: "Total ML-generated orders successfully filled" - usage: "Calculate fill rate: filled_total / submitted_total" - - # ML Orders Rejected Counter - # Rejected orders with reasons - # Labels: model_id, symbol, reason - - record: ml_orders_rejected_total - expr: ml_orders_rejected_total - labels: - component: ml_trading - metric_type: counter - annotations: - description: "Total ML-generated orders rejected with reason" - usage: "Monitor rejection patterns and reasons" - reasons: "risk_limit, insufficient_margin, invalid_price, market_closed" - - - name: ml_trading_performance_metrics - interval: 10s - rules: - # ML Model Sharpe Ratio Gauge - # Risk-adjusted returns - # Labels: model_id - - record: ml_model_sharpe_ratio - expr: ml_model_sharpe_ratio - labels: - component: ml_trading - metric_type: gauge - annotations: - description: "ML model Sharpe ratio (risk-adjusted returns)" - usage: "Monitor risk-adjusted performance" - target: ">1.5 for production trading" - calculation: "(avg_return - risk_free_rate) / stddev * sqrt(252)" - - # ML Model Win Rate Gauge - # Percentage of profitable trades (0.0-1.0) - # Labels: model_id - - record: ml_model_win_rate - expr: ml_model_win_rate - labels: - component: ml_trading - metric_type: gauge - annotations: - description: "ML model win rate (percentage of profitable trades)" - usage: "Track profitability success rate" - target: ">0.55 (55% win rate)" - - # ML Model Average Return Gauge - # Average dollars per trade - # Labels: model_id - - record: ml_model_avg_return - expr: ml_model_avg_return - labels: - component: ml_trading - metric_type: gauge - annotations: - description: "ML model average return per trade (dollars)" - usage: "Track per-trade profitability" - - # ML Model Inference Latency Histogram - # Microseconds per prediction - # Labels: model_id - # Buckets: 10, 50, 100, 500, 1000, 5000, 10000 ΞΌs - - record: ml_model_inference_latency - expr: ml_model_inference_latency - labels: - component: ml_trading - metric_type: histogram - annotations: - description: "ML model inference latency in microseconds" - usage: "Monitor prediction speed for real-time trading" - target: "P99 < 1000ΞΌs (1ms)" - - # ML Model Cumulative PnL Gauge - # Total profit/loss (dollars) - # Labels: model_id - - record: ml_model_cumulative_pnl - expr: ml_model_cumulative_pnl - labels: - component: ml_trading - metric_type: gauge - annotations: - description: "ML model cumulative profit/loss in dollars" - usage: "Track total profitability since deployment" - - # ML Model Maximum Drawdown Gauge - # Worst peak-to-trough decline (dollars) - # Labels: model_id - - record: ml_model_max_drawdown - expr: ml_model_max_drawdown - labels: - component: ml_trading - metric_type: gauge - annotations: - description: "ML model maximum drawdown in dollars" - usage: "Risk metric for capital preservation" - - - name: ml_trading_ensemble_metrics - interval: 10s - rules: - # Ensemble Agreement Rate Gauge - # Model agreement percentage (0.0-1.0) - # Labels: symbol - - record: ml_ensemble_agreement_rate - expr: ml_ensemble_agreement_rate - labels: - component: ml_trading - metric_type: gauge - annotations: - description: "Ensemble model agreement rate" - usage: "Monitor model consensus" - interpretation: "1.0=all agree, 0.0=all disagree" - alert_threshold: "<0.5 indicates high uncertainty" - - # Ensemble Disagreement Events Counter - # High disagreement occurrences - # Labels: symbol, threshold (0.5, 0.7, 0.9) - - record: ml_ensemble_disagreement_events - expr: ml_ensemble_disagreement_events - labels: - component: ml_trading - metric_type: counter - annotations: - description: "Count of high ensemble disagreement events" - usage: "Track model conflict frequency" - thresholds: "0.5, 0.7, 0.9" - indicators: "Regime shift, data quality issues, strategy conflicts" - -# ============================================================================ -# Derived Metrics (Computed from Base Metrics) -# ============================================================================ - - - name: ml_trading_derived_metrics - interval: 10s - rules: - # Order Fill Rate (percentage) - # Ratio of filled to submitted orders - - record: ml_order_fill_rate - expr: | - 100 * ( - sum by (model_id, symbol) (rate(ml_orders_filled_total[5m])) - / - sum by (model_id, symbol) (rate(ml_orders_submitted_total[5m])) - ) - labels: - component: ml_trading - metric_type: derived - annotations: - description: "Order fill rate percentage by model and symbol" - usage: "Monitor order execution quality" - target: ">90% fill rate" - - # Order Rejection Rate (percentage) - # Ratio of rejected to submitted orders - - record: ml_order_rejection_rate - expr: | - 100 * ( - sum by (model_id, symbol) (rate(ml_orders_rejected_total[5m])) - / - sum by (model_id, symbol) (rate(ml_orders_submitted_total[5m])) - ) - labels: - component: ml_trading - metric_type: derived - annotations: - description: "Order rejection rate percentage by model and symbol" - usage: "Identify problematic models or risk issues" - alert_threshold: ">10% rejection rate" - - # Prediction Rate (predictions per second) - # Velocity of predictions - - record: ml_prediction_rate - expr: | - sum by (model_id, symbol) (rate(ml_predictions_total[1m])) - labels: - component: ml_trading - metric_type: derived - annotations: - description: "Prediction velocity (predictions per second)" - usage: "Monitor model activity levels" - - # Average Prediction Confidence (P50) - # Median confidence score - - record: ml_avg_prediction_confidence - expr: | - histogram_quantile(0.50, sum by (model_id, le) (rate(ml_predictions_confidence_bucket[5m]))) - labels: - component: ml_trading - metric_type: derived - annotations: - description: "Median prediction confidence by model" - usage: "Track typical model uncertainty" - alert_threshold: "<0.7 indicates low confidence" - - # Model Inference Latency P99 - # 99th percentile latency - - record: ml_inference_latency_p99 - expr: | - histogram_quantile(0.99, sum by (model_id, le) (rate(ml_model_inference_latency_bucket[5m]))) - labels: - component: ml_trading - metric_type: derived - annotations: - description: "P99 inference latency by model (microseconds)" - usage: "Monitor worst-case inference speed" - target: "<1000ΞΌs (1ms)" - -# ============================================================================ -# Query Examples for Grafana Dashboards -# ============================================================================ - -# Model Performance Comparison (Sharpe Ratio) -# Query: ml_model_sharpe_ratio > 1.0 -# Panel: Table with model_id and value - -# Order Fill Rate by Model -# Query: ml_order_fill_rate -# Panel: Time series graph with model_id legend - -# Ensemble Disagreement Heatmap -# Query: ml_ensemble_disagreement_rate{symbol="ES.FUT"} -# Panel: Heatmap over time - -# Prediction Volume by Action -# Query: sum by (action) (rate(ml_predictions_total[5m])) -# Panel: Pie chart (buy/sell/hold distribution) - -# Model PnL Leaderboard -# Query: topk(5, ml_model_cumulative_pnl) -# Panel: Bar chart of top 5 models by PnL - -# Inference Latency Distribution -# Query: sum by (le) (rate(ml_model_inference_latency_bucket[5m])) -# Panel: Heatmap histogram - -# Prediction Confidence Over Time -# Query: ml_avg_prediction_confidence -# Panel: Time series with alert threshold annotation - -# High Disagreement Events (Rate) -# Query: rate(ml_ensemble_disagreement_events{threshold="0.7"}[5m]) -# Panel: Counter gauge with alert threshold diff --git a/monitoring/server.rs b/monitoring/server.rs deleted file mode 100644 index a8c1cf608..000000000 --- a/monitoring/server.rs +++ /dev/null @@ -1,420 +0,0 @@ -//! Prometheus Metrics HTTP Server -//! -//! Provides HTTP endpoint for Prometheus to scrape metrics from the Foxhunt trading system. - -use std::net::SocketAddr; -use std::sync::Arc; -use std::fmt; -use std::error::Error as StdError; -use axum::{ - extract::State, - http::StatusCode, - response::{IntoResponse, Response}, - routing::get, - Router, -}; -use prometheus::{Encoder, TextEncoder, gather}; -use tokio::net::TcpListener; -use tracing::{info, error, warn}; -use serde::{Deserialize, Serialize}; - -use crate::monitoring::metrics::FoxhuntMetrics; - -/// Metrics server configuration -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct MetricsServerConfig { - /// Server bind address - pub bind_address: String, - /// Server port - pub port: u16, - /// Enable basic authentication - pub enable_auth: bool, - /// Basic auth username (if auth enabled) - pub auth_username: Option, - /// Basic auth password (if auth enabled) - pub auth_password: Option, - /// Enable detailed system metrics - pub enable_system_metrics: bool, - /// Metrics endpoint path - pub metrics_path: String, -} - -impl Default for MetricsServerConfig { - fn default() -> Self { - Self { - bind_address: "0.0.0.0".to_string(), - port: 9090, - enable_auth: false, - auth_username: None, - auth_password: None, - enable_system_metrics: true, - metrics_path: "/metrics".to_string(), - } - } -} - -/// Metrics server state -#[derive(Debug, Clone)] -pub struct MetricsServerState { - config: MetricsServerConfig, - metrics: Arc, -} - -/// Prometheus metrics HTTP server -pub struct MetricsServer { - config: MetricsServerConfig, - metrics: Arc, -} - -impl MetricsServer { - /// Create new metrics server - pub fn new(config: MetricsServerConfig, metrics: Arc) -> Self { - Self { config, metrics } - } - - /// Start the metrics server - pub async fn start(&self) -> Result<(), Box> { - let addr = format!("{}:{}", self.config.bind_address, self.config.port); - let socket_addr: SocketAddr = addr.parse()?; - - let state = MetricsServerState { - config: self.config.clone(), - metrics: self.metrics.clone(), - }; - - let app = self.create_router(state); - let listener = TcpListener::bind(socket_addr).await?; - - info!("πŸš€ Prometheus metrics server starting on http://{}", addr); - info!("πŸ“Š Metrics endpoint: http://{}{}", addr, self.config.metrics_path); - - axum::serve(listener, app).await?; - - Ok(()) - } - - /// Create the router with all endpoints - fn create_router(&self, state: MetricsServerState) -> Router { - let metrics_path = state.config.metrics_path.clone(); - - Router::new() - .route(&metrics_path, get(metrics_handler)) - .route("/health", get(health_handler)) - .route("/", get(root_handler)) - .with_state(state) - } -} - -/// Handler for the main metrics endpoint -async fn metrics_handler( - State(state): State, -) -> Result { - let start_time = std::time::Instant::now(); - - // Gather all metrics - let metric_families = gather(); - - // Encode to Prometheus text format - let encoder = TextEncoder::new(); - let mut buffer = Vec::new(); - - encoder.encode(&metric_families, &mut buffer) - .map_err(|e| MetricsError::EncodingError(e.to_string()))?; - - let metrics_text = String::from_utf8(buffer) - .map_err(|e| MetricsError::EncodingError(e.to_string()))?; - - // Add custom metrics summary if enabled - let response_body = if state.config.enable_system_metrics { - let system_metrics = collect_system_metrics().await; - format!("{}\n{}", metrics_text, system_metrics) - } else { - metrics_text - }; - - let duration = start_time.elapsed(); - - // Log slow metrics collection - if duration.as_millis() > 100 { - warn!("Slow metrics collection: {}ms", duration.as_millis()); - } - - Ok(( - StatusCode::OK, - [("Content-Type", "text/plain; version=0.0.4; charset=utf-8")], - response_body, - ).into_response()) -} - -/// Handler for health check endpoint -async fn health_handler() -> impl IntoResponse { - let health_status = HealthStatus { - status: "healthy".to_string(), - timestamp: chrono::Utc::now(), - version: env!("CARGO_PKG_VERSION").to_string(), - uptime_seconds: get_uptime_seconds(), - }; - - (StatusCode::OK, serde_json::to_string(&health_status).unwrap_or_else(|_| - "{\"status\":\"healthy\"}".to_string())) -} - -/// Handler for root endpoint -async fn root_handler(State(state): State) -> impl IntoResponse { - let html = format!( - r#" - - - Foxhunt Metrics Server - - - -

🦊 Foxhunt HFT Trading System

-

Prometheus Metrics Server

- -
-

Server Status: Running

-

Version: {}

-

Metrics Endpoint: {}

-

Health Check: /health

-
- - πŸ“Š View Metrics - ❀️ Health Check - -

Available Metrics

-
    -
  • foxhunt_orders_total - Total orders processed
  • -
  • foxhunt_latency_microseconds - System latency distribution
  • -
  • foxhunt_position_value_usd - Current position values
  • -
  • foxhunt_ml_predictions_total - ML predictions generated
  • -
  • foxhunt_risk_breaches_total - Risk limit breaches
  • -
  • foxhunt_throughput_ops_per_second - System throughput
  • -
  • And many more...
  • -
- -

Integration

-

Add this target to your Prometheus configuration:

-
-scrape_configs:
-  - job_name: 'foxhunt-trading'
-    static_configs:
-      - targets: ['{}:{}']
-    scrape_interval: 5s
-    metrics_path: '{}'
- -"#, - env!("CARGO_PKG_VERSION"), - state.config.metrics_path, - state.config.metrics_path, - state.config.bind_address, - state.config.port, - state.config.metrics_path - ); - - (StatusCode::OK, [("Content-Type", "text/html")], html) -} - -/// Collect additional system metrics -async fn collect_system_metrics() -> String { - let mut system_metrics = Vec::new(); - - // Add timestamp - let timestamp = chrono::Utc::now().timestamp_millis(); - system_metrics.push(format!( - "# HELP foxhunt_metrics_collection_timestamp_ms Timestamp when metrics were collected\n\ - # TYPE foxhunt_metrics_collection_timestamp_ms gauge\n\ - foxhunt_metrics_collection_timestamp_ms {}", - timestamp - )); - - // Add uptime - let uptime = get_uptime_seconds(); - system_metrics.push(format!( - "# HELP foxhunt_uptime_seconds System uptime in seconds\n\ - # TYPE foxhunt_uptime_seconds counter\n\ - foxhunt_uptime_seconds {}", - uptime - )); - - // Add memory info (if available) - if let Ok(memory_info) = get_memory_info() { - system_metrics.push(format!( - "# HELP foxhunt_memory_total_bytes Total system memory\n\ - # TYPE foxhunt_memory_total_bytes gauge\n\ - foxhunt_memory_total_bytes {}\n\ - # HELP foxhunt_memory_available_bytes Available system memory\n\ - # TYPE foxhunt_memory_available_bytes gauge\n\ - foxhunt_memory_available_bytes {}", - memory_info.total, memory_info.available - )); - } - - system_metrics.join("\n") -} - -/// Get system uptime in seconds -fn get_uptime_seconds() -> u64 { - // Simplified uptime calculation - // In production, this would read from /proc/uptime or use system calls - static START_TIME: std::sync::LazyLock = - std::sync::LazyLock::new(|| std::time::Instant::now()); - - START_TIME.elapsed().as_secs() -} - -/// Memory information structure -#[derive(Debug)] -struct MemoryInfo { - total: u64, - available: u64, -} - -/// Get memory information (Linux-specific) -fn get_memory_info() -> Result { - #[cfg(target_os = "linux")] - { - use std::fs; - let meminfo = fs::read_to_string("/proc/meminfo")?; - - let mut total = 0u64; - let mut available = 0u64; - - for line in meminfo.lines() { - if line.starts_with("MemTotal:") { - if let Some(value) = line.split_whitespace().nth(1) { - total = value.parse::().unwrap_or(0) * 1024; // Convert KB to bytes - } - } else if line.starts_with("MemAvailable:") { - if let Some(value) = line.split_whitespace().nth(1) { - available = value.parse::().unwrap_or(0) * 1024; // Convert KB to bytes - } - } - } - - Ok(MemoryInfo { total, available }) - } - - #[cfg(not(target_os = "linux"))] - { - // Fallback for non-Linux systems - Ok(MemoryInfo { - total: 8 * 1024 * 1024 * 1024, // 8GB default - available: 4 * 1024 * 1024 * 1024, // 4GB default - }) - } -} - -/// Health status structure -#[derive(Debug, Serialize)] -struct HealthStatus { - status: String, - timestamp: chrono::DateTime, - version: String, - uptime_seconds: u64, -} - -/// Metrics server errors -#[derive(Debug)] -pub enum MetricsError { - EncodingError(String), - AuthenticationFailed, - ServerError(String), -} - -impl fmt::Display for MetricsError { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - MetricsError::EncodingError(msg) => write!(f, "Encoding error: {}", msg), - MetricsError::AuthenticationFailed => write!(f, "Authentication failed"), - MetricsError::ServerError(msg) => write!(f, "Server error: {}", msg), - } - } -} - -impl StdError for MetricsError {} - -impl IntoResponse for MetricsError { - fn into_response(self) -> Response { - let (status, message) = match self { - MetricsError::EncodingError(msg) => (StatusCode::INTERNAL_SERVER_ERROR, msg), - MetricsError::AuthenticationFailed => (StatusCode::UNAUTHORIZED, "Authentication required".to_string()), - MetricsError::ServerError(msg) => (StatusCode::INTERNAL_SERVER_ERROR, msg), - }; - - (status, message).into_response() - } -} - -/// Start metrics server with default configuration -pub async fn start_metrics_server() -> Result<(), Box> { - let config = MetricsServerConfig::default(); - let metrics = Arc::new(FoxhuntMetrics::new()); - let server = MetricsServer::new(config, metrics); - - info!("Starting Foxhunt metrics server..."); - server.start().await -} - -/// Start metrics server with custom configuration -pub async fn start_metrics_server_with_config( - config: MetricsServerConfig, -) -> Result<(), Box> { - let metrics = Arc::new(FoxhuntMetrics::new()); - let server = MetricsServer::new(config, metrics); - server.start().await -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn test_config_defaults() { - let config = MetricsServerConfig::default(); - assert_eq!(config.port, 9090); - assert_eq!(config.bind_address, "0.0.0.0"); - assert_eq!(config.metrics_path, "/metrics"); - assert!(!config.enable_auth); - } - - #[test] - fn test_memory_info() { - // Should not panic on any platform - let _result = get_memory_info(); - } - - #[test] - fn test_uptime() { - let uptime = get_uptime_seconds(); - assert!(uptime >= 0); - } - - #[tokio::test] - async fn test_system_metrics_collection() { - let metrics = collect_system_metrics().await; - assert!(metrics.contains("foxhunt_uptime_seconds")); - assert!(metrics.contains("foxhunt_metrics_collection_timestamp_ms")); - } - - #[tokio::test] - async fn test_health_handler() { - let response = health_handler().await; - // Should not panic and return a response - let _response = response.into_response(); - } -} \ No newline at end of file