#![allow( unused_variables, clippy::assertions_on_constants, clippy::len_zero, clippy::useless_vec )] //! Unit Tests for Ensemble Metrics Module //! //! This test suite validates Prometheus metrics for ensemble ML monitoring //! including aggregation latency, confidence, disagreement, weights, and A/B testing. use prometheus::core::Collector; use trading_service::ensemble_metrics::*; #[test] fn test_ensemble_aggregation_latency_metric() { let metric = &*ENSEMBLE_AGGREGATION_LATENCY_US; let desc = metric.desc(); assert!( desc.len() > 0, "Ensemble aggregation latency histogram should exist" ); // Test different aggregation methods metric .with_label_values(&["weighted_average"]) .observe(10.0); metric.with_label_values(&["majority_vote"]).observe(5.0); metric .with_label_values(&["confidence_weighted"]) .observe(15.0); // Verify buckets: [1.0, 5.0, 10.0, 25.0, 50.0, 100.0] let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_ensemble_confidence_metric() { let metric = &*ENSEMBLE_CONFIDENCE; let desc = metric.desc(); assert!(desc.len() > 0, "Ensemble confidence gauge should exist"); // Test confidence scores for different symbols (0.0-1.0) metric.with_label_values(&["ES.FUT"]).set(0.85); metric.with_label_values(&["NQ.FUT"]).set(0.92); metric.with_label_values(&["ZN.FUT"]).set(0.78); metric.with_label_values(&["6E.FUT"]).set(0.68); let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_ensemble_disagreement_rate_metric() { let metric = &*ENSEMBLE_DISAGREEMENT_RATE; let desc = metric.desc(); assert!( desc.len() > 0, "Ensemble disagreement rate gauge should exist" ); // Test disagreement rates (0.0 = all agree, 1.0 = all disagree) metric.with_label_values(&["ES.FUT"]).set(0.15); // Low disagreement metric.with_label_values(&["NQ.FUT"]).set(0.55); // High disagreement (alert threshold) metric.with_label_values(&["ZN.FUT"]).set(0.05); // Very low disagreement metric.with_label_values(&["6E.FUT"]).set(0.75); // Very high disagreement let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_ensemble_predictions_counter() { let metric = &*ENSEMBLE_PREDICTIONS_TOTAL; let desc = metric.desc(); assert!(desc.len() > 0, "Ensemble predictions counter should exist"); // Test predictions by action type and symbol metric.with_label_values(&["buy", "ES.FUT"]).inc(); metric.with_label_values(&["sell", "ES.FUT"]).inc(); metric.with_label_values(&["hold", "ES.FUT"]).inc(); metric.with_label_values(&["buy", "NQ.FUT"]).inc_by(5.0); metric.with_label_values(&["sell", "ZN.FUT"]).inc_by(3.0); metric.with_label_values(&["hold", "6E.FUT"]).inc_by(10.0); let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_ensemble_model_weight_metric() { let metric = &*ENSEMBLE_MODEL_WEIGHT; let desc = metric.desc(); assert!(desc.len() > 0, "Ensemble model weight gauge should exist"); // Test model weights (should sum to ~1.0 for each symbol) // ES.FUT weights metric.with_label_values(&["DQN", "ES.FUT"]).set(0.25); metric.with_label_values(&["PPO", "ES.FUT"]).set(0.30); metric.with_label_values(&["MAMBA-2", "ES.FUT"]).set(0.20); metric.with_label_values(&["TFT", "ES.FUT"]).set(0.25); // Sum = 1.00 ✓ // NQ.FUT weights (different distribution) metric.with_label_values(&["DQN", "NQ.FUT"]).set(0.20); metric.with_label_values(&["PPO", "NQ.FUT"]).set(0.35); metric.with_label_values(&["TFT", "NQ.FUT"]).set(0.45); // Sum = 1.00 ✓ let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_ensemble_high_disagreement_counter() { let metric = &*ENSEMBLE_HIGH_DISAGREEMENT_TOTAL; let desc = metric.desc(); assert!(desc.len() > 0, "High disagreement counter should exist"); // Test high disagreement events (symbol, threshold) metric.with_label_values(&["ES.FUT", "0.5"]).inc(); metric.with_label_values(&["NQ.FUT", "0.7"]).inc_by(3.0); metric.with_label_values(&["ZN.FUT", "0.9"]).inc_by(2.0); let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_ensemble_model_pnl_attribution_histogram() { let metric = &*ENSEMBLE_MODEL_PNL_CONTRIBUTION; let desc = metric.desc(); assert!(desc.len() > 0, "PnL attribution histogram should exist"); // Test PnL contributions from different models metric.with_label_values(&["DQN", "ES.FUT"]).observe(125.50); metric.with_label_values(&["PPO", "ES.FUT"]).observe(-45.25); metric.with_label_values(&["TFT", "NQ.FUT"]).observe(350.75); metric .with_label_values(&["MAMBA-2", "ZN.FUT"]) .observe(-120.00); // Buckets: [-1000, -500, -100, 0, 100, 500, 1000] let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_checkpoint_swaps_counter() { let metric = &*CHECKPOINT_SWAPS_TOTAL; let desc = metric.desc(); assert!(desc.len() > 0, "Checkpoint swaps counter should exist"); // Test checkpoint hot-swaps (model_id, status) metric.with_label_values(&["DQN", "success"]).inc(); metric.with_label_values(&["PPO", "failure"]).inc(); metric.with_label_values(&["TFT", "rollback"]).inc(); metric .with_label_values(&["MAMBA-2", "success"]) .inc_by(2.0); let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_ab_test_assignments_counter() { let metric = &*AB_TEST_ASSIGNMENTS_TOTAL; let desc = metric.desc(); assert!(desc.len() > 0, "A/B test assignments counter should exist"); // Test A/B test assignments (test_id, group: control, treatment_a, treatment_b) metric .with_label_values(&["test_001", "control"]) .inc_by(50.0); metric .with_label_values(&["test_001", "treatment_a"]) .inc_by(25.0); metric .with_label_values(&["test_001", "treatment_b"]) .inc_by(25.0); metric .with_label_values(&["test_002", "control"]) .inc_by(40.0); metric .with_label_values(&["test_002", "treatment_a"]) .inc_by(60.0); let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_ab_test_metric_difference_gauge() { let metric = &*AB_TEST_METRIC_DIFF; let desc = metric.desc(); assert!( desc.len() > 0, "A/B test metric difference gauge should exist" ); // Test metric differences (test_id, metric_name, percentage difference) metric .with_label_values(&["test_001", "sharpe_ratio"]) .set(15.5); // +15.5% improvement metric .with_label_values(&["test_001", "max_drawdown"]) .set(-8.3); // -8.3% improvement metric.with_label_values(&["test_002", "win_rate"]).set(3.2); // +3.2% improvement metric .with_label_values(&["test_002", "profit_factor"]) .set(22.7); // +22.7% improvement let collected = metric.collect(); assert!(!collected.is_empty(), "Should have collected metrics"); } #[test] fn test_all_ensemble_metrics_registered() { // Verify all ensemble metrics initialize without panics let _ = &*ENSEMBLE_AGGREGATION_LATENCY_US; let _ = &*ENSEMBLE_CONFIDENCE; let _ = &*ENSEMBLE_DISAGREEMENT_RATE; let _ = &*ENSEMBLE_PREDICTIONS_TOTAL; let _ = &*ENSEMBLE_MODEL_WEIGHT; let _ = &*ENSEMBLE_HIGH_DISAGREEMENT_TOTAL; let _ = &*ENSEMBLE_MODEL_PNL_CONTRIBUTION; let _ = &*CHECKPOINT_SWAPS_TOTAL; let _ = &*AB_TEST_ASSIGNMENTS_TOTAL; let _ = &*AB_TEST_METRIC_DIFF; assert!(true, "All 10 ensemble metrics initialized successfully"); } #[test] fn test_ensemble_metrics_independence() { // Verify metrics for different symbols are tracked independently ENSEMBLE_CONFIDENCE.with_label_values(&["ES.FUT"]).set(0.85); ENSEMBLE_CONFIDENCE.with_label_values(&["NQ.FUT"]).set(0.92); ENSEMBLE_DISAGREEMENT_RATE .with_label_values(&["ES.FUT"]) .set(0.15); ENSEMBLE_DISAGREEMENT_RATE .with_label_values(&["NQ.FUT"]) .set(0.55); // Predictions should be tracked separately ENSEMBLE_PREDICTIONS_TOTAL .with_label_values(&["buy", "ES.FUT"]) .inc(); ENSEMBLE_PREDICTIONS_TOTAL .with_label_values(&["buy", "NQ.FUT"]) .inc(); let confidence_collected = ENSEMBLE_CONFIDENCE.collect(); let disagreement_collected = ENSEMBLE_DISAGREEMENT_RATE.collect(); let predictions_collected = ENSEMBLE_PREDICTIONS_TOTAL.collect(); assert!(!confidence_collected.is_empty()); assert!(!disagreement_collected.is_empty()); assert!(!predictions_collected.is_empty()); } #[test] fn test_model_weight_distribution() { // Test realistic weight distributions across models let symbols = vec!["ES.FUT", "NQ.FUT", "ZN.FUT"]; let models = vec!["DQN", "PPO", "MAMBA-2", "TFT"]; for symbol in &symbols { // Simulate adaptive weights ENSEMBLE_MODEL_WEIGHT .with_label_values(&["DQN", symbol]) .set(0.25); ENSEMBLE_MODEL_WEIGHT .with_label_values(&["PPO", symbol]) .set(0.30); ENSEMBLE_MODEL_WEIGHT .with_label_values(&["MAMBA-2", symbol]) .set(0.20); ENSEMBLE_MODEL_WEIGHT .with_label_values(&["TFT", symbol]) .set(0.25); } let collected = ENSEMBLE_MODEL_WEIGHT.collect(); assert!( !collected.is_empty(), "Model weights should be tracked per symbol" ); } #[test] fn test_aggregation_latency_buckets() { // Test latency observations in different buckets // Buckets: [1.0, 5.0, 10.0, 25.0, 50.0, 100.0] ENSEMBLE_AGGREGATION_LATENCY_US .with_label_values(&["weighted_average"]) .observe(0.5); // < 1μs (very fast) ENSEMBLE_AGGREGATION_LATENCY_US .with_label_values(&["weighted_average"]) .observe(7.5); // 5-10μs ENSEMBLE_AGGREGATION_LATENCY_US .with_label_values(&["weighted_average"]) .observe(18.0); // 10-25μs ENSEMBLE_AGGREGATION_LATENCY_US .with_label_values(&["weighted_average"]) .observe(35.0); // 25-50μs ENSEMBLE_AGGREGATION_LATENCY_US .with_label_values(&["weighted_average"]) .observe(120.0); // > 100μs (alert threshold) let collected = ENSEMBLE_AGGREGATION_LATENCY_US.collect(); assert!( !collected.is_empty(), "Histogram should capture all buckets" ); } #[test] fn test_high_disagreement_threshold() { // Test that high disagreement counter tracks events > 0.5 threshold // Simulate disagreement detection logic let disagreement_rates = vec![ ("ES.FUT", 0.15), // Low - no alert ("NQ.FUT", 0.55), // High - alert ✓ ("ZN.FUT", 0.05), // Low - no alert ("6E.FUT", 0.75), // Very high - alert ✓ ]; for (symbol, rate) in disagreement_rates { ENSEMBLE_DISAGREEMENT_RATE .with_label_values(&[symbol]) .set(rate); if rate >= 0.5 { ENSEMBLE_HIGH_DISAGREEMENT_TOTAL .with_label_values(&[symbol, "0.5"]) .inc(); } if rate >= 0.7 { ENSEMBLE_HIGH_DISAGREEMENT_TOTAL .with_label_values(&[symbol, "0.7"]) .inc(); } } let collected = ENSEMBLE_HIGH_DISAGREEMENT_TOTAL.collect(); assert!( !collected.is_empty(), "High disagreement events should be counted" ); } #[test] fn test_pnl_attribution_positive_and_negative() { // Test both profitable and unprofitable model contributions // Profitable models ENSEMBLE_MODEL_PNL_CONTRIBUTION .with_label_values(&["DQN", "ES.FUT"]) .observe(250.50); ENSEMBLE_MODEL_PNL_CONTRIBUTION .with_label_values(&["PPO", "NQ.FUT"]) .observe(180.25); // Unprofitable models (negative P&L) ENSEMBLE_MODEL_PNL_CONTRIBUTION .with_label_values(&["TFT", "ZN.FUT"]) .observe(-75.30); ENSEMBLE_MODEL_PNL_CONTRIBUTION .with_label_values(&["MAMBA-2", "6E.FUT"]) .observe(-150.00); let collected = ENSEMBLE_MODEL_PNL_CONTRIBUTION.collect(); assert!(!collected.is_empty(), "Should track both gains and losses"); } #[test] fn test_checkpoint_swap_scenarios() { // Test different checkpoint swap outcomes (model_id, status) // Successful swaps CHECKPOINT_SWAPS_TOTAL .with_label_values(&["DQN", "success"]) .inc(); CHECKPOINT_SWAPS_TOTAL .with_label_values(&["PPO", "success"]) .inc(); // Failed swaps (model loading error) CHECKPOINT_SWAPS_TOTAL .with_label_values(&["TFT", "failure"]) .inc(); // Rollback swaps (performance degradation detected) CHECKPOINT_SWAPS_TOTAL .with_label_values(&["MAMBA-2", "rollback"]) .inc(); let collected = CHECKPOINT_SWAPS_TOTAL.collect(); assert!(!collected.is_empty(), "Should track all swap outcomes"); } #[test] fn test_ab_test_balanced_assignment() { // Test balanced A/B test assignment (50/50 split) AB_TEST_ASSIGNMENTS_TOTAL .with_label_values(&["test_balanced", "control"]) .inc_by(50.0); AB_TEST_ASSIGNMENTS_TOTAL .with_label_values(&["test_balanced", "treatment_a"]) .inc_by(50.0); // Test imbalanced assignment (70/30 split) AB_TEST_ASSIGNMENTS_TOTAL .with_label_values(&["test_imbalanced", "control"]) .inc_by(70.0); AB_TEST_ASSIGNMENTS_TOTAL .with_label_values(&["test_imbalanced", "treatment_a"]) .inc_by(30.0); let collected = AB_TEST_ASSIGNMENTS_TOTAL.collect(); assert!( !collected.is_empty(), "Should track assignment distributions" ); }