//! WAVE 15 AGENT 34: DQN Backtesting Integration Test //! //! This test verifies that backtesting metrics are correctly integrated //! into the DQN hyperopt objective function and that objectives vary //! meaningfully across different hyperparameter configurations. //! //! ## Test Coverage //! //! 1. **Metrics Structure**: Verify DQNMetrics includes all 6 fields //! 2. **Composite Objective**: Verify objective calculation formula //! 3. **Objective Variance**: Verify objectives differ across trials //! 4. **Backtesting Population**: Verify backtesting metrics are populated use ml::hyperopt::adapters::dqn::{DQNMetrics, DQNParams, DQNTrainer}; use ml::hyperopt::traits::{HyperparameterOptimizable, ParameterSpace}; /// Test 1: Verify DQNMetrics structure includes backtesting fields #[test] fn test_dqn_metrics_structure() { let metrics = DQNMetrics { train_loss: 0.5, val_loss: 0.4, avg_q_value: 2.5, final_epsilon: 0.01, epochs_completed: 100, avg_episode_reward: -50.0, buy_action_pct: 0.3, sell_action_pct: 0.3, hold_action_pct: 0.4, gradient_norm: 2.0, q_value_std: 1.5, sharpe_ratio: Some(1.5), // Backtesting metric max_drawdown_pct: Some(-15.0), // Backtesting metric win_rate: Some(60.0), // Backtesting metric }; // Verify all fields are accessible assert_eq!(metrics.sharpe_ratio, Some(1.5)); assert_eq!(metrics.max_drawdown_pct, Some(-15.0)); assert_eq!(metrics.win_rate, Some(60.0)); println!("✓ DQNMetrics structure includes all backtesting fields"); } /// Test 2: Verify composite objective calculation #[test] fn test_composite_objective_calculation() { let metrics = DQNMetrics { train_loss: 0.5, val_loss: 0.4, avg_q_value: 5.0, final_epsilon: 0.01, epochs_completed: 100, avg_episode_reward: 0.0, // Neutral reward (maps to 0.5 score) buy_action_pct: 0.3, sell_action_pct: 0.3, hold_action_pct: 0.4, gradient_norm: 2.0, q_value_std: 1.5, sharpe_ratio: Some(2.0), // 2.0/5.0 = 0.4 score max_drawdown_pct: Some(-10.0), // 10/100 = 0.1 penalty → 0.9 score win_rate: Some(60.0), // 60/100 = 0.6 score }; let objective = DQNTrainer::extract_objective(&metrics); // Expected calculation: // rl_reward_score = (0.0 + 10.0) / 20.0 = 0.5 // sharpe_ratio_score = 2.0 / 5.0 = 0.4 // drawdown_penalty = 10.0 / 100.0 = 0.1 → drawdown_score = 1.0 - 0.1 = 0.9 // win_rate_score = 60.0 / 100.0 = 0.6 // // composite_objective = 0.40 * 0.5 + 0.30 * 0.4 + 0.20 * 0.9 + 0.10 * 0.6 // = 0.20 + 0.12 + 0.18 + 0.06 // = 0.56 // // Final objective = -0.56 (negated for minimization) let expected = -0.56; assert!( (objective - expected).abs() < 0.01, "Objective mismatch: expected {:.4}, got {:.4}", expected, objective ); println!( "✓ Composite objective calculation correct: {:.4}", objective ); } /// Test 3: Verify objective varies across different configurations #[test] fn test_objective_variance_across_configs() { // Configuration 1: Good RL reward, poor backtesting let metrics1 = DQNMetrics { train_loss: 0.5, val_loss: 0.4, avg_q_value: 5.0, final_epsilon: 0.01, epochs_completed: 100, avg_episode_reward: 5.0, // High reward (maps to 0.75 score) buy_action_pct: 0.3, sell_action_pct: 0.3, hold_action_pct: 0.4, gradient_norm: 2.0, q_value_std: 1.5, sharpe_ratio: Some(0.5), // Poor Sharpe (0.1 score) max_drawdown_pct: Some(-30.0), // High drawdown (0.7 score) win_rate: Some(45.0), // Poor win rate (0.45 score) }; // Configuration 2: Poor RL reward, good backtesting let metrics2 = DQNMetrics { train_loss: 0.5, val_loss: 0.4, avg_q_value: 5.0, final_epsilon: 0.01, epochs_completed: 100, avg_episode_reward: -5.0, // Low reward (maps to 0.25 score) buy_action_pct: 0.3, sell_action_pct: 0.3, hold_action_pct: 0.4, gradient_norm: 2.0, q_value_std: 1.5, sharpe_ratio: Some(4.0), // High Sharpe (0.8 score) max_drawdown_pct: Some(-5.0), // Low drawdown (0.95 score) win_rate: Some(75.0), // High win rate (0.75 score) }; // Configuration 3: Balanced performance let metrics3 = DQNMetrics { train_loss: 0.5, val_loss: 0.4, avg_q_value: 5.0, final_epsilon: 0.01, epochs_completed: 100, avg_episode_reward: 0.0, // Neutral reward (0.5 score) buy_action_pct: 0.3, sell_action_pct: 0.3, hold_action_pct: 0.4, gradient_norm: 2.0, q_value_std: 1.5, sharpe_ratio: Some(2.5), // Medium Sharpe (0.5 score) max_drawdown_pct: Some(-15.0), // Medium drawdown (0.85 score) win_rate: Some(60.0), // Medium win rate (0.6 score) }; let obj1 = DQNTrainer::extract_objective(&metrics1); let obj2 = DQNTrainer::extract_objective(&metrics2); let obj3 = DQNTrainer::extract_objective(&metrics3); println!("Objective 1 (good RL, poor backtest): {:.4}", obj1); println!("Objective 2 (poor RL, good backtest): {:.4}", obj2); println!("Objective 3 (balanced): {:.4}", obj3); // Verify objectives are NOT identical assert_ne!(obj1, obj2, "Objectives should vary across configurations"); assert_ne!(obj2, obj3, "Objectives should vary across configurations"); assert_ne!(obj1, obj3, "Objectives should vary across configurations"); // Calculate coefficient of variation (CV) to verify variance let mean = (obj1 + obj2 + obj3) / 3.0; let variance = ((obj1 - mean).powi(2) + (obj2 - mean).powi(2) + (obj3 - mean).powi(2)) / 3.0; let std_dev = variance.sqrt(); let cv = (std_dev / mean.abs()) * 100.0; println!("Mean objective: {:.4}", mean); println!("Std dev: {:.4}", std_dev); println!("Coefficient of variation: {:.2}%", cv); // Verify reasonable variance (CV > 5%) assert!( cv > 5.0, "Coefficient of variation too low: {:.2}% (expected > 5%)", cv ); println!( "✓ Objectives vary meaningfully across configurations (CV={:.2}%)", cv ); } /// Test 4: Verify backtesting metrics are populated (when available) #[test] fn test_backtesting_metrics_populated() { // Scenario 1: Backtesting metrics available let metrics_with_backtest = DQNMetrics { train_loss: 0.5, val_loss: 0.4, avg_q_value: 5.0, final_epsilon: 0.01, epochs_completed: 100, avg_episode_reward: 0.0, buy_action_pct: 0.3, sell_action_pct: 0.3, hold_action_pct: 0.4, gradient_norm: 2.0, q_value_std: 1.5, sharpe_ratio: Some(2.0), max_drawdown_pct: Some(-15.0), win_rate: Some(60.0), }; assert!(metrics_with_backtest.sharpe_ratio.is_some()); assert!(metrics_with_backtest.max_drawdown_pct.is_some()); assert!(metrics_with_backtest.win_rate.is_some()); // Scenario 2: Backtesting metrics unavailable (None) let metrics_without_backtest = DQNMetrics { train_loss: 0.5, val_loss: 0.4, avg_q_value: 5.0, final_epsilon: 0.01, epochs_completed: 100, avg_episode_reward: 0.0, buy_action_pct: 0.3, sell_action_pct: 0.3, hold_action_pct: 0.4, gradient_norm: 2.0, q_value_std: 1.5, sharpe_ratio: None, max_drawdown_pct: None, win_rate: None, }; assert!(metrics_without_backtest.sharpe_ratio.is_none()); assert!(metrics_without_backtest.max_drawdown_pct.is_none()); assert!(metrics_without_backtest.win_rate.is_none()); // Verify objectives differ between scenarios let obj_with = DQNTrainer::extract_objective(&metrics_with_backtest); let obj_without = DQNTrainer::extract_objective(&metrics_without_backtest); println!("Objective with backtesting: {:.4}", obj_with); println!("Objective without backtesting: {:.4}", obj_without); // With backtesting: uses actual metrics // Without backtesting: uses neutral fallback (0.5) assert_ne!( obj_with, obj_without, "Objectives should differ based on backtesting availability" ); println!("✓ Backtesting metrics correctly handled (Some vs None)"); } /// Test 5: Verify parameter space bounds (sanity check) #[test] fn test_parameter_space_consistency() { let bounds = DQNParams::continuous_bounds(); // Verify we have 6 parameters (Wave 13 configuration) assert_eq!(bounds.len(), 6, "Expected 6 parameters in search space"); // Verify all bounds are valid (lower < upper) for (i, (lower, upper)) in bounds.iter().enumerate() { assert!( lower < upper, "Invalid bounds for parameter {}: [{}, {}]", i, lower, upper ); } println!("✓ Parameter space bounds are consistent"); } /// Test 6: Verify objective normalization (prevents outliers) #[test] fn test_objective_normalization() { // Test outlier reward (should be clamped) let metrics_outlier = DQNMetrics { train_loss: 0.5, val_loss: 0.4, avg_q_value: 5.0, final_epsilon: 0.01, epochs_completed: 100, avg_episode_reward: 100.0, // Extreme outlier (clamped to 10.0 → score 1.0) buy_action_pct: 0.3, sell_action_pct: 0.3, hold_action_pct: 0.4, gradient_norm: 2.0, q_value_std: 1.5, sharpe_ratio: Some(2.0), max_drawdown_pct: Some(-15.0), win_rate: Some(60.0), }; let obj_outlier = DQNTrainer::extract_objective(&metrics_outlier); // Test normal reward (within expected range) let metrics_normal = DQNMetrics { train_loss: 0.5, val_loss: 0.4, avg_q_value: 5.0, final_epsilon: 0.01, epochs_completed: 100, avg_episode_reward: 10.0, // Max expected reward (score 1.0) buy_action_pct: 0.3, sell_action_pct: 0.3, hold_action_pct: 0.4, gradient_norm: 2.0, q_value_std: 1.5, sharpe_ratio: Some(2.0), max_drawdown_pct: Some(-15.0), win_rate: Some(60.0), }; let obj_normal = DQNTrainer::extract_objective(&metrics_normal); // After clamping, both should have same objective (reward clamped to 1.0) assert!( (obj_outlier - obj_normal).abs() < 0.001, "Outlier should be clamped to same objective as max: outlier={:.4}, normal={:.4}", obj_outlier, obj_normal ); println!("✓ Objective normalization prevents outlier domination"); }