//! PPO Hyperopt Policy Learning Rate Upper Bound Tests //! //! Validates that the policy learning rate upper bound has been narrowed //! from 1e-3 to 5e-5 based on DQN breakthrough findings (best policy LR = 1e-6). //! //! Context: DQN hyperopt showed that policy LR of 1e-6 was optimal. The old //! upper bound of 1e-3 was 1000x higher than optimal, wasting hyperopt trials. //! New upper bound of 5e-5 is 50x higher than best (still allows exploration) //! but 20x lower than old bound (avoids catastrophic forgetting). use ml::hyperopt::adapters::ppo::PPOParams; use ml::hyperopt::traits::ParameterSpace; #[test] fn test_policy_lr_upper_bound_narrowed() { let bounds = PPOParams::continuous_bounds(); let policy_lr_bounds = bounds[0]; // policy_learning_rate is 1st parameter // Upper bound should be ln(5e-5) = -9.903 let expected_upper = 5e-5_f64.ln(); let actual_upper = policy_lr_bounds.1; assert!( (actual_upper - expected_upper).abs() < 1e-6, "Policy LR upper bound incorrect. Expected ln(5e-5) = {:.6}, got {:.6}", expected_upper, actual_upper ); } #[test] fn test_policy_lr_at_new_upper_bound() { let params = PPOParams::from_continuous(&[ 5e-5_f64.ln(), // policy_lr (NEW UPPER BOUND) 0.001_f64.ln(), // value_lr 0.2, // clip_epsilon 1.0, // value_loss_coeff 0.01_f64.ln(), // entropy_coeff 128.0, // minibatch_size ]) .unwrap(); assert!( (params.policy_learning_rate - 0.00005).abs() < 1e-8, "Policy LR at upper bound incorrect. Expected 5e-5, got {}", params.policy_learning_rate ); } #[test] fn test_policy_lr_prevents_catastrophic_forgetting() { // Upper bound (5e-5) should be 50x higher than best (1e-6) // but 20x lower than old bound (1e-3) let bounds = PPOParams::continuous_bounds(); let upper = bounds[0].1.exp(); let _lower = bounds[0].0.exp(); assert!( upper < 1e-3, "Upper bound should be less than old bound (1e-3), got {}", upper ); assert!( upper > 1e-6, "Upper bound should be more than lower bound (1e-6), got {}", upper ); // Check that upper bound is in realistic range (1e-5 to 1e-4) assert!( upper >= 1e-5 && upper <= 1e-4, "Upper bound should be in realistic range [1e-5, 1e-4], got {}", upper ); // Check ratio: upper/best should be 50x (5e-5 / 1e-6 = 50) let best_policy_lr = 1e-6; let ratio = upper / best_policy_lr; assert!( (ratio - 50.0).abs() < 0.1, "Upper/best ratio should be 50x, got {:.1}x", ratio ); } #[test] fn test_policy_lr_lower_bound_unchanged() { // Lower bound should remain at 1e-6 (proven optimal by DQN hyperopt) let bounds = PPOParams::continuous_bounds(); let lower = bounds[0].0.exp(); assert!( (lower - 1e-6).abs() < 1e-9, "Lower bound should be 1e-6 (optimal from DQN hyperopt), got {}", lower ); } #[test] fn test_value_lr_bounds_unchanged() { // Value LR bounds should remain unchanged (1e-5 to 1e-3) // NOTE: Upper bound was expanded from 1e-3 to 5e-3 in separate change let bounds = PPOParams::continuous_bounds(); let value_lr_bounds = bounds[1]; // value_learning_rate is 2nd parameter let lower = value_lr_bounds.0.exp(); let upper = value_lr_bounds.1.exp(); assert!( (lower - 1e-5).abs() < 1e-9, "Value LR lower bound should be 1e-5, got {}", lower ); assert!( (upper - 5e-3).abs() < 1e-9, "Value LR upper bound should be 5e-3, got {}", upper ); } #[test] fn test_other_bounds_unchanged() { // Verify other parameter bounds are unchanged let bounds = PPOParams::continuous_bounds(); // clip_epsilon: (0.1, 0.3) assert_eq!( bounds[2], (0.1, 0.3), "Clip epsilon bounds changed unexpectedly" ); // value_loss_coeff: (0.5, 2.0) assert_eq!( bounds[3], (0.5, 2.0), "Value loss coeff bounds changed unexpectedly" ); // entropy_coeff: (ln(0.001), ln(0.1)) let entropy_lower = bounds[4].0.exp(); let entropy_upper = bounds[4].1.exp(); assert!( (entropy_lower - 0.001).abs() < 1e-6, "Entropy coeff lower bound changed" ); assert!( (entropy_upper - 0.1).abs() < 1e-6, "Entropy coeff upper bound changed" ); // minibatch_size: (64, 230) assert_eq!( bounds[5], (64.0, 230.0), "Minibatch size bounds changed unexpectedly" ); } #[test] fn test_roundtrip_with_new_bounds() { // Test that roundtrip conversion works correctly with new bounds let params = PPOParams { policy_learning_rate: 2.5e-5, // In middle of new range value_learning_rate: 5e-4, clip_epsilon: 0.15, value_loss_coeff: 1.2, entropy_coeff: 0.02, minibatch_size: 128, }; let continuous = params.to_continuous(); let recovered = PPOParams::from_continuous(&continuous).unwrap(); assert!( (recovered.policy_learning_rate - params.policy_learning_rate).abs() < 1e-10, "Policy LR roundtrip failed" ); assert!( (recovered.value_learning_rate - params.value_learning_rate).abs() < 1e-10, "Value LR roundtrip failed" ); assert_eq!( recovered.minibatch_size, params.minibatch_size, "Minibatch size roundtrip failed" ); } #[test] fn test_parameter_count() { // Verify we have exactly 6 parameters let bounds = PPOParams::continuous_bounds(); assert_eq!( bounds.len(), 6, "PPOParams should have 6 continuous parameters" ); let param_names = PPOParams::param_names(); assert_eq!( param_names.len(), 6, "PPOParams should have 6 parameter names" ); }