//! Test to verify that PPO hyperopt adapter only samples minibatch_size values //! that divide batch_size=2048 evenly. //! //! This prevents the assertion failure in PPO training: //! ``` //! assert!(config.batch_size % config.mini_batch_size == 0) //! ``` //! //! Valid divisors for batch_size=2048: {64, 128, 256, 512, 1024, 2048} use ml::hyperopt::adapters::ppo::PPOParams; use ml::hyperopt::traits::ParameterSpace; /// Batch size used in PPO training (from PPOConfig in ml/src/hyperopt/adapters/ppo.rs line 384) const BATCH_SIZE: usize = 2048; /// Valid divisors of 2048 (powers of 2 from 64 to 2048) const VALID_DIVISORS: [usize; 6] = [64, 128, 256, 512, 1024, 2048]; #[test] fn test_all_valid_divisors_sample_correctly() { // Test that discrete sampling produces all valid divisors correctly for (idx, &expected_divisor) in VALID_DIVISORS.iter().enumerate() { // Create continuous vector with minibatch_size index let continuous = vec![ 1e-6_f64.ln(), // policy_learning_rate (log scale) 1e-5_f64.ln(), // value_learning_rate (log scale) 0.2, // clip_epsilon 1.0, // value_loss_coeff 0.01_f64.ln(), // entropy_coeff (log scale) idx as f64, // minibatch_size index [0-5] ]; let params = PPOParams::from_continuous(&continuous).expect("Failed to convert from continuous"); assert_eq!( params.minibatch_size, expected_divisor, "Index {} should map to divisor {}, got {}", idx, expected_divisor, params.minibatch_size ); // Verify it divides batch_size evenly assert_eq!( BATCH_SIZE % params.minibatch_size, 0, "Divisor {} does not divide batch_size {} evenly", params.minibatch_size, BATCH_SIZE ); } } #[test] fn test_random_continuous_values_produce_valid_divisors() { // Test that random continuous values in range [0.0, 5.0] always produce valid divisors use rand::Rng; let mut rng = rand::thread_rng(); for _ in 0..100 { // Sample random minibatch_size index in range [0.0, 5.0] let minibatch_idx = rng.gen_range(0.0..=5.0); let continuous = vec![ rng.gen_range(1e-6_f64.ln()..5e-5_f64.ln()), // policy_learning_rate rng.gen_range(1e-5_f64.ln()..5e-3_f64.ln()), // value_learning_rate rng.gen_range(0.1..0.3), // clip_epsilon rng.gen_range(0.5..2.0), // value_loss_coeff rng.gen_range(0.001_f64.ln()..0.1_f64.ln()), // entropy_coeff minibatch_idx, // minibatch_size index ]; let params = PPOParams::from_continuous(&continuous).expect("Failed to convert from continuous"); // Verify minibatch_size is one of the valid divisors assert!( VALID_DIVISORS.contains(¶ms.minibatch_size), "Sampled minibatch_size {} is not in valid divisors {:?}", params.minibatch_size, VALID_DIVISORS ); // Verify it divides batch_size evenly assert_eq!( BATCH_SIZE % params.minibatch_size, 0, "Sampled minibatch_size {} does not divide batch_size {} evenly", params.minibatch_size, BATCH_SIZE ); } } #[test] fn test_roundtrip_preserves_valid_divisors() { // Test that to_continuous() → from_continuous() roundtrip preserves valid divisors for &divisor in &VALID_DIVISORS { let params = PPOParams { policy_learning_rate: 1e-5, value_learning_rate: 1e-4, clip_epsilon: 0.2, value_loss_coeff: 1.0, entropy_coeff: 0.01, minibatch_size: divisor, }; let continuous = params.to_continuous(); let recovered = PPOParams::from_continuous(&continuous).expect("Failed to recover from continuous"); assert_eq!( recovered.minibatch_size, divisor, "Roundtrip failed: {} → continuous → {}", divisor, recovered.minibatch_size ); } } #[test] fn test_boundary_indices_clamp_correctly() { // Test that indices outside [0, 5] clamp to valid divisors // Below range: -1.0 should clamp to index 0 → divisor 64 let continuous_below = vec![ 1e-6_f64.ln(), 1e-5_f64.ln(), 0.2, 1.0, 0.01_f64.ln(), -1.0, // Below range ]; let params_below = PPOParams::from_continuous(&continuous_below).unwrap(); assert_eq!( params_below.minibatch_size, 64, "Index -1.0 should clamp to 64" ); // Above range: 10.0 should clamp to index 5 → divisor 2048 let continuous_above = vec![ 1e-6_f64.ln(), 1e-5_f64.ln(), 0.2, 1.0, 0.01_f64.ln(), 10.0, // Above range ]; let params_above = PPOParams::from_continuous(&continuous_above).unwrap(); assert_eq!( params_above.minibatch_size, 2048, "Index 10.0 should clamp to 2048" ); } #[test] fn test_fractional_indices_round_to_nearest() { // Test that fractional indices round to nearest integer index // 0.4 rounds to 0 → divisor 64 let continuous_0_4 = vec![1e-6_f64.ln(), 1e-5_f64.ln(), 0.2, 1.0, 0.01_f64.ln(), 0.4]; let params_0_4 = PPOParams::from_continuous(&continuous_0_4).unwrap(); assert_eq!( params_0_4.minibatch_size, 64, "Index 0.4 should round to 0 → 64" ); // 0.6 rounds to 1 → divisor 128 let continuous_0_6 = vec![1e-6_f64.ln(), 1e-5_f64.ln(), 0.2, 1.0, 0.01_f64.ln(), 0.6]; let params_0_6 = PPOParams::from_continuous(&continuous_0_6).unwrap(); assert_eq!( params_0_6.minibatch_size, 128, "Index 0.6 should round to 1 → 128" ); // 2.5 rounds to 2 → divisor 256 (banker's rounding, but we'll accept either 2 or 3) let continuous_2_5 = vec![1e-6_f64.ln(), 1e-5_f64.ln(), 0.2, 1.0, 0.01_f64.ln(), 2.5]; let params_2_5 = PPOParams::from_continuous(&continuous_2_5).unwrap(); // Accept either 256 (round to 2) or 512 (round to 3) due to rounding mode assert!( params_2_5.minibatch_size == 256 || params_2_5.minibatch_size == 512, "Index 2.5 should round to 2 or 3, got minibatch_size={}", params_2_5.minibatch_size ); } #[test] fn test_no_invalid_divisors_in_range() { // Verify that NO invalid divisors (96, 160, 192, 230, etc.) can be sampled let invalid_divisors = [96, 160, 192, 230, 320, 400, 500, 1000, 1500]; // Sample 1000 random continuous values use rand::Rng; let mut rng = rand::thread_rng(); for _ in 0..1000 { let continuous = vec![ rng.gen_range(1e-6_f64.ln()..5e-5_f64.ln()), rng.gen_range(1e-5_f64.ln()..5e-3_f64.ln()), rng.gen_range(0.1..0.3), rng.gen_range(0.5..2.0), rng.gen_range(0.001_f64.ln()..0.1_f64.ln()), rng.gen_range(0.0..5.0), // minibatch_size index ]; let params = PPOParams::from_continuous(&continuous).unwrap(); // Verify NO invalid divisors are sampled assert!( !invalid_divisors.contains(¶ms.minibatch_size), "Sampled INVALID divisor {} (should only sample {:?})", params.minibatch_size, VALID_DIVISORS ); } }