diff --git a/crates/ml/src/hyperopt/adapters/dqn.rs b/crates/ml/src/hyperopt/adapters/dqn.rs index ae13fa3f3..113b64e29 100644 --- a/crates/ml/src/hyperopt/adapters/dqn.rs +++ b/crates/ml/src/hyperopt/adapters/dqn.rs @@ -2084,7 +2084,7 @@ fn calculate_diversity_penalty(action_distribution: &[f64; 3]) -> f64 { .collect(); let entropy = calculate_action_entropy(&*action_counts); - let max_entropy = (9.0_f64).log2(); // ~3.170 + let max_entropy = (3.0_f64).log2(); // ~1.585 (3 buckets: BUY/SELL/HOLD) // Smooth quadratic penalty: -5.0 * (1 - entropy/max_entropy)² // At entropy=0: -5.0, at entropy=max: 0.0 @@ -4237,12 +4237,12 @@ mod tests { assert!(penalty_mid > -1.0, "Medium entropy mild: {}", penalty_mid); let penalty_uniform = calculate_diversity_penalty(&[0.33, 0.33, 0.34]); - assert!(penalty_uniform.abs() < 0.1, "Uniform ~0: {}", penalty_uniform); + assert!(penalty_uniform.abs() < 0.1, "Uniform 3-bucket ~0: {}", penalty_uniform); - // Monotonic - assert!(penalty_uniform > penalty_mid); - assert!(penalty_mid > penalty_low); - assert!(penalty_low > penalty_zero); + // Monotonic: uniform > mid > low > zero (less negative = better) + assert!(penalty_uniform > penalty_mid, "uniform > mid: {} > {}", penalty_uniform, penalty_mid); + assert!(penalty_mid > penalty_low, "mid > low: {} > {}", penalty_mid, penalty_low); + assert!(penalty_low > penalty_zero, "low > zero: {} > {}", penalty_low, penalty_zero); } #[test]