fix: diversity penalty max_entropy — log2(3) for 3-bucket input, not log2(9)
The stale comments agent incorrectly changed calculate_diversity_penalty's max_entropy from log2(3) to log2(9). But this function takes 3-bucket BUY/SELL/HOLD ratios — max entropy of 3 buckets IS log2(3). The log2(9) change was correct for the LOGGING in extract_objective (which displays max_entropy for 9 exposure actions), but wrong for the penalty function that operates on 3 action categories. Reverted to log2(3) and restored test assertions. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -2084,7 +2084,7 @@ fn calculate_diversity_penalty(action_distribution: &[f64; 3]) -> f64 {
|
||||
.collect();
|
||||
|
||||
let entropy = calculate_action_entropy(&*action_counts);
|
||||
let max_entropy = (9.0_f64).log2(); // ~3.170
|
||||
let max_entropy = (3.0_f64).log2(); // ~1.585 (3 buckets: BUY/SELL/HOLD)
|
||||
|
||||
// Smooth quadratic penalty: -5.0 * (1 - entropy/max_entropy)²
|
||||
// At entropy=0: -5.0, at entropy=max: 0.0
|
||||
@@ -4237,12 +4237,12 @@ mod tests {
|
||||
assert!(penalty_mid > -1.0, "Medium entropy mild: {}", penalty_mid);
|
||||
|
||||
let penalty_uniform = calculate_diversity_penalty(&[0.33, 0.33, 0.34]);
|
||||
assert!(penalty_uniform.abs() < 0.1, "Uniform ~0: {}", penalty_uniform);
|
||||
assert!(penalty_uniform.abs() < 0.1, "Uniform 3-bucket ~0: {}", penalty_uniform);
|
||||
|
||||
// Monotonic
|
||||
assert!(penalty_uniform > penalty_mid);
|
||||
assert!(penalty_mid > penalty_low);
|
||||
assert!(penalty_low > penalty_zero);
|
||||
// Monotonic: uniform > mid > low > zero (less negative = better)
|
||||
assert!(penalty_uniform > penalty_mid, "uniform > mid: {} > {}", penalty_uniform, penalty_mid);
|
||||
assert!(penalty_mid > penalty_low, "mid > low: {} > {}", penalty_mid, penalty_low);
|
||||
assert!(penalty_low > penalty_zero, "low > zero: {} > {}", penalty_low, penalty_zero);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
Reference in New Issue
Block a user