Files
foxhunt/crates/ml/tests/preprocessing_test.rs
jgrusewski cf91106e32 fix: migrate 44 test files from Candle to native CUDA — zero test compile errors
Complete Candle→cudarc migration for all test code. The workspace
now compiles clean with `cargo check --workspace --tests` (0 errors)
and `cargo clippy --workspace --lib -D warnings` (0 errors).

Migration patterns applied across all files:
- Tensor → GpuTensor (from_host, zeros, randn, full)
- Device → MlDevice (cuda, cuda_if_available, new_cuda)
- All GpuTensor ops now take &Arc<CudaStream>
- VarMap/VarBuilder → GpuVarStore or removed
- DType removed (everything f32)
- Candle autograd tests (Var, GradStore, backward) → #[ignore]
- Preprocessing tests → host-side Vec<f32> (CPU-side by design)
- PPO hidden state → host-side Vec<f32> slices
- UnifiedTrainable: forward_loss(&[f32], &[f32]) → f64

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-03-19 10:02:26 +01:00

378 lines
12 KiB
Rust

#![allow(
clippy::assertions_on_constants,
clippy::assertions_on_result_states,
clippy::clone_on_copy,
clippy::decimal_literal_representation,
clippy::doc_markdown,
clippy::empty_line_after_doc_comments,
clippy::field_reassign_with_default,
clippy::get_unwrap,
clippy::identity_op,
clippy::inconsistent_digit_grouping,
clippy::indexing_slicing,
clippy::integer_division,
clippy::len_zero,
clippy::let_underscore_must_use,
clippy::manual_div_ceil,
clippy::manual_let_else,
clippy::manual_range_contains,
clippy::modulo_arithmetic,
clippy::needless_range_loop,
clippy::non_ascii_literal,
clippy::redundant_clone,
clippy::shadow_reuse,
clippy::shadow_same,
clippy::shadow_unrelated,
clippy::single_match_else,
clippy::str_to_string,
clippy::string_slice,
clippy::tests_outside_test_module,
clippy::too_many_lines,
clippy::unnecessary_wraps,
clippy::unseparated_literal_suffix,
clippy::use_debug,
clippy::useless_vec,
clippy::wildcard_enum_match_arm,
clippy::else_if_without_else,
clippy::expect_used,
clippy::missing_const_for_fn,
clippy::similar_names,
clippy::type_complexity,
clippy::collapsible_else_if,
clippy::doc_lazy_continuation,
clippy::items_after_test_module,
clippy::map_clone,
clippy::multiple_unsafe_ops_per_block,
clippy::unwrap_or_default,
clippy::assign_op_pattern,
clippy::needless_borrow,
clippy::println_empty_string,
clippy::unnecessary_cast,
clippy::used_underscore_binding,
clippy::create_dir,
clippy::implicit_saturating_sub,
clippy::exit,
clippy::expect_fun_call,
clippy::too_many_arguments,
clippy::unnecessary_map_or,
clippy::unwrap_used,
dead_code,
unused_imports,
unused_variables,
clippy::cloned_ref_to_slice_refs,
clippy::neg_multiply,
clippy::while_let_loop,
clippy::bool_assert_comparison,
clippy::excessive_precision,
clippy::trivially_copy_pass_by_ref,
clippy::op_ref,
clippy::redundant_closure,
clippy::unnecessary_lazy_evaluations,
clippy::if_then_some_else_none,
clippy::unnecessary_to_owned,
clippy::single_component_path_imports,
)]
//! Preprocessing module tests
//!
//! Tests for data preprocessing functions that transform raw OHLCV data
//! into stationary log returns with windowed normalization.
//!
//! Test coverage:
//! 1. Log returns transformation
//! 2. Windowed normalization (z-score)
//! 3. Outlier clipping (+-N sigma)
//! 4. Full preprocessing pipeline
//!
//! Wave 14 - Agent 28
//!
//! All preprocessing functions operate on host `&[f32]` slices — no GPU needed.
#[test]
fn test_log_returns_transformation() {
// GIVEN: Price series [100, 105, 103, 110]
let prices = [100.0f32, 105.0, 103.0, 110.0];
// WHEN: Log returns calculated
let returns =
ml::preprocessing::compute_log_returns(&prices).expect("Failed to compute log returns");
// THEN: Should be log(P_t / P_{t-1})
// Expected: [0.0 (placeholder), 0.04879, -0.01942, 0.06567]
assert_eq!(returns.len(), 4, "Should have 4 return values");
// First value should be 0.0 (placeholder for missing value)
let val0 = returns[0];
assert!(
(val0 - 0.0).abs() < 0.0001,
"First return should be 0.0 (placeholder), got {}",
val0
);
// Second value: log(105/100) ~ 0.04879
let val1 = returns[1];
assert!(
(val1 - 0.04879).abs() < 0.0001,
"Second return should be ~0.04879, got {}",
val1
);
// Third value: log(103/105) ~ -0.01942
let val2 = returns[2];
assert!(
(val2 - (-0.01942)).abs() < 0.001,
"Third return should be ~-0.01942, got {}",
val2
);
// Fourth value: log(110/103) ~ 0.06567
let val3 = returns[3];
assert!(
(val3 - 0.06567).abs() < 0.001,
"Fourth return should be ~0.06567, got {}",
val3
);
}
#[test]
fn test_windowed_normalization() {
// GIVEN: Returns with changing volatility
let returns = [0.01f32, 0.02, 0.10, 0.15, 0.01, 0.02];
// WHEN: Windowed normalization applied (window=3)
let normalized =
ml::preprocessing::windowed_normalize(&returns, 3).expect("Failed to normalize");
// THEN: Each window should have mean~0, std~1
assert_eq!(normalized.len(), 6, "Should have 6 normalized values");
// Check that all values are roughly normalized (should be in range -5 to +5 for z-scores)
let max_abs = normalized.iter().map(|x| x.abs()).fold(0.0_f32, f32::max);
assert!(
max_abs < 5.0,
"All normalized values should be bounded, max abs = {}",
max_abs
);
// Verify normalization is working by checking the last window [0.10, 0.15, 0.01]
// After normalization, they should have different z-scores
let last_three = &normalized[3..6];
// Calculate mean and variance of normalized values in last window
let mean_normalized: f32 = last_three.iter().sum::<f32>() / last_three.len() as f32;
let var_normalized: f32 = last_three
.iter()
.map(|&x| (x - mean_normalized).powi(2))
.sum::<f32>()
/ last_three.len() as f32;
// Normalized values should have mean close to 0 and variance close to 1
// (within the specific window that was used for normalization)
assert!(
mean_normalized.abs() < 0.5,
"Normalized mean should be close to 0, got {}",
mean_normalized
);
assert!(
(var_normalized - 1.0).abs() < 1.5,
"Normalized variance should be close to 1.0, got {}",
var_normalized
);
}
#[test]
fn test_outlier_clipping() {
// GIVEN: Returns with extreme outliers
let returns = [0.01f32, 0.02, 10.0, 0.01, -8.0, 0.02];
// WHEN: Clip to +-3 sigma
let clipped = ml::preprocessing::clip_outliers(&returns, 3.0).expect("Failed to clip outliers");
assert_eq!(clipped.len(), 6, "Should have 6 clipped values");
// THEN: Outliers should be clipped
// Calculate mean and std of original data
let mean: f32 = returns.iter().sum::<f32>() / returns.len() as f32;
let variance: f32 =
returns.iter().map(|&x| (x - mean).powi(2)).sum::<f32>() / returns.len() as f32;
let std = variance.sqrt();
let upper_bound = mean + 3.0 * std;
let lower_bound = mean - 3.0 * std;
// All values should be within bounds
let clipped_min = clipped.iter().copied().fold(f32::INFINITY, f32::min);
let clipped_max = clipped.iter().copied().fold(f32::NEG_INFINITY, f32::max);
assert!(
clipped_min >= lower_bound && clipped_max <= upper_bound,
"All values should be within [{}, {}], got min={}, max={}",
lower_bound,
upper_bound,
clipped_min,
clipped_max
);
// Extreme values should have been clipped (they are within the calculated bounds)
// With data [0.01, 0.02, 10.0, 0.01, -8.0, 0.02]:
// Mean ~ 0.343, Std ~ 5.79, so +-3s ~ [-17.03, 17.71]
// Thus 10.0 and -8.0 are actually WITHIN bounds and won't be clipped!
// This is expected behavior - the clipping threshold adapts to data distribution.
// Verify that clipping function is working correctly by checking bounds
let val2 = clipped[2];
assert!(
val2 <= upper_bound,
"Value at index 2 should be <= upper_bound {}, got {}",
upper_bound,
val2
);
let val4 = clipped[4];
assert!(
val4 >= lower_bound,
"Value at index 4 should be >= lower_bound {}, got {}",
lower_bound,
val4
);
}
#[test]
fn test_full_preprocessing_pipeline() {
// GIVEN: Simulated OHLCV data (20 bars for quick test)
// Simulate realistic price movement: trending with some volatility
let mut prices = vec![100.0f32];
for i in 1..20 {
let prev = prices[i - 1];
// Add small random-like changes
let change = if i % 3 == 0 {
1.0
} else if i % 5 == 0 {
-0.5
} else {
0.5
};
prices.push(prev + change);
}
// WHEN: Full preprocessing applied
let config = ml::preprocessing::PreprocessConfig {
window_size: 5,
clip_sigma: 3.0,
use_log_returns: true,
};
let preprocessed = ml::preprocessing::preprocess_prices(&prices, config)
.expect("Failed to preprocess data");
// THEN: Verify properties
assert_eq!(
preprocessed.len(),
20,
"Should have 20 preprocessed values"
);
// 1. Should not have NaNs or Infs
let sum_all: f32 = preprocessed.iter().sum();
assert!(
sum_all.is_finite(),
"Preprocessed data should contain no NaN/Inf values, sum_all = {}",
sum_all
);
// 2. Should be bounded (after normalization and clipping)
let max_abs = preprocessed
.iter()
.map(|x| x.abs())
.fold(0.0_f32, f32::max);
assert!(
max_abs < 10.0,
"All preprocessed values should be bounded, max abs = {}",
max_abs
);
// 3. Skip first value (placeholder) when calculating preprocessed variance
let tail = &preprocessed[1..];
let preprocessed_variance: f32 =
tail.iter().map(|x| x.powi(2)).sum::<f32>() / tail.len() as f32;
// Preprocessed should have more normalized variance
// (Not necessarily smaller, but should be in a reasonable range for normalized data)
assert!(
preprocessed_variance.is_finite() && preprocessed_variance >= 0.0,
"Preprocessed variance should be finite and non-negative, got {}",
preprocessed_variance
);
}
#[test]
fn test_preprocessing_handles_flat_prices() {
// GIVEN: Flat price series (no volatility)
let prices = [100.0f32, 100.0, 100.0, 100.0, 100.0];
// WHEN: Preprocessing applied (with small window for short data)
let config = ml::preprocessing::PreprocessConfig {
window_size: 3, // Use small window for short test data
clip_sigma: 3.0,
use_log_returns: true,
};
let preprocessed = ml::preprocessing::preprocess_prices(&prices, config)
.expect("Failed to preprocess flat prices");
// THEN: Should handle gracefully (all zeros or very small values)
// No NaN/Inf
let sum_all: f32 = preprocessed.iter().sum();
assert!(
sum_all.is_finite(),
"Preprocessed data should contain no NaN/Inf, sum_all = {}",
sum_all
);
// All values should be near zero for flat prices
let max_abs = preprocessed
.iter()
.map(|x| x.abs())
.fold(0.0_f32, f32::max);
assert!(
max_abs < 0.0001,
"Flat prices should produce near-zero returns, max abs = {}",
max_abs
);
}
#[test]
fn test_preprocessing_handles_single_spike() {
// GIVEN: Mostly flat prices with one spike
let prices = [100.0f32, 100.0, 100.0, 150.0, 100.0, 100.0, 100.0];
// WHEN: Preprocessing with aggressive clipping
let config = ml::preprocessing::PreprocessConfig {
window_size: 3,
clip_sigma: 2.0, // More aggressive clipping
use_log_returns: true,
};
let preprocessed =
ml::preprocessing::preprocess_prices(&prices, config).expect("Failed to preprocess");
// THEN: Spike should be clipped/normalized
// No NaN/Inf
let sum_all: f32 = preprocessed.iter().sum();
assert!(
sum_all.is_finite(),
"Preprocessed data should contain no NaN/Inf, sum_all = {}",
sum_all
);
// Find the spike location (index 3 corresponds to 150.0 price)
// The return at index 3 would be log(150/100) ~ 0.405
// After normalization and clipping, it should be bounded
let max_abs = preprocessed
.iter()
.map(|x| x.abs())
.fold(0.0_f32, f32::max);
assert!(
max_abs < 5.0,
"Spike should be clipped/normalized, max abs = {}",
max_abs
);
}