Files
foxhunt/ml/tests/parquet_timestamp_loading_test.rs
jgrusewski e166a4fc02 Wave 3: Update LOW RISK test files (225→54 features)
- Updated 73 test files across 10 categories
- Total 557 replacements (225 → 54)
- DQN tests: 252/262 passing (9 failures - slice index blocker)
- TFT tests: 98/98 passing
- MAMBA-2 tests: 11/11 passing
- Hyperopt tests: 98/98 passing

Critical findings:
- Blocker: ml/src/trainers/dqn.rs:3444 hardcoded slice indices
- Architecture mismatch: extract_current_features() vs extract_current_features_v2()

Wave 3 Agent breakdown:
- Agent 1: DQN test files (12 files)
- Agent 2: PPO test files (2 files)
- Agent 3: TFT test files (6 files)
- Agent 4: MAMBA-2 test files (2 files)
- Agent 5: Feature extraction tests (3 files)
- Agent 6: Integration test files (9 files)
- Agent 7: Data loader test files (3 files)
- Agent 8: Hyperopt test files (1 file)
- Agent 9: Benchmark test files (9 files)
- Agent 10: Utility & misc test files (73 files)

Next: Fix slice index blocker, then Wave 4 (OFI integration 46→54)
2025-11-23 01:22:32 +01:00

216 lines
6.3 KiB
Rust

//! Test Suite for Parquet Timestamp Loading
//!
//! Validates that `load_parquet_data_with_timestamps()` returns:
//! 1. Three vectors with the same length (features, timestamps, bars)
//! 2. Correct count after warmup (after 50 bars warmup)
//! 3. Timestamps monotonically increasing
//! 4. Bars have valid OHLCV data
//!
//! ## Test Data
//! - File: test_data/ES_FUT_unseen.parquet
//! - Warmup: 50 bars
//! - Expected output: Feature vectors with timestamps and bars (all same length)
use ml::data_loaders::parquet_utils::load_parquet_data_with_timestamps;
use std::path::PathBuf;
/// Helper function to find test data file across different working directories
fn find_test_data_file() -> Option<PathBuf> {
let possible_paths = [
"test_data/ES_FUT_unseen.parquet",
"../test_data/ES_FUT_unseen.parquet",
"/home/jgrusewski/Work/foxhunt/test_data/ES_FUT_unseen.parquet",
];
possible_paths
.iter()
.map(|p| PathBuf::from(p))
.find(|p| p.exists())
}
#[test]
fn test_load_parquet_data_with_timestamps() {
// Arrange: Use production unseen data
let path = find_test_data_file()
.expect("Could not find test_data/ES_FUT_unseen.parquet. Tried: [test_data/, ../test_data/, /home/jgrusewski/Work/foxhunt/test_data/]");
let warmup_bars = 50;
// Act: Load features, timestamps, and bars
let result = load_parquet_data_with_timestamps(&path, warmup_bars);
assert!(
result.is_ok(),
"Failed to load Parquet data with timestamps: {:?}",
result.err()
);
let (features, timestamps, bars) = result.unwrap();
// Assert 1: All three vectors have the same length
assert_eq!(
features.len(),
timestamps.len(),
"Features and timestamps must have the same length"
);
assert_eq!(
features.len(),
bars.len(),
"Features and bars must have the same length"
);
// Assert 2: Reasonable count (should have data after warmup)
assert!(
features.len() > 0,
"Expected at least some feature vectors after warmup, got {}",
features.len()
);
println!(
"Loaded {} feature vectors after {} warmup bars",
features.len(),
warmup_bars
);
// Assert 3: Timestamps are monotonically increasing
assert!(
timestamps.windows(2).all(|w| w[0] <= w[1]),
"Timestamps must be monotonically increasing"
);
// Assert 4: Bars have valid OHLCV data
for (i, bar) in bars.iter().enumerate() {
assert!(
bar.open > 0.0 && bar.open.is_finite(),
"Bar {} has invalid open price: {}",
i,
bar.open
);
assert!(
bar.high > 0.0 && bar.high.is_finite(),
"Bar {} has invalid high price: {}",
i,
bar.high
);
assert!(
bar.low > 0.0 && bar.low.is_finite(),
"Bar {} has invalid low price: {}",
i,
bar.low
);
assert!(
bar.close > 0.0 && bar.close.is_finite(),
"Bar {} has invalid close price: {}",
i,
bar.close
);
assert!(
bar.volume > 0.0 && bar.volume.is_finite(),
"Bar {} has invalid volume: {}",
i,
bar.volume
);
}
// Assert 5: Features have correct dimensionality (54)
for (i, feature_vec) in features.iter().enumerate() {
assert_eq!(
feature_vec.len(),
54,
"Feature vector {} has incorrect length: {}",
i,
feature_vec.len()
);
// Validate no NaN/Inf in features
for (feat_idx, &value) in feature_vec.iter().enumerate() {
assert!(
value.is_finite(),
"Feature vector {} has NaN/Inf at index {}: {}",
i,
feat_idx,
value
);
}
}
println!(
"✅ Successfully loaded {} feature vectors with timestamps and OHLCV bars",
features.len()
);
}
#[test]
fn test_timestamps_match_bars() {
// Arrange: Load data
let path = find_test_data_file().expect("Could not find test_data/ES_FUT_unseen.parquet");
let warmup_bars = 50;
// Act
let (_features, timestamps, bars) =
load_parquet_data_with_timestamps(&path, warmup_bars).expect("Failed to load Parquet data");
// Assert: Timestamps from return value match timestamps from bars
for (i, (ts, bar)) in timestamps.iter().zip(bars.iter()).enumerate() {
assert_eq!(
ts, &bar.timestamp,
"Timestamp mismatch at index {}: returned timestamp {:?} != bar timestamp {:?}",
i, ts, bar.timestamp
);
}
println!(
"✅ All {} timestamps match corresponding bars",
timestamps.len()
);
}
#[test]
fn test_features_match_bars() {
// Arrange: Load data
let path = find_test_data_file().expect("Could not find test_data/ES_FUT_unseen.parquet");
let warmup_bars = 50;
// Act
let (features, _timestamps, bars) =
load_parquet_data_with_timestamps(&path, warmup_bars).expect("Failed to load Parquet data");
// Assert: Number of features matches number of bars
assert_eq!(
features.len(),
bars.len(),
"Number of feature vectors must match number of bars"
);
// Assert: OHLCV values in bars are valid (basic sanity checks)
for (i, bar) in bars.iter().enumerate() {
assert!(
bar.open > 0.0 && bar.open.is_finite(),
"Bar {} has invalid open price: {}",
i,
bar.open
);
assert!(
bar.high >= bar.low,
"Bar {} has high < low: high={}, low={}",
i,
bar.high,
bar.low
);
assert!(
bar.close >= bar.low && bar.close <= bar.high,
"Bar {} has close outside [low, high]: close={}, low={}, high={}",
i,
bar.close,
bar.low,
bar.high
);
assert!(
bar.volume > 0.0 && bar.volume.is_finite(),
"Bar {} has invalid volume: {}",
i,
bar.volume
);
}
println!("✅ All {} bars have valid OHLCV relationships", bars.len());
}