diff --git a/crates/ml/examples/train_baseline_rl.rs b/crates/ml/examples/train_baseline_rl.rs index c186cdd3c..be3d1516f 100644 --- a/crates/ml/examples/train_baseline_rl.rs +++ b/crates/ml/examples/train_baseline_rl.rs @@ -816,6 +816,7 @@ fn run_training(args: &Args) -> Result> { }); // Try loading from fxcache + eprintln!("[DEBUG] fxcache_dir = {:?}", fxcache_dir); let fxcache_data = fxcache_dir.and_then(|cache_dir| { let symbol_dir = args.data_dir.join(&args.symbol); let default_mbp10 = PathBuf::from("test_data/futures-baseline-mbp10"); @@ -861,7 +862,8 @@ fn run_training(args: &Args) -> Result> { n, data_load_start.elapsed().as_secs_f64()); (bars, cached.features) } else { - // Fall back to DBN loading + // Fall back to DBN loading — this is SLOW (148GB MBP-10 parsing) + eprintln!("[DEBUG] fxcache MISS — falling back to DBN loading (SLOW!)"); info!(" Loading OHLCV bars from DBN files..."); let bars = load_all_bars(&args.data_dir, &args.symbol)?; if bars.is_empty() { diff --git a/crates/ml/src/trainers/dqn/trainer/training_loop.rs b/crates/ml/src/trainers/dqn/trainer/training_loop.rs index d086bde3f..4093af169 100644 --- a/crates/ml/src/trainers/dqn/trainer/training_loop.rs +++ b/crates/ml/src/trainers/dqn/trainer/training_loop.rs @@ -286,11 +286,13 @@ impl DQNTrainer { }; // ── Phase 2: GPU experience collection ── + eprintln!("[DEBUG] Phase 2: starting GPU experience collection (epoch {})", epoch); let phase2_start = std::time::Instant::now(); let gpu_experiences_collected = self.collect_gpu_experiences( training_data, ).await?; let phase2_ms = phase2_start.elapsed().as_secs_f64() * 1000.0; + eprintln!("[DEBUG] Phase 2: done in {:.1}ms, collected={}", phase2_ms, gpu_experiences_collected); // CUDA builds: GPU experience collector is MANDATORY if !gpu_experiences_collected { @@ -315,9 +317,11 @@ impl DQNTrainer { } // ── Phase 3: Batched training from replay buffer ── + eprintln!("[DEBUG] Phase 3: starting training steps (epoch {})", epoch); let phase3_start = std::time::Instant::now(); let train_step_count = self.run_training_steps(training_data).await?; let phase3_ms = phase3_start.elapsed().as_secs_f64() * 1000.0; + eprintln!("[DEBUG] Phase 3: done in {:.1}ms, steps={}", phase3_ms, train_step_count); // ── Phase 4: Epoch boundary + validation ── let phase4_start = std::time::Instant::now(); @@ -1169,11 +1173,13 @@ impl DQNTrainer { // #33 Saboteur state already set above (before collector borrow) // Zero-roundtrip GPU path — GPU PER is always active in CUDA builds + eprintln!("[DEBUG] collect_experiences_gpu: n_episodes={}, timesteps={}, starts_len={}", n_episodes, timesteps, episode_starts.len()); let gpu_batch = collector.collect_experiences_gpu( features_buf, targets_buf, &episode_starts, &config, ).map_err(|e| anyhow::anyhow!( "GPU zero-roundtrip collection FAILED (no CPU fallback): {e}" ))?; + eprintln!("[DEBUG] collect_experiences_gpu: DONE, batch.n_episodes={}, batch.timesteps={}", gpu_batch.n_episodes, gpu_batch.timesteps); let count = gpu_batch.n_episodes * gpu_batch.timesteps;