tft_real_dbn_data: StreamTensor::from_vec, quantile loss returns f32 ppo_recurrent_integration: PPO::new() API, get_policy_state &[f32] test_dbn_sequence_256: to_host + manual indexing instead of .i() ops ppo_checkpoint_roundtrip: save/load_checkpoint(&PathBuf) API mamba2_accuracy_fix: pure f64 arithmetic, no GPU tensors needed ppo_lstm_training_loop: PPO::new() API ppo_step_counter_fix: new checkpoint API ppo_recurrent_performance: forward_host, LSTM batch_size arg Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
145 lines
4.3 KiB
Rust
145 lines
4.3 KiB
Rust
#![allow(
|
|
clippy::assertions_on_constants,
|
|
clippy::assertions_on_result_states,
|
|
clippy::clone_on_copy,
|
|
clippy::decimal_literal_representation,
|
|
clippy::doc_markdown,
|
|
clippy::empty_line_after_doc_comments,
|
|
clippy::field_reassign_with_default,
|
|
clippy::get_unwrap,
|
|
clippy::identity_op,
|
|
clippy::inconsistent_digit_grouping,
|
|
clippy::indexing_slicing,
|
|
clippy::integer_division,
|
|
clippy::len_zero,
|
|
clippy::let_underscore_must_use,
|
|
clippy::manual_div_ceil,
|
|
clippy::manual_let_else,
|
|
clippy::manual_range_contains,
|
|
clippy::modulo_arithmetic,
|
|
clippy::needless_range_loop,
|
|
clippy::non_ascii_literal,
|
|
clippy::redundant_clone,
|
|
clippy::shadow_reuse,
|
|
clippy::shadow_same,
|
|
clippy::shadow_unrelated,
|
|
clippy::single_match_else,
|
|
clippy::str_to_string,
|
|
clippy::string_slice,
|
|
clippy::tests_outside_test_module,
|
|
clippy::too_many_lines,
|
|
clippy::unnecessary_wraps,
|
|
clippy::unseparated_literal_suffix,
|
|
clippy::use_debug,
|
|
clippy::useless_vec,
|
|
clippy::wildcard_enum_match_arm,
|
|
clippy::else_if_without_else,
|
|
clippy::expect_used,
|
|
clippy::missing_const_for_fn,
|
|
clippy::similar_names,
|
|
clippy::type_complexity,
|
|
clippy::collapsible_else_if,
|
|
clippy::doc_lazy_continuation,
|
|
clippy::items_after_test_module,
|
|
clippy::map_clone,
|
|
clippy::multiple_unsafe_ops_per_block,
|
|
clippy::unwrap_or_default,
|
|
clippy::assign_op_pattern,
|
|
clippy::needless_borrow,
|
|
clippy::println_empty_string,
|
|
clippy::unnecessary_cast,
|
|
clippy::used_underscore_binding,
|
|
clippy::create_dir,
|
|
clippy::implicit_saturating_sub,
|
|
clippy::exit,
|
|
clippy::expect_fun_call,
|
|
clippy::too_many_arguments,
|
|
clippy::unnecessary_map_or,
|
|
clippy::unwrap_used,
|
|
dead_code,
|
|
unused_imports,
|
|
unused_variables,
|
|
clippy::cloned_ref_to_slice_refs,
|
|
clippy::neg_multiply,
|
|
clippy::while_let_loop,
|
|
clippy::bool_assert_comparison,
|
|
clippy::excessive_precision,
|
|
clippy::trivially_copy_pass_by_ref,
|
|
clippy::op_ref,
|
|
clippy::redundant_closure,
|
|
clippy::unnecessary_lazy_evaluations,
|
|
clippy::if_then_some_else_none,
|
|
clippy::unnecessary_to_owned,
|
|
clippy::single_component_path_imports,
|
|
)]
|
|
//! PPO Checkpoint Round-Trip Validation Test
|
|
//!
|
|
//! Proves that saving a PPO model checkpoint and loading it back
|
|
//! produces identical action probabilities for the same input state.
|
|
//!
|
|
//! Run manually:
|
|
//! ```sh
|
|
//! SQLX_OFFLINE=true cargo test -p ml --test ppo_checkpoint_roundtrip_test -- --ignored --nocapture
|
|
//! ```
|
|
|
|
#![allow(unused_crate_dependencies)]
|
|
|
|
use anyhow::Result;
|
|
use ml::ppo::ppo::{PPOConfig, PPO};
|
|
use std::path::PathBuf;
|
|
use tracing::info;
|
|
|
|
#[tokio::test]
|
|
#[ignore]
|
|
async fn test_ppo_checkpoint_roundtrip() -> Result<()> {
|
|
let checkpoint_dir = tempfile::tempdir()?;
|
|
let state_dim = 54;
|
|
let num_actions = 45;
|
|
|
|
// Create PPO model with known config
|
|
let config = PPOConfig {
|
|
state_dim,
|
|
num_actions,
|
|
policy_hidden_dims: vec![128, 64],
|
|
value_hidden_dims: vec![128, 64],
|
|
use_lstm: false,
|
|
..PPOConfig::default()
|
|
};
|
|
|
|
let ppo = PPO::new(config.clone())?;
|
|
|
|
// Create test state
|
|
let test_state: Vec<f32> = (0..state_dim).map(|i| (i as f32 * 0.1).sin()).collect();
|
|
|
|
// Get greedy action before save (GPU-side argmax, no bulk transfer)
|
|
let action_before = ppo.greedy_action(&test_state)?;
|
|
|
|
// Save checkpoint (API takes &PathBuf)
|
|
let checkpoint_path = checkpoint_dir.path().join("checkpoint");
|
|
|
|
ppo.save_checkpoint(&checkpoint_path)?;
|
|
|
|
// Verify config file exists and is non-empty
|
|
let config_path = checkpoint_path.with_extension("json");
|
|
assert!(config_path.exists(), "Config file not saved");
|
|
assert!(std::fs::metadata(&config_path)?.len() > 0);
|
|
|
|
// Load into a fresh model (API takes &PathBuf)
|
|
let loaded_ppo = PPO::load_checkpoint(&checkpoint_path)?;
|
|
|
|
// Get greedy action after load — must match exactly
|
|
let action_after = loaded_ppo.greedy_action(&test_state)?;
|
|
|
|
assert_eq!(
|
|
action_before, action_after,
|
|
"Greedy action mismatch after checkpoint roundtrip: {:?} vs {:?}",
|
|
action_before, action_after,
|
|
);
|
|
|
|
let config_bytes = std::fs::metadata(&config_path)?.len();
|
|
info!("Checkpoint round-trip validation passed");
|
|
info!(config_bytes, "Config checkpoint size");
|
|
|
|
Ok(())
|
|
}
|