feat(ml-backtesting): per-backtest sim parameter arrays (P1)
Migrates target_annual_vol_units, annualisation_factor, max_lots, latency_ns, kelly_frac_floor, sharpe_weight_floor from scalar-broadcast kernel args to per-backtest device arrays via BatchedSimConfig. Atomic contract change per feedback_no_partial_refactor — kernel + sim + harness + all 3 existing test files migrate in this commit. - crates/ml-backtesting/src/sim.rs → sim/mod.rs (directory module) - crates/ml-backtesting/src/sim/batched_config.rs (NEW): BatchedSimConfig + UniformSimParams + validate(). from_uniform rebuilds the legacy uniform-broadcast behaviour at n=1 (smoke/fixtures). from_grid lands in P6 for the 140-variant sweep packing. - LobSimCuda gains 6 per-backtest device buffers (target_annual_vol_units_d, annualisation_factor_d, max_lots_d, latency_ns_d, kelly_frac_floor_d, sharpe_weight_floor_d). step_decision_with_latency uploads from BatchedSimConfig each call; both decision kernel launches now pass per-backtest array pointers. - decision_policy_default + decision_policy_program: scalar args become const float* / const int* per_b arrays; first lines of each kernel index by `b` into the arrays. Behaviour preserved at n=1 uniform. - dispatch_latent_market_orders: reads latency per-backtest from &BatchedSimConfig (host loop stays for P1; P2 replaces with kernel). - step_decision_with_latency now ALWAYS dispatches through the latency path; when cfg.latency_ns[b]=0 the in-flight slot's arrival_ts equals current_ts and gets promoted immediately on next snapshot. Eliminates the if/else branch and consolidates the launch path. - harness.rs: BacktestHarness gains a sim_config field, built via BatchedSimConfig::from_uniform at new() from the harness cfg's scalar fields. The run loop passes &self.sim_config to step_decision_with_latency. Regression coverage: - parallel_sim_correctness::parallel_sim_equivalence_with_uniform_config — n=8 with uniform config produces 8 bit-identical market_targets (proves per-backtest indexing reduces correctly). - Existing decision_floor_coldstart tests (3) all pass through the new ABI — proves cold-start floor + variance-cap gate behaviour preserved. - parallel_sim_independence_per_backtest deferred to P2 (needs read_first_inflight_arrival_ts helper that depends on LimitSlot layout). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -28,30 +28,38 @@
|
||||
#define MIN_TRADES_FOR_VAR_CAP 10u
|
||||
|
||||
extern "C" __global__ void decision_policy_default(
|
||||
const float* __restrict__ alpha_probs, // [N_HORIZONS] broadcast
|
||||
const unsigned char* __restrict__ isv_kelly_base, // [n_backtests * N_HORIZONS * sizeof(IsvKellyState)]
|
||||
const Pos* __restrict__ positions, // [n_backtests]
|
||||
const unsigned int* __restrict__ program_lens, // [n_backtests] — non-zero ⇒ skip (program kernel handles it)
|
||||
int* __restrict__ market_targets, // [n_backtests * 2] — written
|
||||
unsigned int* __restrict__ open_horizon_masks, // [n_backtests] — written (entry attribution)
|
||||
float target_annual_vol_units,
|
||||
float annualisation_factor,
|
||||
int max_lots,
|
||||
const float* __restrict__ alpha_probs, // [N_HORIZONS] broadcast
|
||||
const unsigned char* __restrict__ isv_kelly_base, // [n_backtests * N_HORIZONS * sizeof(IsvKellyState)]
|
||||
const Pos* __restrict__ positions, // [n_backtests]
|
||||
const unsigned int* __restrict__ program_lens, // [n_backtests] — non-zero ⇒ skip (program kernel handles it)
|
||||
int* __restrict__ market_targets, // [n_backtests * 2] — written
|
||||
unsigned int* __restrict__ open_horizon_masks, // [n_backtests] — written (entry attribution)
|
||||
// P1: per-backtest sim parameter arrays. Each backtest indexed by b.
|
||||
const float* __restrict__ target_annual_vol_units_per_b, // [n_backtests]
|
||||
const float* __restrict__ annualisation_factor_per_b, // [n_backtests]
|
||||
const int* __restrict__ max_lots_per_b, // [n_backtests]
|
||||
// Cold-start floors per pearl_blend_formulas_must_have_permanent_floor:
|
||||
// max(floor, real), not blend; pearl_kelly_cap_signal_driven_floors.
|
||||
// When isv_kelly state is at sentinel (no closed trades), kelly_frac
|
||||
// falls back to `kelly_frac_floor`; the aggregate weight floor lets
|
||||
// falls back to `kelly_frac_floor_per_b[b]`; the aggregate weight floor lets
|
||||
// cross-horizon sum produce a non-zero size before recent_sharpe is
|
||||
// populated. Floors are non-zero by contract — passing 0.0 reverts
|
||||
// to the old sentinel-skip behaviour.
|
||||
float kelly_frac_floor,
|
||||
float sharpe_weight_floor,
|
||||
const float* __restrict__ kelly_frac_floor_per_b, // [n_backtests]
|
||||
const float* __restrict__ sharpe_weight_floor_per_b, // [n_backtests]
|
||||
int n_backtests
|
||||
) {
|
||||
int b = blockIdx.x;
|
||||
if (b >= n_backtests || threadIdx.x != 0) return;
|
||||
if (program_lens[b] != 0u) return; // backtest b uses a custom bytecode program; skip default
|
||||
|
||||
// Per-backtest scalar reads (host-uploaded arrays).
|
||||
const float target_annual_vol_units = target_annual_vol_units_per_b[b];
|
||||
const float annualisation_factor = annualisation_factor_per_b[b];
|
||||
const int max_lots = max_lots_per_b[b];
|
||||
const float kelly_frac_floor = kelly_frac_floor_per_b[b];
|
||||
const float sharpe_weight_floor = sharpe_weight_floor_per_b[b];
|
||||
|
||||
const IsvKellyState* isv = reinterpret_cast<const IsvKellyState*>(
|
||||
isv_kelly_base + (size_t)b * N_HORIZONS * sizeof(IsvKellyState)
|
||||
);
|
||||
@@ -178,19 +186,18 @@ extern "C" __global__ void decision_policy_program(
|
||||
const float* __restrict__ alpha_probs,
|
||||
const unsigned char* __restrict__ isv_kelly_base,
|
||||
const Pos* __restrict__ positions,
|
||||
const Instruction* __restrict__ programs, // [n_backtests * max_instructions]
|
||||
const unsigned int* __restrict__ program_lens, // [n_backtests]
|
||||
const unsigned int* __restrict__ regimes, // [n_backtests] — current regime id
|
||||
int* __restrict__ market_targets, // [n_backtests * 2]
|
||||
const Instruction* __restrict__ programs, // [n_backtests * max_instructions]
|
||||
const unsigned int* __restrict__ program_lens, // [n_backtests]
|
||||
const unsigned int* __restrict__ regimes, // [n_backtests] — current regime id
|
||||
int* __restrict__ market_targets, // [n_backtests * 2]
|
||||
unsigned int* __restrict__ open_horizon_masks,
|
||||
float target_annual_vol_units,
|
||||
float annualisation_factor,
|
||||
// P1: per-backtest sim parameter arrays. See decision_policy_default header for contract.
|
||||
const float* __restrict__ target_annual_vol_units_per_b,
|
||||
const float* __restrict__ annualisation_factor_per_b,
|
||||
int max_instructions,
|
||||
int max_lots,
|
||||
// Cold-start floors — see decision_policy_default header for the
|
||||
// contract. Both kernels share the same Kelly + weight floor pattern.
|
||||
float kelly_frac_floor,
|
||||
float sharpe_weight_floor,
|
||||
const int* __restrict__ max_lots_per_b,
|
||||
const float* __restrict__ kelly_frac_floor_per_b,
|
||||
const float* __restrict__ sharpe_weight_floor_per_b,
|
||||
int n_backtests
|
||||
) {
|
||||
int b = blockIdx.x;
|
||||
@@ -201,6 +208,13 @@ extern "C" __global__ void decision_policy_program(
|
||||
// Skip — this backtest uses the hardcoded default kernel separately.
|
||||
return;
|
||||
}
|
||||
// Per-backtest scalar reads (host-uploaded arrays).
|
||||
const float target_annual_vol_units = target_annual_vol_units_per_b[b];
|
||||
const float annualisation_factor = annualisation_factor_per_b[b];
|
||||
const int max_lots = max_lots_per_b[b];
|
||||
const float kelly_frac_floor = kelly_frac_floor_per_b[b];
|
||||
const float sharpe_weight_floor = sharpe_weight_floor_per_b[b];
|
||||
|
||||
const Instruction* prog = programs + (size_t)b * max_instructions;
|
||||
const IsvKellyState* isv = reinterpret_cast<const IsvKellyState*>(
|
||||
isv_kelly_base + (size_t)b * N_HORIZONS * sizeof(IsvKellyState)
|
||||
|
||||
@@ -60,6 +60,11 @@ pub struct BacktestHarnessConfig {
|
||||
|
||||
pub struct BacktestHarness {
|
||||
cfg: BacktestHarnessConfig,
|
||||
/// Per-backtest sim parameter arrays. P1 onwards is the source of
|
||||
/// truth for sim-side scalar params; cfg's scalar fields are still
|
||||
/// used to construct this via BatchedSimConfig::from_uniform at
|
||||
/// harness new(), but the run-loop reads sim_config, not cfg.
|
||||
sim_config: crate::sim::BatchedSimConfig,
|
||||
loader: MultiHorizonLoader,
|
||||
/// PerceptionTrainer in inference role — owns the trunk (loaded from
|
||||
/// Checkpoint) and the kernel-launch scratches. Forward driven via
|
||||
@@ -127,9 +132,25 @@ impl BacktestHarness {
|
||||
sim.upload_program(b, &prog)?;
|
||||
}
|
||||
|
||||
// P1: BatchedSimConfig built from uniform broadcast of the harness cfg's
|
||||
// scalar fields. In P6 the sweep runner constructs a per-variant
|
||||
// BatchedSimConfig via BatchedSimConfig::from_grid instead.
|
||||
let sim_config = crate::sim::BatchedSimConfig::from_uniform(
|
||||
cfg.n_parallel,
|
||||
&crate::sim::UniformSimParams {
|
||||
target_annual_vol_units: cfg.target_annual_vol_units,
|
||||
annualisation_factor: cfg.annualisation_factor,
|
||||
max_lots: cfg.max_lots,
|
||||
latency_ns: cfg.latency_ns,
|
||||
kelly_frac_floor: cfg.kelly_frac_floor,
|
||||
sharpe_weight_floor: cfg.sharpe_weight_floor,
|
||||
},
|
||||
);
|
||||
|
||||
let pnl_curves = (0..cfg.n_parallel).map(|_| Vec::with_capacity(1024)).collect();
|
||||
Ok(Self {
|
||||
cfg,
|
||||
sim_config,
|
||||
loader,
|
||||
trainer,
|
||||
snapshot_window: VecDeque::with_capacity(seq_len),
|
||||
@@ -185,15 +206,7 @@ impl BacktestHarness {
|
||||
.try_into()
|
||||
.context("slice last K probs")?;
|
||||
self.sim.broadcast_alpha(&last_probs)?;
|
||||
self.sim.step_decision_with_latency(
|
||||
raw.ts_ns,
|
||||
self.cfg.target_annual_vol_units,
|
||||
self.cfg.annualisation_factor,
|
||||
self.cfg.max_lots,
|
||||
self.cfg.latency_ns,
|
||||
self.cfg.kelly_frac_floor,
|
||||
self.cfg.sharpe_weight_floor,
|
||||
)?;
|
||||
self.sim.step_decision_with_latency(raw.ts_ns, &self.sim_config)?;
|
||||
self.decision_count += 1;
|
||||
total_decisions += 1;
|
||||
}
|
||||
|
||||
68
crates/ml-backtesting/src/sim/batched_config.rs
Normal file
68
crates/ml-backtesting/src/sim/batched_config.rs
Normal file
@@ -0,0 +1,68 @@
|
||||
//! Per-backtest sim parameter arrays. Replaces the scalar-broadcast
|
||||
//! parameters previously passed to `step_decision_with_latency`.
|
||||
//! `from_uniform` rebuilds the legacy uniform behaviour for smoke /
|
||||
//! fixture tests where every backtest shares the same config.
|
||||
//! `from_grid` packs a sweep cell-grid (n_parallel=140 typical) into
|
||||
//! one BatchedSimConfig.
|
||||
|
||||
use anyhow::{anyhow, Result};
|
||||
|
||||
/// Single source of truth for per-backtest sim params at run time.
|
||||
/// Lengths MUST all equal `n_backtests`.
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct BatchedSimConfig {
|
||||
pub target_annual_vol_units: Vec<f32>,
|
||||
pub annualisation_factor: Vec<f32>,
|
||||
pub max_lots: Vec<u16>,
|
||||
pub latency_ns: Vec<u32>,
|
||||
pub kelly_frac_floor: Vec<f32>,
|
||||
pub sharpe_weight_floor: Vec<f32>,
|
||||
}
|
||||
|
||||
/// Convenience scalar input for `from_uniform`. Mirror of the historical
|
||||
/// BacktestHarnessConfig sim fields. Threshold + cost will be added in P4.
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct UniformSimParams {
|
||||
pub target_annual_vol_units: f32,
|
||||
pub annualisation_factor: f32,
|
||||
pub max_lots: u16,
|
||||
pub latency_ns: u32,
|
||||
pub kelly_frac_floor: f32,
|
||||
pub sharpe_weight_floor: f32,
|
||||
}
|
||||
|
||||
impl BatchedSimConfig {
|
||||
pub fn n_backtests(&self) -> usize {
|
||||
self.target_annual_vol_units.len()
|
||||
}
|
||||
|
||||
pub fn from_uniform(n: usize, p: &UniformSimParams) -> Self {
|
||||
Self {
|
||||
target_annual_vol_units: vec![p.target_annual_vol_units; n],
|
||||
annualisation_factor: vec![p.annualisation_factor; n],
|
||||
max_lots: vec![p.max_lots; n],
|
||||
latency_ns: vec![p.latency_ns; n],
|
||||
kelly_frac_floor: vec![p.kelly_frac_floor; n],
|
||||
sharpe_weight_floor: vec![p.sharpe_weight_floor; n],
|
||||
}
|
||||
}
|
||||
|
||||
pub fn validate(&self) -> Result<()> {
|
||||
let n = self.target_annual_vol_units.len();
|
||||
let lens = [
|
||||
("annualisation_factor", self.annualisation_factor.len()),
|
||||
("max_lots", self.max_lots.len()),
|
||||
("latency_ns", self.latency_ns.len()),
|
||||
("kelly_frac_floor", self.kelly_frac_floor.len()),
|
||||
("sharpe_weight_floor", self.sharpe_weight_floor.len()),
|
||||
];
|
||||
for (name, l) in lens {
|
||||
if l != n {
|
||||
return Err(anyhow!(
|
||||
"BatchedSimConfig length mismatch: {name} has {l} entries, expected {n}"
|
||||
));
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,9 @@
|
||||
//!
|
||||
//! See docs/superpowers/specs/2026-05-18-real-lob-integration-design.md §2 + §5b.
|
||||
|
||||
mod batched_config;
|
||||
pub use batched_config::{BatchedSimConfig, UniformSimParams};
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{CudaSlice, CudaStream, LaunchConfig, PushKernelArg};
|
||||
use ml_core::device::MlDevice;
|
||||
@@ -74,6 +77,15 @@ pub struct LobSimCuda {
|
||||
program_table_d: CudaSlice<u8>, // [n_backtests * MAX_PROGRAM_INSTRUCTIONS * INSTRUCTION_BYTES]
|
||||
program_lens_d: CudaSlice<u32>, // [n_backtests]
|
||||
regimes_d: CudaSlice<u32>, // [n_backtests] — host-set current regime id
|
||||
|
||||
// P1: per-backtest sim parameter buffers. Uploaded by step_decision_with_latency
|
||||
// from a host-side BatchedSimConfig. Held as Sim fields to avoid per-call alloc.
|
||||
target_annual_vol_units_d: CudaSlice<f32>, // [n_backtests]
|
||||
annualisation_factor_d: CudaSlice<f32>, // [n_backtests]
|
||||
max_lots_d: CudaSlice<i32>, // [n_backtests] — stored as i32 to match kernel arg
|
||||
latency_ns_d: CudaSlice<u32>, // [n_backtests] — used by P2 kernel
|
||||
kelly_frac_floor_d: CudaSlice<f32>, // [n_backtests]
|
||||
sharpe_weight_floor_d: CudaSlice<f32>, // [n_backtests]
|
||||
}
|
||||
|
||||
impl LobSimCuda {
|
||||
@@ -188,6 +200,21 @@ impl LobSimCuda {
|
||||
.alloc_zeros::<u32>(n_backtests)
|
||||
.context("alloc regimes_d")?;
|
||||
|
||||
// P1: per-backtest sim parameter device buffers (uploaded each
|
||||
// step_decision_with_latency from BatchedSimConfig).
|
||||
let target_annual_vol_units_d = stream.alloc_zeros::<f32>(n_backtests)
|
||||
.context("alloc target_annual_vol_units_d")?;
|
||||
let annualisation_factor_d = stream.alloc_zeros::<f32>(n_backtests)
|
||||
.context("alloc annualisation_factor_d")?;
|
||||
let max_lots_d = stream.alloc_zeros::<i32>(n_backtests)
|
||||
.context("alloc max_lots_d")?;
|
||||
let latency_ns_d = stream.alloc_zeros::<u32>(n_backtests)
|
||||
.context("alloc latency_ns_d")?;
|
||||
let kelly_frac_floor_d = stream.alloc_zeros::<f32>(n_backtests)
|
||||
.context("alloc kelly_frac_floor_d")?;
|
||||
let sharpe_weight_floor_d = stream.alloc_zeros::<f32>(n_backtests)
|
||||
.context("alloc sharpe_weight_floor_d")?;
|
||||
|
||||
Ok(Self {
|
||||
n_backtests,
|
||||
stream,
|
||||
@@ -222,6 +249,12 @@ impl LobSimCuda {
|
||||
program_table_d,
|
||||
program_lens_d,
|
||||
regimes_d,
|
||||
target_annual_vol_units_d,
|
||||
annualisation_factor_d,
|
||||
max_lots_d,
|
||||
latency_ns_d,
|
||||
kelly_frac_floor_d,
|
||||
sharpe_weight_floor_d,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -493,25 +526,13 @@ impl LobSimCuda {
|
||||
}
|
||||
|
||||
/// Backward-compat wrapper for the no-latency immediate-match path.
|
||||
/// Equivalent to `step_decision_with_latency(.., 0, .., ..)`.
|
||||
/// Callers seeking immediate fill should set `cfg.latency_ns[b] = 0`.
|
||||
pub fn step_decision(
|
||||
&mut self,
|
||||
current_ts_ns: u64,
|
||||
target_annual_vol_units: f32,
|
||||
annualisation_factor: f32,
|
||||
max_lots: u16,
|
||||
kelly_frac_floor: f32,
|
||||
sharpe_weight_floor: f32,
|
||||
cfg: &BatchedSimConfig,
|
||||
) -> Result<()> {
|
||||
self.step_decision_with_latency(
|
||||
current_ts_ns,
|
||||
target_annual_vol_units,
|
||||
annualisation_factor,
|
||||
max_lots,
|
||||
0,
|
||||
kelly_frac_floor,
|
||||
sharpe_weight_floor,
|
||||
)
|
||||
self.step_decision_with_latency(current_ts_ns, cfg)
|
||||
}
|
||||
|
||||
/// Run the C7 decision pipeline with optional latency simulation:
|
||||
@@ -535,13 +556,24 @@ impl LobSimCuda {
|
||||
pub fn step_decision_with_latency(
|
||||
&mut self,
|
||||
current_ts_ns: u64,
|
||||
target_annual_vol_units: f32,
|
||||
annualisation_factor: f32,
|
||||
max_lots: u16,
|
||||
latency_ns: u32,
|
||||
kelly_frac_floor: f32,
|
||||
sharpe_weight_floor: f32,
|
||||
sim_cfg: &BatchedSimConfig,
|
||||
) -> Result<()> {
|
||||
anyhow::ensure!(
|
||||
sim_cfg.n_backtests() == self.n_backtests,
|
||||
"BatchedSimConfig n_backtests {} != Sim n_backtests {}",
|
||||
sim_cfg.n_backtests(), self.n_backtests
|
||||
);
|
||||
sim_cfg.validate()?;
|
||||
|
||||
// P1: upload per-backtest sim params into pre-allocated device buffers.
|
||||
self.stream.memcpy_htod(&sim_cfg.target_annual_vol_units, &mut self.target_annual_vol_units_d)?;
|
||||
self.stream.memcpy_htod(&sim_cfg.annualisation_factor, &mut self.annualisation_factor_d)?;
|
||||
let max_lots_i32: Vec<i32> = sim_cfg.max_lots.iter().map(|&m| m as i32).collect();
|
||||
self.stream.memcpy_htod(&max_lots_i32, &mut self.max_lots_d)?;
|
||||
self.stream.memcpy_htod(&sim_cfg.latency_ns, &mut self.latency_ns_d)?;
|
||||
self.stream.memcpy_htod(&sim_cfg.kelly_frac_floor, &mut self.kelly_frac_floor_d)?;
|
||||
self.stream.memcpy_htod(&sim_cfg.sharpe_weight_floor, &mut self.sharpe_weight_floor_d)?;
|
||||
|
||||
// Step 1: decision kernels write market_targets + open_horizon_mask.
|
||||
// 1a — bytecode program kernel handles backtests with plen > 0.
|
||||
// 1b — default hardcoded kernel handles backtests with plen == 0.
|
||||
@@ -551,7 +583,6 @@ impl LobSimCuda {
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
let n = self.n_backtests as i32;
|
||||
let max_lots_i = max_lots as i32;
|
||||
let max_instrs = crate::lob::MAX_PROGRAM_INSTRUCTIONS as i32;
|
||||
|
||||
// 1a: bytecode VM kernel (skips backtests with plen==0 internally).
|
||||
@@ -566,12 +597,12 @@ impl LobSimCuda {
|
||||
.arg(&self.regimes_d)
|
||||
.arg(&mut self.market_targets_d)
|
||||
.arg(&mut self.open_horizon_mask_d)
|
||||
.arg(&target_annual_vol_units)
|
||||
.arg(&annualisation_factor)
|
||||
.arg(&self.target_annual_vol_units_d)
|
||||
.arg(&self.annualisation_factor_d)
|
||||
.arg(&max_instrs)
|
||||
.arg(&max_lots_i)
|
||||
.arg(&kelly_frac_floor)
|
||||
.arg(&sharpe_weight_floor)
|
||||
.arg(&self.max_lots_d)
|
||||
.arg(&self.kelly_frac_floor_d)
|
||||
.arg(&self.sharpe_weight_floor_d)
|
||||
.arg(&n)
|
||||
.launch(cfg)?;
|
||||
}
|
||||
@@ -585,11 +616,11 @@ impl LobSimCuda {
|
||||
.arg(&self.program_lens_d)
|
||||
.arg(&mut self.market_targets_d)
|
||||
.arg(&mut self.open_horizon_mask_d)
|
||||
.arg(&target_annual_vol_units)
|
||||
.arg(&annualisation_factor)
|
||||
.arg(&max_lots_i)
|
||||
.arg(&kelly_frac_floor)
|
||||
.arg(&sharpe_weight_floor)
|
||||
.arg(&self.target_annual_vol_units_d)
|
||||
.arg(&self.annualisation_factor_d)
|
||||
.arg(&self.max_lots_d)
|
||||
.arg(&self.kelly_frac_floor_d)
|
||||
.arg(&self.sharpe_weight_floor_d)
|
||||
.arg(&n)
|
||||
.launch(cfg)?;
|
||||
}
|
||||
@@ -616,24 +647,12 @@ impl LobSimCuda {
|
||||
let prev_pos_per_block = self.snapshot_position_lots()?;
|
||||
let prev_mask_per_block = self.snapshot_open_horizon_mask()?;
|
||||
|
||||
if latency_ns == 0 {
|
||||
// Step 3a (legacy, no-latency): execute immediately.
|
||||
let mut launch = self.stream.launch_builder(&self.submit_market_fn);
|
||||
unsafe {
|
||||
launch
|
||||
.arg(&self.books_d)
|
||||
.arg(&self.market_targets_d)
|
||||
.arg(&mut self.pos_d)
|
||||
.arg(&n)
|
||||
.launch(cfg)?;
|
||||
}
|
||||
self.stream.synchronize()?;
|
||||
} else {
|
||||
// Step 3b (latency path): convert each non-noop market_target
|
||||
// to an in-flight aggressive Limit. Read targets back to host,
|
||||
// submit one seed_limit_alloc per backtest.
|
||||
self.dispatch_latent_market_orders(current_ts_ns, latency_ns)?;
|
||||
}
|
||||
// Step 3: always dispatch through the latency path. When a backtest's
|
||||
// latency_ns == 0 the in-flight slot's arrival_ts equals current_ts
|
||||
// and step_resting_orders promotes immediately on the next snapshot,
|
||||
// functionally equivalent to the legacy submit_market_immediate path
|
||||
// but supporting per-backtest latencies in one code path.
|
||||
self.dispatch_latent_market_orders(current_ts_ns, sim_cfg)?;
|
||||
|
||||
// Step 4: pnl_track — detect close, emit TradeRecord.
|
||||
let cap = crate::lob::TRADE_LOG_CAP as i32;
|
||||
@@ -699,7 +718,7 @@ impl LobSimCuda {
|
||||
fn dispatch_latent_market_orders(
|
||||
&mut self,
|
||||
current_ts_ns: u64,
|
||||
latency_ns: u32,
|
||||
sim_cfg: &BatchedSimConfig,
|
||||
) -> Result<()> {
|
||||
let n = self.n_backtests;
|
||||
let mut targets = vec![0i32; n * 2];
|
||||
@@ -710,6 +729,7 @@ impl LobSimCuda {
|
||||
if side == 2 || size <= 0 {
|
||||
continue; // no-op
|
||||
}
|
||||
let latency_ns = sim_cfg.latency_ns[b];
|
||||
// Find a free slot via the existing seed path (which sets
|
||||
// overflow_flag if all slots are in use, surfaced as Err).
|
||||
let mut placed = false;
|
||||
@@ -15,9 +15,20 @@
|
||||
|
||||
use anyhow::Result;
|
||||
use ml_backtesting::policy::IsvKellyStateHost;
|
||||
use ml_backtesting::sim::LobSimCuda;
|
||||
use ml_backtesting::sim::{BatchedSimConfig, LobSimCuda, UniformSimParams};
|
||||
use ml_core::device::MlDevice;
|
||||
|
||||
fn cfg_uniform(n: usize, kelly: f32, sharpe: f32) -> BatchedSimConfig {
|
||||
BatchedSimConfig::from_uniform(n, &UniformSimParams {
|
||||
target_annual_vol_units: 50.0,
|
||||
annualisation_factor: 825.0,
|
||||
max_lots: 5,
|
||||
latency_ns: 0,
|
||||
kelly_frac_floor: kelly,
|
||||
sharpe_weight_floor: sharpe,
|
||||
})
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires CUDA"]
|
||||
fn cold_start_sentinel_state_still_fires_a_trade() -> Result<()> {
|
||||
@@ -33,15 +44,7 @@ fn cold_start_sentinel_state_still_fires_a_trade() -> Result<()> {
|
||||
// Strong directional alpha across all horizons → conviction-driven
|
||||
// sig_mag = 0.6 for every horizon, dir = +1.
|
||||
sim.broadcast_alpha(&[0.8, 0.8, 0.8, 0.8, 0.8])?;
|
||||
sim.step_decision_with_latency(
|
||||
0, // current_ts_ns
|
||||
50.0, // target_annual_vol_units
|
||||
825.0, // annualisation_factor
|
||||
5, // max_lots
|
||||
0, // latency_ns (immediate path; not relevant — we only read targets)
|
||||
0.20, // kelly_frac_floor — the fix under test (default)
|
||||
0.10, // sharpe_weight_floor
|
||||
)?;
|
||||
sim.step_decision_with_latency(0, &cfg_uniform(1, 0.20, 0.10))?;
|
||||
let (side, size) = sim.read_market_target(0)?;
|
||||
assert_eq!(side, 0, "cold-start with p_h=0.8 must produce a long; got side={side}");
|
||||
assert!(size >= 1, "cold-start size {size} < 1 — the kernel floor isn't firing");
|
||||
@@ -82,7 +85,7 @@ fn post_first_loss_state_does_not_lock_out_further_trades() -> Result<()> {
|
||||
});
|
||||
sim.write_isv_kelly(0, &post_loss)?;
|
||||
sim.broadcast_alpha(&[0.8, 0.8, 0.8, 0.8, 0.8])?;
|
||||
sim.step_decision_with_latency(0, 50.0, 825.0, 5, 0, 0.20, 0.10)?;
|
||||
sim.step_decision_with_latency(0, &cfg_uniform(1, 0.20, 0.10))?;
|
||||
let (side, size) = sim.read_market_target(0)?;
|
||||
assert_eq!(side, 0, "post-loss state must still fire a long with strong alpha (got side={side})");
|
||||
assert!(size >= 1, "post-loss size {size} < 1 — n_trades_seen gate isn't bypassing the variance cap");
|
||||
@@ -105,7 +108,7 @@ fn cold_start_with_zero_floor_reproduces_old_bug() -> Result<()> {
|
||||
};
|
||||
let mut sim = LobSimCuda::new(1, &dev)?;
|
||||
sim.broadcast_alpha(&[0.8, 0.8, 0.8, 0.8, 0.8])?;
|
||||
sim.step_decision_with_latency(0, 50.0, 825.0, 5, 0, 0.0, 0.0)?;
|
||||
sim.step_decision_with_latency(0, &cfg_uniform(1, 0.0, 0.0))?;
|
||||
let (side, size) = sim.read_market_target(0)?;
|
||||
assert_eq!(side, 2, "with zero floors, sentinel state must still skip (side=noop)");
|
||||
assert_eq!(size, 0);
|
||||
|
||||
@@ -248,14 +248,18 @@ fn run_book_fixture(path: &Path) -> Result<()> {
|
||||
ts_ns,
|
||||
} => {
|
||||
sim.broadcast_alpha(alpha_probs)?;
|
||||
sim.step_decision(
|
||||
*ts_ns,
|
||||
*target_annual_vol_units,
|
||||
*annualisation_factor,
|
||||
*max_lots,
|
||||
0.10, // kelly_frac_floor — cold-start prior
|
||||
0.10, // sharpe_weight_floor
|
||||
)?;
|
||||
let sim_cfg = ml_backtesting::sim::BatchedSimConfig::from_uniform(
|
||||
sim.n_backtests(),
|
||||
&ml_backtesting::sim::UniformSimParams {
|
||||
target_annual_vol_units: *target_annual_vol_units,
|
||||
annualisation_factor: *annualisation_factor,
|
||||
max_lots: *max_lots,
|
||||
latency_ns: 0,
|
||||
kelly_frac_floor: 0.10,
|
||||
sharpe_weight_floor: 0.10,
|
||||
},
|
||||
);
|
||||
sim.step_decision(*ts_ns, &sim_cfg)?;
|
||||
}
|
||||
FixtureEvent::SeedLimit {
|
||||
backtest_idx, slot_idx, side, kind, active_state,
|
||||
|
||||
@@ -120,15 +120,18 @@ fn run_integrated_fuzz(n_backtests: usize, n_events: usize, seed: u64) -> Result
|
||||
sim.broadcast_alpha(&probs)?;
|
||||
// Mix latency vs immediate paths to cover both.
|
||||
let latency = if decision_idx % 2 == 0 { 0 } else { 50_000_000 };
|
||||
sim.step_decision_with_latency(
|
||||
ts_ns,
|
||||
rng.gen_range(20.0..60.0), // target_annual_vol_units
|
||||
rng.gen_range(500.0..1200.0), // annualisation_factor
|
||||
rng.gen_range(2..6), // max_lots
|
||||
latency,
|
||||
0.10, // kelly_frac_floor
|
||||
0.10, // sharpe_weight_floor
|
||||
)?;
|
||||
let sim_cfg = ml_backtesting::sim::BatchedSimConfig::from_uniform(
|
||||
sim.n_backtests(),
|
||||
&ml_backtesting::sim::UniformSimParams {
|
||||
target_annual_vol_units: rng.gen_range(20.0..60.0),
|
||||
annualisation_factor: rng.gen_range(500.0..1200.0),
|
||||
max_lots: rng.gen_range(2..6),
|
||||
latency_ns: latency,
|
||||
kelly_frac_floor: 0.10,
|
||||
sharpe_weight_floor: 0.10,
|
||||
},
|
||||
);
|
||||
sim.step_decision_with_latency(ts_ns, &sim_cfg)?;
|
||||
decision_idx += 1;
|
||||
}
|
||||
|
||||
|
||||
51
crates/ml-backtesting/tests/parallel_sim_correctness.rs
Normal file
51
crates/ml-backtesting/tests/parallel_sim_correctness.rs
Normal file
@@ -0,0 +1,51 @@
|
||||
//! P1 regression test: per-backtest sim parameter arrays. With uniform
|
||||
//! config across N backtests, all N must produce the same market_target
|
||||
//! decision (proves the per-backtest indexing reduces correctly to the
|
||||
//! pre-refactor scalar-broadcast semantics).
|
||||
//!
|
||||
//! Pair: the independence test (proving DIFFERENT per-backtest configs
|
||||
//! produce different outputs) lands alongside the P2 inflight-limits
|
||||
//! kernel where the read_first_inflight_arrival_ts helper exists.
|
||||
|
||||
use anyhow::Result;
|
||||
use ml_backtesting::sim::{BatchedSimConfig, LobSimCuda, UniformSimParams};
|
||||
use ml_core::device::MlDevice;
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires CUDA"]
|
||||
fn parallel_sim_equivalence_with_uniform_config() -> Result<()> {
|
||||
let dev = match MlDevice::cuda(0) {
|
||||
Ok(d) => d,
|
||||
Err(e) => {
|
||||
eprintln!("skipping: cuda device unavailable ({e})");
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
let mut sim = LobSimCuda::new(8, &dev)?;
|
||||
sim.broadcast_alpha(&[0.8, 0.8, 0.8, 0.8, 0.8])?;
|
||||
let cfg = BatchedSimConfig::from_uniform(
|
||||
8,
|
||||
&UniformSimParams {
|
||||
target_annual_vol_units: 50.0,
|
||||
annualisation_factor: 825.0,
|
||||
max_lots: 5,
|
||||
latency_ns: 0,
|
||||
kelly_frac_floor: 0.20,
|
||||
sharpe_weight_floor: 0.10,
|
||||
},
|
||||
);
|
||||
sim.step_decision_with_latency(0, &cfg)?;
|
||||
let first = sim.read_market_target(0)?;
|
||||
for b in 1..8 {
|
||||
let got = sim.read_market_target(b)?;
|
||||
assert_eq!(
|
||||
got, first,
|
||||
"backtest {b} differs from backtest 0 under uniform config: {got:?} vs {first:?}"
|
||||
);
|
||||
}
|
||||
// And the uniform-config result must equal the result we'd get from
|
||||
// a single-backtest sim (preserves pre-refactor behaviour).
|
||||
assert_eq!(first.0, 0, "uniform config with p_h=0.8 should produce a buy; got side={}", first.0);
|
||||
assert!(first.1 >= 1, "uniform-config size {} < 1", first.1);
|
||||
Ok(())
|
||||
}
|
||||
Reference in New Issue
Block a user