Phase 2A scaffolding lands FIRST per spec §4.4 ABI contract. Phase 1.2 cost kernel reads LobBar; both dev synthetic and prod fxcache produce LobBar — dev/prod parity per Q3. Generators: flat_market, drift_market, ou_market, regime_switch_market (seeded RNG for reproducible tests). Regime-switch test uses sticky 0.99/0.01 transitions (true regime persistence; spec's 50/50 was a random walk, not a regime switch — corrected with code comment). behavioral_suite test target wired into Cargo.toml; will run all 22 Phase 2 tests once they land in Phase 2B/2C. Audit doc: SP15 Phase 2A.1 entry appended to docs/dqn-wire-up-audit.md per Invariant 7 (component changes require audit-doc update). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
293 lines
11 KiB
TOML
293 lines
11 KiB
TOML
[package]
|
|
name = "ml"
|
|
version.workspace = true
|
|
edition.workspace = true
|
|
rust-version.workspace = true
|
|
authors.workspace = true
|
|
license.workspace = true
|
|
repository.workspace = true
|
|
homepage.workspace = true
|
|
documentation.workspace = true
|
|
publish.workspace = true
|
|
keywords.workspace = true
|
|
categories.workspace = true
|
|
|
|
[features]
|
|
# CUDA default: all ML training is GPU-only. Service crates on CPU nodes
|
|
# must opt out with `default-features = false, features = ["minimal-inference"]`.
|
|
#
|
|
# `default` includes `full-stack` so existing consumers depending on
|
|
# `ml.workspace = true` (no extra opts) continue to see the full module
|
|
# surface. Services that only need a subset should disable defaults and
|
|
# opt in to specific feature flags below (see services/*/Cargo.toml).
|
|
default = ["minimal-inference", "cuda", "full-stack"]
|
|
|
|
# PRODUCTION FEATURES - LIGHTWEIGHT ONLY
|
|
minimal-inference = [] # Minimal inference with no optional deps
|
|
financial = [] # Basic financial calculations
|
|
high-precision = ["rust_decimal/serde-float"]
|
|
|
|
# PERFORMANCE FEATURES - NO HEAVY ML
|
|
mimalloc-allocator = ["mimalloc"] # Fast memory allocator for 10-25% speedup
|
|
simd = [] # SIMD without heavy dependencies
|
|
|
|
# Storage and memory management features
|
|
gc = [] # Garbage collection features
|
|
s3-storage = ["ml-checkpoint/s3-storage", "aws-config", "aws-sdk-s3", "aws-types", "aws-credential-types", "urlencoding"] # S3 storage backend with AWS SDK
|
|
cuda = ["cudarc", "ml-core/cuda", "ml-dqn/cuda", "ml-ppo/cuda", "ml-supervised/cuda", "ml-ensemble/cuda", "ml-labeling/cuda", "ml-explainability/cuda", "ml-hyperopt/cuda"] # CUDA support — enabled by compile-training CI step via --features ml/cuda
|
|
nccl = ["cuda"] # NCCL multi-GPU data parallelism (requires NCCL library + cudarc nccl feature)
|
|
|
|
# ============================================================================
|
|
# OPT-IN MODULE FEATURES — gate optional ml-* sub-crates and the `pub mod`
|
|
# items in lib.rs that depend on them. Services should enable only what they
|
|
# actually `use`. Each leaf sub-crate maps to exactly one `pub mod` in lib.rs.
|
|
# ============================================================================
|
|
backtest-mod = ["dep:ml-backtesting"] # ml::backtesting
|
|
paper-trading-mod = ["dep:ml-paper-trading"] # ml::paper_trading
|
|
stress-testing-mod = ["dep:ml-stress-testing"] # ml::stress_testing
|
|
explainability-mod = ["dep:ml-explainability"] # ml::explainability
|
|
universe-mod = ["dep:ml-universe"] # ml::universe
|
|
regime-detection-mod = ["dep:ml-regime-detection"] # ml::regime_detection
|
|
validation-mod = ["dep:ml-validation"] # ml::validation
|
|
data-validation-mod = ["dep:ml-data-validation"] # ml::data_validation
|
|
|
|
# Convenience: full module surface (used by `default` to preserve existing
|
|
# behavior). Services that opt out of `default` and want everything can
|
|
# still enable just `full-stack`.
|
|
full-stack = [
|
|
"backtest-mod",
|
|
"paper-trading-mod",
|
|
"stress-testing-mod",
|
|
"explainability-mod",
|
|
"universe-mod",
|
|
"regime-detection-mod",
|
|
"validation-mod",
|
|
"data-validation-mod",
|
|
]
|
|
|
|
# ALL HEAVY ML FEATURES REMOVED:
|
|
# gpu, pytorch, linfa-ml - MOVED TO ml_training_service
|
|
# optimization, graph-models, reinforcement-learning - MOVED TO ml_training_service
|
|
# transformers-advanced - MOVED TO ml_training_service
|
|
|
|
[dependencies]
|
|
# Core async and utilities
|
|
tokio.workspace = true
|
|
futures.workspace = true
|
|
async-trait.workspace = true
|
|
clap.workspace = true # CLI argument parsing for train_tft binary
|
|
|
|
# Serialization and error handling
|
|
serde.workspace = true
|
|
serde_json.workspace = true
|
|
serde_yaml = "0.9" # YAML serialization for hyperparameter exports
|
|
toml.workspace = true
|
|
uuid.workspace = true
|
|
thiserror.workspace = true
|
|
anyhow.workspace = true
|
|
chrono.workspace = true
|
|
chrono-tz = "0.10" # Timezone support for market hours calculations (Wave C)
|
|
time = "0.3" # Required by dbn::TsSymbolMap for instrument_id → symbol resolution
|
|
csv = "1.3" # CSV serialization for action export (Wave 3 Task 3.1)
|
|
rand.workspace = true
|
|
rand_chacha = "0.3" # ChaCha RNG for reproducible Monte Carlo permutation tests
|
|
|
|
# System and I/O
|
|
memmap2.workspace = true
|
|
tempfile.workspace = true
|
|
tracing.workspace = true
|
|
tracing-subscriber.workspace = true # For train_tft binary logging
|
|
prometheus.workspace = true
|
|
reqwest.workspace = true
|
|
colored = "2.1" # Terminal color output for evaluation reports
|
|
|
|
# Internal workspace crates
|
|
# Always-on: required by inference, training_pipeline, safety, ensemble, and
|
|
# many cross-module references inside ml/src/.
|
|
ml-core.workspace = true
|
|
ml-ensemble.workspace = true
|
|
ml-dqn.workspace = true
|
|
ml-ppo.workspace = true
|
|
ml-supervised.workspace = true
|
|
ml-hyperopt.workspace = true
|
|
ml-features.workspace = true
|
|
ml-labeling.workspace = true
|
|
ml-regime.workspace = true
|
|
ml-checkpoint.workspace = true
|
|
ml-risk.workspace = true
|
|
ml-security.workspace = true
|
|
ml-asset-selection.workspace = true
|
|
ml-observability.workspace = true
|
|
# Opt-in via features (gated `pub mod` items in lib.rs). Services that don't
|
|
# use these modules don't pay the compile cost. See [features] above for
|
|
# which feature gates which crate / module.
|
|
ml-backtesting = { workspace = true, optional = true }
|
|
ml-paper-trading = { workspace = true, optional = true }
|
|
ml-stress-testing = { workspace = true, optional = true }
|
|
ml-explainability = { workspace = true, optional = true }
|
|
ml-universe = { workspace = true, optional = true }
|
|
ml-regime-detection = { workspace = true, optional = true }
|
|
ml-validation = { workspace = true, optional = true }
|
|
ml-data-validation = { workspace = true, optional = true }
|
|
config.workspace = true
|
|
common = { workspace = true, features = ["questdb"] }
|
|
risk = { path = "../risk" }
|
|
# Model loading functionality is in storage crate
|
|
storage = { path = "../storage" }
|
|
# Data crate for test helpers (dev-dependency in tests)
|
|
data = { path = "../data" }
|
|
|
|
# Database for model registry
|
|
sqlx.workspace = true
|
|
|
|
|
|
# candle-core, candle-nn, candle-optimisers — REMOVED (replaced by ml-core native CUDA autograd)
|
|
|
|
# HEAVY ML FRAMEWORKS REMOVED - MOVED TO ml_training_service
|
|
# ort (ONNX Runtime) - REMOVED (1000+ dependencies alone!)
|
|
# tch, torch-sys (PyTorch bindings) - REMOVED (500+ dependencies!)
|
|
|
|
|
|
# Mathematical libraries.
|
|
# ndarray's `blas` feature is intentionally not enabled — it requires
|
|
# libopenblas-dev on the build host, which the CI compile pool does not
|
|
# provide. Training hot paths run through CUDA cuBLAS on GPU, so the
|
|
# host-side BLAS is unnecessary.
|
|
ndarray = { workspace = true, features = ["rayon"] }
|
|
nalgebra = { version = "0.33", features = ["serde-serialize"] }
|
|
|
|
# MINIMAL statistics only - ALL HEAVY ML ALGORITHMS REMOVED
|
|
# linfa ecosystem (linfa, linfa-clustering, linfa-linear, linfa-reduction) - REMOVED (200+ deps)
|
|
# smartcore - REMOVED (100+ dependencies)
|
|
# Basic statistics - always included (not optional)
|
|
statrs.workspace = true # Required for statistical computations
|
|
|
|
|
|
rust_decimal.workspace = true
|
|
|
|
|
|
# gymnasium, rerun - REMOVED (RL frameworks moved to ml_training_service)
|
|
|
|
|
|
# cudarc — vendored fork (vendor/cudarc): has_async_alloc=false fix for cublasLtMatmul on H100
|
|
cudarc = { version = "0.19", optional = true, default-features = false, features = ["driver", "cublas", "cublaslt", "dynamic-linking", "std", "cuda-version-from-build-system", "f16"] }
|
|
rayon.workspace = true
|
|
crossbeam = { version = "0.8", features = ["std"] }
|
|
|
|
|
|
petgraph = { version = "0.6", features = ["serde"] } # Required for TGNN graphs
|
|
semver = "1.0"
|
|
lru.workspace = true # Required for model caching
|
|
|
|
|
|
|
|
# chronoutil, ta, polars - REMOVED or moved to workspace dependencies
|
|
|
|
|
|
# argmin, nlopt, ipopt - REMOVED (optimization frameworks moved to ml_training_service)
|
|
|
|
|
|
rand_distr.workspace = true
|
|
dbn.workspace = true # Databento Binary format for real market data loading
|
|
zstd.workspace = true # Zstd decompression for .dbn.zst trade files
|
|
databento = "0.34" # Databento API client for downloading data (includes async by default)
|
|
|
|
# Performance allocators
|
|
mimalloc = { version = "0.1", optional = true } # Fast memory allocator
|
|
dotenv = "0.15" # Load .env files for API keys
|
|
parking_lot = { version = "0.12", features = ["hardware-lock-elision"] }
|
|
dashmap = { workspace = true }
|
|
once_cell = "1.19"
|
|
lazy_static.workspace = true
|
|
flate2 = "1.0"
|
|
sha2 = "0.10"
|
|
safetensors = "0.7" # Direct dep for checkpoint metadata (config hash validation)
|
|
hmac = "0.12" # HMAC for checkpoint signatures (SEC-001 fix)
|
|
hex = "0.4" # Hex encoding for signatures
|
|
bincode = "1.3"
|
|
fastrand = "2.1"
|
|
# wide - REMOVED (SIMD moved to trading_engine)
|
|
num-traits = "0.2"
|
|
|
|
# Parquet I/O for feature caching (Wave 2 Agent 8)
|
|
# Updated to workspace version 56 to fix arrow-arith compilation conflict
|
|
parquet.workspace = true
|
|
arrow.workspace = true
|
|
bytes = "1.5" # For Parquet in-memory serialization
|
|
num = "0.4"
|
|
libc = "0.2"
|
|
fs2 = "0.4"
|
|
num_cpus = "1.16"
|
|
approx.workspace = true
|
|
sysinfo = "0.33" # System information for benchmarks
|
|
|
|
# AWS SDK dependencies for S3 checkpoint storage (optional, s3-storage feature)
|
|
aws-config = { version = "1.1", optional = true }
|
|
aws-sdk-s3 = { version = "1.14", optional = true }
|
|
aws-types = { version = "1.1", optional = true }
|
|
aws-credential-types = { version = "1.1", optional = true }
|
|
urlencoding = { version = "2.1", optional = true }
|
|
|
|
# Bayesian optimization for hyperparameter tuning (using argmin instead of egobox due to ndarray conflict)
|
|
argmin = { version = "0.8", features = ["rayon"] } # Optimization framework with parallel execution
|
|
argmin-math = "0.3" # Math utilities for argmin
|
|
|
|
[dev-dependencies]
|
|
tokio-test = "0.4"
|
|
proptest = "1.5"
|
|
tempfile = "3.12"
|
|
futures-test = "0.3"
|
|
test-case = "3.0"
|
|
rstest = "0.22"
|
|
criterion = { version = "0.5", features = ["html_reports", "async_tokio"] }
|
|
fastrand = "2.1"
|
|
|
|
tokio = { workspace = true, features = ["test-util", "macros"] }
|
|
insta = "1.34" # Snapshot testing for ML outputs
|
|
serial_test = "3.0" # Sequential testing for GPU resources
|
|
tracing-subscriber = { version = "0.3", features = ["env-filter", "fmt"] }
|
|
rand_chacha = "0.3" # ChaCha RNG for hyperopt tests
|
|
|
|
[[example]]
|
|
name = "cuda_test"
|
|
path = "examples/cuda_test.rs"
|
|
|
|
[[example]]
|
|
name = "train_baseline_rl"
|
|
path = "examples/train_baseline_rl.rs"
|
|
|
|
[[example]]
|
|
name = "train_baseline_supervised"
|
|
path = "examples/train_baseline_supervised.rs"
|
|
|
|
[[example]]
|
|
name = "evaluate_baseline"
|
|
path = "examples/evaluate_baseline.rs"
|
|
|
|
[[example]]
|
|
name = "hyperopt_baseline_rl"
|
|
path = "examples/hyperopt_baseline_rl.rs"
|
|
|
|
[[example]]
|
|
name = "hyperopt_baseline_supervised"
|
|
path = "examples/hyperopt_baseline_supervised.rs"
|
|
|
|
[[test]]
|
|
name = "behavioral_suite"
|
|
path = "tests/behavioral/main.rs"
|
|
|
|
[[bench]]
|
|
name = "microstructure_bench"
|
|
harness = false
|
|
|
|
[[bench]]
|
|
name = "alternative_bars_bench"
|
|
harness = false
|
|
|
|
[[bench]]
|
|
name = "bench_feature_extraction"
|
|
harness = false
|
|
|
|
[lints]
|
|
workspace = true
|