Three fixes for GPU PER hot path: 1. IQN quantile loss used empty CPU weights Vec instead of GPU-resident weights_tensor_cached — caused CUDA_ERROR_ILLEGAL_ADDRESS from uninitialized GPU memory. Now uses cached GPU tensor matching C51 and standard DQN paths. 2. GpuPrioritized add()/add_batch() replaced with StagedGpuBuffer: add() stages on CPU (Vec::push, zero GPU ops), sample() batch-flushes staging→GPU in one DMA before sampling. Production path (insert_batch_tensors) bypasses staging entirely — GPU→GPU with zero CPU. 3. All 9 ML sub-crates default to cuda feature so `cargo test -p ml-dqn` exercises GPU code paths on CUDA workstations. CI service crates use default-features=false, unaffected. Test results: 350 passed (was 343+7 failed), 0 failed, 1 ignored. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
71 lines
1.7 KiB
TOML
71 lines
1.7 KiB
TOML
[package]
|
|
name = "ml-supervised"
|
|
version.workspace = true
|
|
edition.workspace = true
|
|
rust-version.workspace = true
|
|
authors.workspace = true
|
|
license.workspace = true
|
|
repository.workspace = true
|
|
homepage.workspace = true
|
|
documentation.workspace = true
|
|
publish.workspace = true
|
|
keywords.workspace = true
|
|
categories.workspace = true
|
|
description = "Supervised models (TFT, Mamba, Liquid, TGGN, TLOB, KAN, xLSTM, Diffusion)"
|
|
|
|
[features]
|
|
default = ["cuda"]
|
|
cuda = ["candle-core/cuda", "candle-core/cudnn", "candle-nn/cuda", "candle-nn/cudnn"]
|
|
|
|
[dependencies]
|
|
ml-core.workspace = true
|
|
common.workspace = true
|
|
config.workspace = true
|
|
data.workspace = true
|
|
|
|
# Async
|
|
tokio.workspace = true
|
|
|
|
# ML frameworks
|
|
candle-core = { git = "https://github.com/huggingface/candle", rev = "671de1db" }
|
|
candle-nn = { git = "https://github.com/huggingface/candle", rev = "671de1db" }
|
|
|
|
# Serialization
|
|
serde = { workspace = true, features = ["derive"] }
|
|
serde_json.workspace = true
|
|
|
|
# Core utilities
|
|
thiserror.workspace = true
|
|
anyhow.workspace = true
|
|
tracing.workspace = true
|
|
rand.workspace = true
|
|
uuid.workspace = true
|
|
async-trait.workspace = true
|
|
|
|
# Numerics
|
|
ndarray = { workspace = true, features = ["rayon"] }
|
|
nalgebra = { version = "0.33", features = ["serde-serialize"] }
|
|
|
|
# Graph support (TGGN)
|
|
petgraph = { version = "0.6", features = ["serde"] }
|
|
|
|
# Caching (TFT)
|
|
lru.workspace = true
|
|
|
|
# System
|
|
libc = "0.2"
|
|
num_cpus = "1.16"
|
|
|
|
# Concurrency
|
|
rayon.workspace = true
|
|
dashmap = { workspace = true }
|
|
parking_lot = { version = "0.12", features = ["hardware-lock-elision"] }
|
|
|
|
[dev-dependencies]
|
|
tokio = { workspace = true, features = ["test-util", "macros"] }
|
|
approx.workspace = true
|
|
tempfile = "3"
|
|
|
|
[lints]
|
|
workspace = true
|