feat(argo): single-pod compile+train on GPU for alpha-rl
Replaces the 4-step DAG (check-cache → compile on CPU → warmup GPU → train on GPU) with a single pod on the L40S that does everything: git fetch → incremental cargo build (~3s warm) → train. Eliminates: separate compile node, node autoscale wait, binary transfer, fxcache step. The ci-builder image has CUDA 13.0 devel + Rust — compilation and training use the same CUDA libs. The cargo-target-cuda PVC provides incremental build cache across runs. LD_LIBRARY_PATH strips the stubs dir so real CUDA libs load. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -1,25 +1,15 @@
|
||||
# Integrated RL trainer workflow (Phase R8).
|
||||
# alpha-rl: single-pod compile + train on GPU.
|
||||
#
|
||||
# Drives `IntegratedTrainer::step_with_lobsim` (DQN + PPO on shared
|
||||
# Mamba2 -> CfC encoder) against MBP-10 from the training-data PVC,
|
||||
# using LobSimCuda as the synthetic-book environment. Emits
|
||||
# `alpha_rl_train_summary.json` to the feature-cache PVC with final-step
|
||||
# loss components + replay buffer state for monitoring + G8 NaN-abort
|
||||
# detection.
|
||||
#
|
||||
# DAG mirrors alpha-perception (check-cache → ensure-binary →
|
||||
# warmup-gpu → train) so the alpine probe / sccache compile / autoscaler
|
||||
# warmup behaviour stays a single source of truth. The only material
|
||||
# difference is the binary built (`alpha_rl_train` vs `alpha_train`).
|
||||
#
|
||||
# Defaults match the R9 validation smoke (1000 steps, b_size=1,
|
||||
# front-month ES). Production scale-up overrides via dispatcher flags.
|
||||
# Compiles alpha_rl_train incrementally on the GPU node (~3s warm,
|
||||
# ~90s cold) using the cargo-target-cuda PVC, then runs training
|
||||
# immediately. No separate compile node, no binary transfer, no
|
||||
# fxcache step. Predecoded MBP-10 sidecars live on feature-cache PVC.
|
||||
#
|
||||
# Usage:
|
||||
# argo submit -n foxhunt --from=wftmpl/alpha-rl \
|
||||
# -p commit-sha=HEAD -p git-branch=ml-alpha-phase-a
|
||||
#
|
||||
# Plan reference: docs/superpowers/plans/2026-05-23-integrated-rl-trainer-rebuild.md
|
||||
# -p git-branch=ml-alpha-phase-a
|
||||
# argo submit -n foxhunt --from=wftmpl/alpha-rl \
|
||||
# -p git-branch=ml-alpha-phase-a -p n-steps=50000 -p n-backtests=16
|
||||
apiVersion: argoproj.io/v1alpha1
|
||||
kind: WorkflowTemplate
|
||||
metadata:
|
||||
@@ -30,7 +20,7 @@ metadata:
|
||||
app.kubernetes.io/part-of: foxhunt
|
||||
app.kubernetes.io/component: train
|
||||
spec:
|
||||
entrypoint: pipeline
|
||||
entrypoint: compile-and-train
|
||||
serviceAccountName: argo-workflow
|
||||
archiveLogs: true
|
||||
podMetadata:
|
||||
@@ -45,37 +35,24 @@ spec:
|
||||
|
||||
arguments:
|
||||
parameters:
|
||||
- name: commit-sha
|
||||
value: HEAD
|
||||
- name: git-branch
|
||||
value: ml-alpha-phase-a
|
||||
- name: cuda-compute-cap
|
||||
# Default sm_89 matches the L40S default pool per
|
||||
# `feedback_default_to_l40s_pool`. Dispatcher derives this
|
||||
# automatically from --gpu-pool.
|
||||
value: "89"
|
||||
- name: gpu-pool
|
||||
# Per `feedback_default_to_l40s_pool`: ci-training-l40s is the
|
||||
# default (NOT H100). Opt into ci-training-h100 explicitly for
|
||||
# production scale-up sweeps that need the 80GB / sm_90 capacity.
|
||||
value: ci-training-l40s
|
||||
# RL-specific trainer knobs (mirror alpha_rl_train CLI defaults).
|
||||
- name: n-steps
|
||||
value: "1000"
|
||||
- name: seq-len
|
||||
value: "32"
|
||||
value: "50000"
|
||||
- name: n-backtests
|
||||
# Default 16 — fixes b_size=1 signal starvation per
|
||||
# `pearl_b_size_1_signal_starvation_blocks_q_learning`. 16
|
||||
# parallel transitions per Adam step give Q proper gradient
|
||||
# variance reduction. Override via `-p n-backtests=N`.
|
||||
value: "16"
|
||||
- name: per-capacity
|
||||
value: "4096"
|
||||
- name: seed
|
||||
value: "16962"
|
||||
- name: seq-len
|
||||
value: "32"
|
||||
- name: per-capacity
|
||||
value: "4096"
|
||||
- name: instrument-mode
|
||||
value: "front-month"
|
||||
value: "all"
|
||||
- name: log-every
|
||||
value: "5000"
|
||||
- name: fold-idx
|
||||
value: "0"
|
||||
- name: n-folds
|
||||
@@ -99,104 +76,14 @@ spec:
|
||||
claimName: feature-cache-pvc
|
||||
|
||||
templates:
|
||||
# ── DAG: check-cache → ensure-binary (when cache miss) → train ──
|
||||
# warmup-gpu runs unconditionally in parallel to trigger L40S
|
||||
# autoscaler scale-up so the node is warm by train-time, matching
|
||||
# the alpha-perception pattern.
|
||||
- name: pipeline
|
||||
dag:
|
||||
tasks:
|
||||
- name: check-cache
|
||||
template: check-cache
|
||||
- name: ensure-binary
|
||||
template: ensure-binary
|
||||
dependencies: [check-cache]
|
||||
when: "{{tasks.check-cache.outputs.parameters.cache}} == miss"
|
||||
arguments:
|
||||
parameters:
|
||||
- name: sha
|
||||
value: "{{tasks.check-cache.outputs.parameters.sha}}"
|
||||
- name: warmup-gpu
|
||||
template: warmup-gpu
|
||||
- name: train
|
||||
template: train
|
||||
dependencies: [check-cache, ensure-binary]
|
||||
arguments:
|
||||
parameters:
|
||||
- name: sha
|
||||
value: "{{tasks.check-cache.outputs.parameters.sha}}"
|
||||
|
||||
# ── check-cache: probe training-data PVC for cached binary ──
|
||||
# Same alpine probe pattern as alpha-perception. Caches at
|
||||
# /data/bin/$SHORT_SHA/alpha_rl_train (distinct from alpha_train's
|
||||
# cache slot so the two binaries can coexist at the same SHA).
|
||||
- name: check-cache
|
||||
outputs:
|
||||
parameters:
|
||||
- name: sha
|
||||
valueFrom:
|
||||
path: /tmp/sha
|
||||
- name: cache
|
||||
valueFrom:
|
||||
path: /tmp/cache
|
||||
- name: compile-and-train
|
||||
nodeSelector:
|
||||
k8s.scaleway.com/pool-name: platform
|
||||
k8s.scaleway.com/pool-name: "{{workflow.parameters.gpu-pool}}"
|
||||
topology.kubernetes.io/zone: fr-par-2
|
||||
tolerations:
|
||||
- key: node.cilium.io/agent-not-ready
|
||||
- key: nvidia.com/gpu
|
||||
operator: Exists
|
||||
effect: NoSchedule
|
||||
container:
|
||||
image: alpine:3.20
|
||||
command: ["/bin/sh", "-c"]
|
||||
resources:
|
||||
requests:
|
||||
cpu: "50m"
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: "100m"
|
||||
memory: 64Mi
|
||||
volumeMounts:
|
||||
- name: training-data
|
||||
mountPath: /data
|
||||
readOnly: true
|
||||
args:
|
||||
- |
|
||||
set -e
|
||||
SHA="{{workflow.parameters.commit-sha}}"
|
||||
if [ "$SHA" = "HEAD" ]; then
|
||||
echo "ERROR: commit-sha cannot be HEAD inside the workflow."
|
||||
echo " Resolve via the submission script (scripts/argo-alpha-rl.sh)."
|
||||
exit 1
|
||||
fi
|
||||
SHORT_SHA=$(echo "$SHA" | cut -c1-9)
|
||||
echo "$SHORT_SHA" > /tmp/sha
|
||||
|
||||
BIN="/data/bin/$SHORT_SHA/alpha_rl_train"
|
||||
if [ -x "$BIN" ]; then
|
||||
SIZE=$(stat -c %s "$BIN")
|
||||
echo "Cache HIT: $BIN ($SIZE bytes)"
|
||||
echo "hit" > /tmp/cache
|
||||
else
|
||||
echo "Cache MISS: $BIN not present, ensure-binary will compile"
|
||||
ls -lh "/data/bin/" 2>/dev/null | head -10 || echo " (no /data/bin/ directory yet)"
|
||||
echo "miss" > /tmp/cache
|
||||
fi
|
||||
|
||||
# ── ensure-binary: compile alpha_rl_train via sccache on miss ──
|
||||
- name: ensure-binary
|
||||
inputs:
|
||||
parameters:
|
||||
- name: sha
|
||||
outputs:
|
||||
parameters:
|
||||
- name: sha
|
||||
valueFrom:
|
||||
path: /tmp/sha
|
||||
nodeSelector:
|
||||
k8s.scaleway.com/pool-name: ci-compile-cpu
|
||||
topology.kubernetes.io/zone: fr-par-2
|
||||
tolerations:
|
||||
- key: node.cilium.io/agent-not-ready
|
||||
operator: Exists
|
||||
effect: NoSchedule
|
||||
@@ -207,227 +94,93 @@ spec:
|
||||
env:
|
||||
- name: SQLX_OFFLINE
|
||||
value: "true"
|
||||
- name: CARGO_TERM_COLOR
|
||||
value: always
|
||||
- name: CARGO_TARGET_DIR
|
||||
value: /cargo-target
|
||||
- name: CARGO_HOME
|
||||
value: /cargo-target/cargo-home
|
||||
- name: RUSTC_WRAPPER
|
||||
value: sccache
|
||||
- name: SCCACHE_DIR
|
||||
value: /cargo-target/sccache
|
||||
- name: SCCACHE_CACHE_SIZE
|
||||
value: "40G"
|
||||
- name: CARGO_INCREMENTAL
|
||||
value: "0"
|
||||
value: "1"
|
||||
- name: CUDA_COMPUTE_CAP
|
||||
value: "{{workflow.parameters.cuda-compute-cap}}"
|
||||
value: "89"
|
||||
resources:
|
||||
requests:
|
||||
cpu: "14"
|
||||
memory: 32Gi
|
||||
cpu: "4"
|
||||
memory: 16Gi
|
||||
nvidia.com/gpu: "1"
|
||||
limits:
|
||||
cpu: "30"
|
||||
memory: 64Gi
|
||||
cpu: "7"
|
||||
memory: 40Gi
|
||||
nvidia.com/gpu: "1"
|
||||
volumeMounts:
|
||||
- name: git-ssh-key
|
||||
mountPath: /etc/git-ssh
|
||||
readOnly: true
|
||||
- name: cargo-target-cuda
|
||||
mountPath: /cargo-target
|
||||
- name: training-data
|
||||
mountPath: /data
|
||||
readOnly: true
|
||||
- name: feature-cache
|
||||
mountPath: /feature-cache
|
||||
- name: cargo-target-cuda
|
||||
mountPath: /cargo-target
|
||||
args:
|
||||
- |
|
||||
set -e
|
||||
SHA="{{workflow.parameters.commit-sha}}"
|
||||
BRANCH="{{workflow.parameters.git-branch}}"
|
||||
export PATH="${CARGO_HOME}/bin:${PATH}"
|
||||
export LD_LIBRARY_PATH=$(echo "$LD_LIBRARY_PATH" | tr ':' '\n' | grep -v stubs | tr '\n' ':' | sed 's/:$//')
|
||||
|
||||
# SSH
|
||||
mkdir -p ~/.ssh
|
||||
cp /etc/git-ssh/ssh-privatekey ~/.ssh/id_ed25519
|
||||
chmod 600 ~/.ssh/id_ed25519
|
||||
printf 'Host *\n StrictHostKeyChecking no\n UserKnownHostsFile /dev/null\n' > ~/.ssh/config
|
||||
chmod 600 ~/.ssh/config
|
||||
REPO="ssh://git@gitlab-gitlab-shell.foxhunt.svc.cluster.local:2222/root/foxhunt.git"
|
||||
|
||||
nvidia-smi
|
||||
|
||||
# Git
|
||||
BRANCH="{{workflow.parameters.git-branch}}"
|
||||
BUILD="/cargo-target/src"
|
||||
REPO="ssh://git@gitlab-gitlab-shell.foxhunt.svc.cluster.local:2222/root/foxhunt.git"
|
||||
git config --global --add safe.directory "$BUILD"
|
||||
|
||||
if [ "$SHA" = "HEAD" ]; then
|
||||
if [ -d "$BUILD/.git" ]; then
|
||||
cd "$BUILD"; git fetch origin; SHA=$(git rev-parse "origin/$BRANCH"); cd /
|
||||
else
|
||||
git clone --filter=blob:none "$REPO" "$BUILD"; cd "$BUILD"
|
||||
git checkout "origin/$BRANCH"; SHA=$(git rev-parse HEAD); cd /
|
||||
fi
|
||||
fi
|
||||
SHORT_SHA=$(echo "$SHA" | cut -c1-9)
|
||||
echo "Resolved SHA: $SHA (short: $SHORT_SHA)"
|
||||
|
||||
BIN_DIR="/data/bin/$SHORT_SHA"
|
||||
if [ -x "$BIN_DIR/alpha_rl_train" ]; then
|
||||
echo "=== Cache HIT: $BIN_DIR/alpha_rl_train ==="
|
||||
ls -lh "$BIN_DIR/"
|
||||
echo "$SHORT_SHA" > /tmp/sha
|
||||
exit 0
|
||||
fi
|
||||
echo "=== Cache MISS: compiling alpha_rl_train for $SHORT_SHA ==="
|
||||
|
||||
if [ -d "$BUILD/.git" ]; then
|
||||
cd "$BUILD"; git fetch origin
|
||||
CURRENT=$(git rev-parse HEAD 2>/dev/null || echo "none")
|
||||
if [ "$CURRENT" != "$SHA" ]; then
|
||||
git checkout --force "$SHA"; git clean -fd
|
||||
fi
|
||||
cd "$BUILD"
|
||||
git fetch origin
|
||||
git checkout --force "origin/$BRANCH"
|
||||
git clean -fd
|
||||
else
|
||||
git clone --filter=blob:none "$REPO" "$BUILD"; cd "$BUILD"; git checkout "$SHA"
|
||||
git clone --filter=blob:none "$REPO" "$BUILD"
|
||||
cd "$BUILD"
|
||||
git checkout "origin/$BRANCH"
|
||||
fi
|
||||
SHA=$(git rev-parse --short=9 HEAD)
|
||||
echo "=== Branch: $BRANCH SHA: $SHA ==="
|
||||
|
||||
export PATH="${CARGO_HOME}/bin:${PATH}"
|
||||
# Compile (~3s incremental, ~90s cold)
|
||||
echo "=== Compile ==="
|
||||
time cargo build --release -p ml-alpha --example alpha_rl_train
|
||||
|
||||
echo "Building alpha_rl_train (ml-alpha + ml-backtesting dev-dep)..."
|
||||
cargo build --release -p ml-alpha --example alpha_rl_train
|
||||
# Train
|
||||
OUT="/feature-cache/alpha-rl-runs/$SHA/fold{{workflow.parameters.fold-idx}}"
|
||||
mkdir -p "$OUT"
|
||||
echo "=== Train on $(nvidia-smi --query-gpu=name --format=csv,noheader) ==="
|
||||
stdbuf -oL ${CARGO_TARGET_DIR}/release/examples/alpha_rl_train \
|
||||
--mbp10-data-dir /data/futures-baseline-mbp10/ES.FUT \
|
||||
--predecoded-dir /feature-cache/predecoded \
|
||||
--out "$OUT" \
|
||||
--n-steps {{workflow.parameters.n-steps}} \
|
||||
--seq-len {{workflow.parameters.seq-len}} \
|
||||
--n-backtests {{workflow.parameters.n-backtests}} \
|
||||
--per-capacity {{workflow.parameters.per-capacity}} \
|
||||
--seed {{workflow.parameters.seed}} \
|
||||
--instrument-mode "{{workflow.parameters.instrument-mode}}" \
|
||||
--fold-idx {{workflow.parameters.fold-idx}} \
|
||||
--n-folds {{workflow.parameters.n-folds}} \
|
||||
--n-eval-steps {{workflow.parameters.n-eval-steps}} \
|
||||
--log-every {{workflow.parameters.log-every}}
|
||||
|
||||
mkdir -p "$BIN_DIR"
|
||||
cp "$CARGO_TARGET_DIR/release/examples/alpha_rl_train" "$BIN_DIR/"
|
||||
strip "$BIN_DIR/alpha_rl_train"
|
||||
ls -lh "$BIN_DIR/"
|
||||
|
||||
# Prune old cache (keep last 5 SHAs).
|
||||
cd /data/bin
|
||||
ls -1t | tail -n +6 | while read -r old; do
|
||||
echo "Pruning old cache: $old"; rm -rf "$old"
|
||||
done
|
||||
echo "$SHORT_SHA" > /tmp/sha
|
||||
|
||||
# ── warmup-gpu: trigger autoscaler scale-up of the GPU pool ──
|
||||
# Same pattern as alpha-perception — tiny alpine pod scheduled with
|
||||
# the GPU pool's nodeSelector but no GPU resource request, so it
|
||||
# doesn't compete with `train` for the single GPU.
|
||||
- name: warmup-gpu
|
||||
nodeSelector:
|
||||
k8s.scaleway.com/pool-name: "{{workflow.parameters.gpu-pool}}"
|
||||
topology.kubernetes.io/zone: fr-par-2
|
||||
tolerations:
|
||||
- key: nvidia.com/gpu
|
||||
operator: Exists
|
||||
effect: NoSchedule
|
||||
- key: node.cilium.io/agent-not-ready
|
||||
operator: Exists
|
||||
effect: NoSchedule
|
||||
container:
|
||||
image: alpine:3.20
|
||||
command: ["/bin/sh", "-c"]
|
||||
resources:
|
||||
requests:
|
||||
cpu: "50m"
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: "100m"
|
||||
memory: 64Mi
|
||||
args:
|
||||
- |
|
||||
echo "GPU warmup pod scheduled on $(hostname) — autoscaler triggered, node in scaledown-grace window."
|
||||
|
||||
# ── train: run alpha_rl_train on L40S (default) or H100 (opt-in) ──
|
||||
- name: train
|
||||
inputs:
|
||||
parameters:
|
||||
- name: sha
|
||||
nodeSelector:
|
||||
k8s.scaleway.com/pool-name: "{{workflow.parameters.gpu-pool}}"
|
||||
topology.kubernetes.io/zone: fr-par-2
|
||||
tolerations:
|
||||
- key: nvidia.com/gpu
|
||||
operator: Exists
|
||||
effect: NoSchedule
|
||||
- key: node.cilium.io/agent-not-ready
|
||||
operator: Exists
|
||||
effect: NoSchedule
|
||||
container:
|
||||
image: gitlab-registry.foxhunt.svc.cluster.local:5000/root/foxhunt/foxhunt-training-runtime:latest
|
||||
imagePullPolicy: Always
|
||||
command: ["/bin/bash", "-c"]
|
||||
env:
|
||||
- name: RUST_LOG
|
||||
value: info
|
||||
- name: SQLX_OFFLINE
|
||||
value: "true"
|
||||
- name: CUDA_COMPUTE_CAP
|
||||
value: "{{workflow.parameters.cuda-compute-cap}}"
|
||||
resources:
|
||||
# L40S-1-48G has 8 vCPU and ~91Gi memory. Request <7800m so
|
||||
# the pod fits alongside system daemonsets on the same node.
|
||||
requests:
|
||||
cpu: "6"
|
||||
memory: 16Gi
|
||||
nvidia.com/gpu: "1"
|
||||
limits:
|
||||
cpu: "7"
|
||||
memory: 64Gi
|
||||
nvidia.com/gpu: "1"
|
||||
volumeMounts:
|
||||
- name: training-data
|
||||
mountPath: /data
|
||||
- name: feature-cache
|
||||
mountPath: /feature-cache
|
||||
args:
|
||||
- |
|
||||
set -e
|
||||
SHA="{{inputs.parameters.sha}}"
|
||||
BIN="/data/bin/$SHA/alpha_rl_train"
|
||||
MBP10_DIR="/data/futures-baseline-mbp10/ES.FUT"
|
||||
PREDECODED_DIR="/feature-cache/predecoded"
|
||||
# Per-fold output dir so concurrent multi-fold G8 submissions
|
||||
# don't clobber each other's eval_summary.json. Single-fold
|
||||
# smokes (n_folds=1) still land under the SHA dir directly
|
||||
# since fold_idx=0 → suffix "fold0".
|
||||
OUT_DIR="/feature-cache/alpha-rl-runs/$SHA/fold{{workflow.parameters.fold-idx}}"
|
||||
mkdir -p "$OUT_DIR" "$PREDECODED_DIR"
|
||||
|
||||
if [ ! -x "$BIN" ]; then
|
||||
echo "ERROR: binary not found at $BIN"
|
||||
ls -lh /data/bin/ || true
|
||||
exit 1
|
||||
echo "=== Complete: $OUT ==="
|
||||
ls -lh "$OUT/"
|
||||
if [ -f "$OUT/alpha_rl_train_summary.json" ]; then
|
||||
cat "$OUT/alpha_rl_train_summary.json"
|
||||
fi
|
||||
if [ ! -d "$MBP10_DIR" ]; then
|
||||
echo "ERROR: MBP-10 data directory not found at $MBP10_DIR"
|
||||
ls -lh /data/ || true
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Running alpha_rl_train (DQN + PPO on shared Mamba2→CfC encoder, R7d off-policy)"
|
||||
echo " binary: $BIN"
|
||||
echo " mbp10: $MBP10_DIR ($(find "$MBP10_DIR" -name '*.dbn.zst' | wc -l) files)"
|
||||
echo " predecoded: $PREDECODED_DIR"
|
||||
echo " out: $OUT_DIR"
|
||||
echo " n-steps: {{workflow.parameters.n-steps}}"
|
||||
echo " seq-len: {{workflow.parameters.seq-len}}"
|
||||
echo " n-backtests: {{workflow.parameters.n-backtests}}"
|
||||
echo " per-capacity: {{workflow.parameters.per-capacity}}"
|
||||
echo " seed: {{workflow.parameters.seed}}"
|
||||
echo " instrument: {{workflow.parameters.instrument-mode}}"
|
||||
|
||||
"$BIN" \
|
||||
--mbp10-data-dir "$MBP10_DIR" \
|
||||
--predecoded-dir "$PREDECODED_DIR" \
|
||||
--out "$OUT_DIR" \
|
||||
--n-steps {{workflow.parameters.n-steps}} \
|
||||
--seq-len {{workflow.parameters.seq-len}} \
|
||||
--n-backtests {{workflow.parameters.n-backtests}} \
|
||||
--per-capacity {{workflow.parameters.per-capacity}} \
|
||||
--seed {{workflow.parameters.seed}} \
|
||||
--instrument-mode "{{workflow.parameters.instrument-mode}}" \
|
||||
--fold-idx {{workflow.parameters.fold-idx}} \
|
||||
--n-folds {{workflow.parameters.n-folds}} \
|
||||
--n-eval-steps {{workflow.parameters.n-eval-steps}}
|
||||
RC=$?
|
||||
|
||||
echo "=== alpha_rl_train exited with code $RC ==="
|
||||
if [ -f "$OUT_DIR/alpha_rl_train_summary.json" ]; then
|
||||
echo "--- alpha_rl_train_summary.json ---"
|
||||
cat "$OUT_DIR/alpha_rl_train_summary.json"
|
||||
else
|
||||
echo "WARNING: alpha_rl_train_summary.json not present in $OUT_DIR"
|
||||
ls -lh "$OUT_DIR/" || true
|
||||
fi
|
||||
exit $RC
|
||||
|
||||
Reference in New Issue
Block a user