#!/usr/bin/env bash # Train a model via Argo Workflows. # # Usage: # ./scripts/argo-train.sh dqn # defaults: HEAD, L40S, 50 epochs # ./scripts/argo-train.sh dqn --sha abc1234 # specific commit # ./scripts/argo-train.sh dqn --epochs 100 --trials 40 # override training params # ./scripts/argo-train.sh ppo --gpu-pool ci-training-h100 # opt into H100 (80 GB) # ./scripts/argo-train.sh dqn --baseline # skip hyperopt # ./scripts/argo-train.sh dqn --watch # follow logs # # Supported models: # RL: dqn, ppo # Supervised: tft, mamba2, tggn, tlob, liquid, kan, xlstm, diffusion # # Requires: argo CLI set -euo pipefail SHA="HEAD" BRANCH="main" TRIALS="" EPOCHS="" GPU_POOL="" SYMBOL="" WATCH=false BASELINE=false CAPITAL="" SANITIZER="none" MULTI_SEED=1 FOLDS=1 DRY_RUN=false TAG="" PROFILE=false IMBALANCE_BAR_THRESHOLD="" IMBALANCE_BAR_EWMA_ALPHA="" VOLUME_BAR_SIZE="" DATA_SOURCE="" usage() { cat < [OPTIONS] Models: dqn, ppo (RL) tft, mamba2, tggn, tlob, liquid, kan, xlstm (supervised) Options: --sha Git commit SHA (default: HEAD) --branch Git branch (default: main) --trials Hyperopt trials (default: 20) --epochs Training epochs (default: 50) --gpu-pool GPU node pool (default: ci-training-l40s; opt into ci-training-h100 for 80 GB / sm_90 workloads) --symbol Trading symbol (default: ES.FUT) --capital Initial capital (default: 35000) --baseline Skip hyperopt (trials=0) --sanitizer Run under compute-sanitizer (memcheck|racecheck|synccheck) --watch Follow workflow logs --multi-seed Run N seeds in parallel (default: 1, fans out via DAG when >1) --folds Walk-forward fold count (default: 1, fans out via DAG when >1) --tag Label workflow with foxhunt-tag= for log aggregation --imbalance-bar-threshold Override imbalance bar threshold (default: 0.5) MUST match between precompute_features and trainer. --imbalance-bar-ewma-alpha Override imbalance bar EWMA alpha (default: 0.1). α=1.0 disables adaptation (this codebase's reversed convention). MUST match between precompute and trainer. --volume-bar-size Override volume bar size in contracts (default: 100). Used when --data-source != "mbp10". Cache key includes. --data-source Data source mode: "mbp10" or "ohlcv" (default: "mbp10"). Threaded into both precompute and trainer. --profile Wrap training under nsys (NVIDIA Nsight Systems) and upload .nsys-rep artefacts to MinIO bucket foxhunt-training-artifacts/profiles//. Forces the multi-seed render path (template rendered locally) so the nsys wrapper is visible in --dry-run output without cluster contact. Plan 5 Task 3 (A.4.1). --dry-run Print rendered workflow YAML to stdout, do not submit -h, --help Show this help EOF exit 0 } [[ $# -eq 0 ]] && { echo "Error: model argument required"; usage; } MODEL="$1"; shift case "$MODEL" in dqn|ppo|tft|mamba2|tggn|tlob|liquid|kan|xlstm|diffusion) ;; *) echo "Error: unknown model '$MODEL'"; usage ;; esac while [[ $# -gt 0 ]]; do case $1 in --sha) SHA="$2"; shift 2 ;; --branch) BRANCH="$2"; shift 2 ;; --trials) TRIALS="$2"; shift 2 ;; --epochs) EPOCHS="$2"; shift 2 ;; --gpu-pool) GPU_POOL="$2"; shift 2 ;; --symbol) SYMBOL="$2"; shift 2 ;; --capital) CAPITAL="$2"; shift 2 ;; --baseline) BASELINE=true; shift ;; --sanitizer) SANITIZER="$2"; shift 2 ;; --watch) WATCH=true; shift ;; --multi-seed) MULTI_SEED="$2"; shift 2 ;; --folds) FOLDS="$2"; shift 2 ;; --tag) TAG="$2"; shift 2 ;; --profile) PROFILE=true; shift ;; --imbalance-bar-threshold) IMBALANCE_BAR_THRESHOLD="$2"; shift 2 ;; --imbalance-bar-ewma-alpha) IMBALANCE_BAR_EWMA_ALPHA="$2"; shift 2 ;; --volume-bar-size) VOLUME_BAR_SIZE="$2"; shift 2 ;; --data-source) DATA_SOURCE="$2"; shift 2 ;; --dry-run) DRY_RUN=true; shift ;; -h|--help) usage ;; *) echo "Unknown option: $1"; usage ;; esac done # Validate seed/fold counts are positive integers. if ! [[ "$MULTI_SEED" =~ ^[0-9]+$ ]] || [[ "$MULTI_SEED" -lt 1 ]]; then echo "Error: --multi-seed must be a positive integer, got '$MULTI_SEED'" exit 1 fi if ! [[ "$FOLDS" =~ ^[0-9]+$ ]] || [[ "$FOLDS" -lt 1 ]]; then echo "Error: --folds must be a positive integer, got '$FOLDS'" exit 1 fi # Auto-derive cuda-compute-cap from GPU pool — cubins must match device sm_XX. # Default pool is ci-training-l40s (sm_89) as of 2026-05-14 per # `feedback_default_to_l40s_pool.md`. Override for other architectures: # ci-training-h100* → sm_90 (Hopper) # ci-training-l40s → sm_89 (Ada Lovelace, current default) # ci-training → sm_89 (alias for L40S — pool is named bare in some # clusters; fixed 2026-05-04 after train-mnpf7 deployed # with sm_90 cubins on L40S device, requiring terminate # + resubmit with explicit --gpu-pool ci-training-l40s). case "${GPU_POOL:-ci-training-l40s}" in *h100*) CUDA_COMPUTE_CAP="90" ;; # Hopper (opt-in) *l40s*|ci-training) CUDA_COMPUTE_CAP="89" ;; *) CUDA_COMPUTE_CAP="89" ;; # default Ada Lovelace esac # ── Route: single-job (existing template) vs multi-seed DAG (new template) ── # Backward compat: --multi-seed 1 --folds 1 keeps the existing single-template # call path verbatim. Only when N>1 OR K>1 do we render the matrix DAG. # # --profile forces the multi-seed render path even at N=K=1 because the # single-job dry-run goes through `argo submit --dry-run -o yaml` which # resolves the WorkflowTemplate against a live cluster — not available in # local CI. The multi-seed renderer reads the template file directly so # `--profile --dry-run` works without cluster access. Plan 5 Task 3. USE_MULTI_SEED=false if [[ "$MULTI_SEED" -gt 1 || "$FOLDS" -gt 1 || "$PROFILE" == "true" ]]; then USE_MULTI_SEED=true fi if [[ "$USE_MULTI_SEED" == "false" ]]; then # Apply the local train-template.yaml to the cluster BEFORE submission. # Without this, `argo submit --from=wftmpl/train` runs against whatever # was last applied — defaults can drift from source. Caused workflow # train-jpxvn (2026-05-10) to dispatch with a stale threshold=0.5 # default and trigger near-OOM. Multi-seed path already does this; the # single-job path was missing it. kubectl apply -n foxhunt -f infra/k8s/argo/train-template.yaml >/dev/null CMD="argo submit -n foxhunt --from=wftmpl/train" CMD="$CMD -p commit-sha=$SHA" CMD="$CMD -p git-branch=$BRANCH" CMD="$CMD -p model=$MODEL" CMD="$CMD -p cuda-compute-cap=$CUDA_COMPUTE_CAP" [[ -n "$TRIALS" ]] && CMD="$CMD -p hyperopt-trials=$TRIALS" [[ -n "$EPOCHS" ]] && CMD="$CMD -p train-epochs=$EPOCHS" [[ -n "$GPU_POOL" ]] && CMD="$CMD -p gpu-pool=$GPU_POOL" [[ -n "$SYMBOL" ]] && CMD="$CMD -p symbol=$SYMBOL" [[ -n "$CAPITAL" ]] && CMD="$CMD -p initial-capital=$CAPITAL" [[ "$SANITIZER" != "none" ]] && CMD="$CMD -p sanitizer=$SANITIZER" [[ -n "$IMBALANCE_BAR_THRESHOLD" ]] && CMD="$CMD -p imbalance-bar-threshold=$IMBALANCE_BAR_THRESHOLD" [[ -n "$IMBALANCE_BAR_EWMA_ALPHA" ]] && CMD="$CMD -p imbalance-bar-ewma-alpha=$IMBALANCE_BAR_EWMA_ALPHA" [[ -n "$VOLUME_BAR_SIZE" ]] && CMD="$CMD -p volume-bar-size=$VOLUME_BAR_SIZE" [[ -n "$DATA_SOURCE" ]] && CMD="$CMD -p data-source=$DATA_SOURCE" $BASELINE && CMD="$CMD -p hyperopt-trials=0" $WATCH && CMD="$CMD --watch" if [[ "$DRY_RUN" == "true" ]]; then # `argo submit --dry-run -o yaml` renders the resolved Workflow without # contacting the cluster. Single-template path emits one Workflow object # (no `kind: WorkflowTask` lines — that string only appears in the # multi-seed DAG matrix). eval "$CMD --dry-run -o yaml" exit 0 fi echo "Submitting $MODEL training workflow..." echo " sha: $SHA" echo " branch: $BRANCH" echo " model: $MODEL" $BASELINE && echo " mode: baseline (no hyperopt)" [[ -n "$TRIALS" ]] && echo " trials: $TRIALS" [[ -n "$EPOCHS" ]] && echo " epochs: $EPOCHS" [[ -n "$GPU_POOL" ]] && echo " gpu: $GPU_POOL" echo " sm: $CUDA_COMPUTE_CAP" echo "" eval "$CMD" exit 0 fi # ── Multi-seed path (one job per seed; folds run inside the binary) ── # Render the train-multi-seed-template.yaml with the per-seed matrix # expanded inline. Plan 5 Task 5 Phase B pivot: was N*K (seed,fold) jobs; # is now N (seed-only) jobs because `train_baseline_rl` is a multi-fold # walk-forward executor — it accepts `--max-folds K`, NOT `--fold K`. Each # rendered task invokes the binary with `--seed N --max-folds K` so the # walk-forward sweep happens inside the single training process. # # The base template ships with a placeholder marker (`# __MATRIX_TASKS__`) # which we replace with N generated WorkflowTask stanzas. TEMPLATE_SRC="infra/k8s/argo/train-multi-seed-template.yaml" if [[ ! -f "$TEMPLATE_SRC" ]]; then echo "Error: multi-seed template not found at $TEMPLATE_SRC" exit 1 fi # Build matrix YAML. Each task is a dag.tasks[] entry that targets the # `train-single` template with the `seed` parameter bound from inputs. # The template forwards `seed` as `--seed` and reads `--max-folds` from the # workflow-scoped `folds` parameter, so each task internally runs all K folds. build_matrix_tasks() { local seeds="$1" # 10 spaces — matches the sibling `- name: ensure-binary` list-item indent # under `dag.tasks:` (which is itself at 8 spaces). Wrong indent here # produces a YAML parse error in the rendered template. local indent=" " local s for ((s=0; s "$TMP_TEMPLATE" echo "Submitting multi-seed $MODEL training workflow..." echo " sha: $SHA" echo " branch: $BRANCH" echo " model: $MODEL" echo " multi-seed: $MULTI_SEED" echo " folds: $FOLDS (each job runs all folds via --max-folds)" echo " total jobs: $MULTI_SEED" [[ "$PROFILE" == "true" ]] && echo " profile: nsys (per-job .nsys-rep upload)" [[ -n "$TAG" ]] && echo " tag: $TAG" [[ -n "$EPOCHS" ]] && echo " epochs: $EPOCHS" [[ -n "$GPU_POOL" ]] && echo " gpu: $GPU_POOL" echo " sm: $CUDA_COMPUTE_CAP" echo "" # Apply the rendered WorkflowTemplate, then submit a workflow from it. kubectl apply -n foxhunt -f "$TMP_TEMPLATE" CMD="argo submit -n foxhunt --from=wftmpl/train-multi-seed" CMD="$CMD -p commit-sha=$SHA" CMD="$CMD -p git-branch=$BRANCH" CMD="$CMD -p model=$MODEL" CMD="$CMD -p cuda-compute-cap=$CUDA_COMPUTE_CAP" CMD="$CMD -p multi-seed=$MULTI_SEED" CMD="$CMD -p folds=$FOLDS" CMD="$CMD -p profile=$PROFILE" [[ -n "$TRIALS" ]] && CMD="$CMD -p hyperopt-trials=$TRIALS" [[ -n "$EPOCHS" ]] && CMD="$CMD -p train-epochs=$EPOCHS" [[ -n "$GPU_POOL" ]] && CMD="$CMD -p gpu-pool=$GPU_POOL" [[ -n "$SYMBOL" ]] && CMD="$CMD -p symbol=$SYMBOL" [[ -n "$CAPITAL" ]] && CMD="$CMD -p initial-capital=$CAPITAL" [[ -n "$TAG" ]] && CMD="$CMD --labels foxhunt-tag=$TAG" [[ -n "$IMBALANCE_BAR_THRESHOLD" ]] && CMD="$CMD -p imbalance-bar-threshold=$IMBALANCE_BAR_THRESHOLD" [[ -n "$IMBALANCE_BAR_EWMA_ALPHA" ]] && CMD="$CMD -p imbalance-bar-ewma-alpha=$IMBALANCE_BAR_EWMA_ALPHA" [[ -n "$VOLUME_BAR_SIZE" ]] && CMD="$CMD -p volume-bar-size=$VOLUME_BAR_SIZE" [[ -n "$DATA_SOURCE" ]] && CMD="$CMD -p data-source=$DATA_SOURCE" [[ "$SANITIZER" != "none" ]] && CMD="$CMD -p sanitizer=$SANITIZER" $BASELINE && CMD="$CMD -p hyperopt-trials=0" $WATCH && CMD="$CMD --watch" eval "$CMD"