Files
foxhunt/scripts/argo-train.sh
2026-05-25 09:26:19 +02:00

329 lines
14 KiB
Bash
Executable File
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env bash
# Train a model via Argo Workflows.
#
# Usage:
# ./scripts/argo-train.sh dqn # defaults: HEAD, L40S, 50 epochs
# ./scripts/argo-train.sh dqn --sha abc1234 # specific commit
# ./scripts/argo-train.sh dqn --epochs 100 --trials 40 # override training params
# ./scripts/argo-train.sh ppo --gpu-pool ci-training-h100 # opt into H100 (80 GB)
# ./scripts/argo-train.sh dqn --baseline # skip hyperopt
# ./scripts/argo-train.sh dqn --watch # follow logs
#
# Supported models:
# RL: dqn, ppo
# Supervised: tft, mamba2, tggn, tlob, liquid, kan, xlstm, diffusion
#
# Requires: argo CLI
set -euo pipefail
SHA="HEAD"
BRANCH="main"
TRIALS=""
EPOCHS=""
GPU_POOL=""
SYMBOL=""
WATCH=false
BASELINE=false
CAPITAL=""
SANITIZER="none"
MULTI_SEED=1
FOLDS=1
DRY_RUN=false
TAG=""
PROFILE=false
IMBALANCE_BAR_THRESHOLD=""
IMBALANCE_BAR_EWMA_ALPHA=""
VOLUME_BAR_SIZE=""
DATA_SOURCE=""
usage() {
cat <<EOF
Usage: $(basename "$0") <model> [OPTIONS]
Models:
dqn, ppo (RL)
tft, mamba2, tggn, tlob, liquid, kan, xlstm (supervised)
Options:
--sha <commit> Git commit SHA (default: HEAD)
--branch <branch> Git branch (default: main)
--trials <n> Hyperopt trials (default: 20)
--epochs <n> Training epochs (default: 50)
--gpu-pool <pool> GPU node pool (default: ci-training-l40s; opt into
ci-training-h100 for 80 GB / sm_90 workloads)
--symbol <sym> Trading symbol (default: ES.FUT)
--capital <n> Initial capital (default: 35000)
--baseline Skip hyperopt (trials=0)
--sanitizer <tool> Run under compute-sanitizer (memcheck|racecheck|synccheck)
--watch Follow workflow logs
--multi-seed <n> Run N seeds in parallel (default: 1, fans out via DAG when >1)
--folds <k> Walk-forward fold count (default: 1, fans out via DAG when >1)
--tag <t> Label workflow with foxhunt-tag=<t> for log aggregation
--imbalance-bar-threshold <f> Override imbalance bar threshold (default: 0.5)
MUST match between precompute_features and trainer.
--imbalance-bar-ewma-alpha <f> Override imbalance bar EWMA alpha (default: 0.1).
α=1.0 disables adaptation (this codebase's reversed
convention). MUST match between precompute and trainer.
--volume-bar-size <n> Override volume bar size in contracts (default: 100).
Used when --data-source != "mbp10". Cache key includes.
--data-source <s> Data source mode: "mbp10" or "ohlcv" (default: "mbp10").
Threaded into both precompute and trainer.
--profile Wrap training under nsys (NVIDIA Nsight Systems) and
upload .nsys-rep artefacts to MinIO bucket
foxhunt-training-artifacts/profiles/<sha>/. Forces the
multi-seed render path (template rendered locally) so
the nsys wrapper is visible in --dry-run output without
cluster contact. Plan 5 Task 3 (A.4.1).
--dry-run Print rendered workflow YAML to stdout, do not submit
-h, --help Show this help
EOF
exit 0
}
[[ $# -eq 0 ]] && { echo "Error: model argument required"; usage; }
MODEL="$1"; shift
case "$MODEL" in
alpha-rl|dqn|ppo|tft|mamba2|tggn|tlob|liquid|kan|xlstm|diffusion) ;;
*) echo "Error: unknown model '$MODEL'"; usage ;;
esac
while [[ $# -gt 0 ]]; do
case $1 in
--sha) SHA="$2"; shift 2 ;;
--branch) BRANCH="$2"; shift 2 ;;
--trials) TRIALS="$2"; shift 2 ;;
--epochs) EPOCHS="$2"; shift 2 ;;
--gpu-pool) GPU_POOL="$2"; shift 2 ;;
--symbol) SYMBOL="$2"; shift 2 ;;
--capital) CAPITAL="$2"; shift 2 ;;
--baseline) BASELINE=true; shift ;;
--sanitizer) SANITIZER="$2"; shift 2 ;;
--watch) WATCH=true; shift ;;
--multi-seed) MULTI_SEED="$2"; shift 2 ;;
--folds) FOLDS="$2"; shift 2 ;;
--tag) TAG="$2"; shift 2 ;;
--profile) PROFILE=true; shift ;;
--imbalance-bar-threshold) IMBALANCE_BAR_THRESHOLD="$2"; shift 2 ;;
--imbalance-bar-ewma-alpha) IMBALANCE_BAR_EWMA_ALPHA="$2"; shift 2 ;;
--volume-bar-size) VOLUME_BAR_SIZE="$2"; shift 2 ;;
--data-source) DATA_SOURCE="$2"; shift 2 ;;
--dry-run) DRY_RUN=true; shift ;;
-h|--help) usage ;;
*) echo "Unknown option: $1"; usage ;;
esac
done
# Validate seed/fold counts are positive integers.
if ! [[ "$MULTI_SEED" =~ ^[0-9]+$ ]] || [[ "$MULTI_SEED" -lt 1 ]]; then
echo "Error: --multi-seed must be a positive integer, got '$MULTI_SEED'"
exit 1
fi
if ! [[ "$FOLDS" =~ ^[0-9]+$ ]] || [[ "$FOLDS" -lt 1 ]]; then
echo "Error: --folds must be a positive integer, got '$FOLDS'"
exit 1
fi
# Auto-derive cuda-compute-cap from GPU pool — cubins must match device sm_XX.
# Default pool is ci-training-l40s (sm_89) as of 2026-05-14 per
# `feedback_default_to_l40s_pool.md`. Override for other architectures:
# ci-training-h100* → sm_90 (Hopper)
# ci-training-l40s → sm_89 (Ada Lovelace, current default)
# ci-training → sm_89 (alias for L40S — pool is named bare in some
# clusters; fixed 2026-05-04 after train-mnpf7 deployed
# with sm_90 cubins on L40S device, requiring terminate
# + resubmit with explicit --gpu-pool ci-training-l40s).
case "${GPU_POOL:-ci-training-l40s}" in
*h100*) CUDA_COMPUTE_CAP="90" ;; # Hopper (opt-in)
*l40s*|ci-training) CUDA_COMPUTE_CAP="89" ;;
*) CUDA_COMPUTE_CAP="89" ;; # default Ada Lovelace
esac
# ── Route: single-job (existing template) vs multi-seed DAG (new template) ──
# Backward compat: --multi-seed 1 --folds 1 keeps the existing single-template
# call path verbatim. Only when N>1 OR K>1 do we render the matrix DAG.
#
# --profile forces the multi-seed render path even at N=K=1 because the
# single-job dry-run goes through `argo submit --dry-run -o yaml` which
# resolves the WorkflowTemplate against a live cluster — not available in
# local CI. The multi-seed renderer reads the template file directly so
# `--profile --dry-run` works without cluster access. Plan 5 Task 3.
USE_MULTI_SEED=false
if [[ "$MULTI_SEED" -gt 1 || "$FOLDS" -gt 1 || "$PROFILE" == "true" ]]; then
USE_MULTI_SEED=true
fi
if [[ "$USE_MULTI_SEED" == "false" ]]; then
# Apply the local train-template.yaml to the cluster BEFORE submission.
# Without this, `argo submit --from=wftmpl/train` runs against whatever
# was last applied — defaults can drift from source. Caused workflow
# train-jpxvn (2026-05-10) to dispatch with a stale threshold=0.5
# default and trigger near-OOM. Multi-seed path already does this; the
# single-job path was missing it.
kubectl apply -n foxhunt -f infra/k8s/argo/train-template.yaml >/dev/null
CMD="argo submit -n foxhunt --from=wftmpl/train"
CMD="$CMD -p commit-sha=$SHA"
CMD="$CMD -p git-branch=$BRANCH"
CMD="$CMD -p model=$MODEL"
CMD="$CMD -p cuda-compute-cap=$CUDA_COMPUTE_CAP"
[[ -n "$TRIALS" ]] && CMD="$CMD -p hyperopt-trials=$TRIALS"
[[ -n "$EPOCHS" ]] && CMD="$CMD -p train-epochs=$EPOCHS"
[[ -n "$GPU_POOL" ]] && CMD="$CMD -p gpu-pool=$GPU_POOL"
[[ -n "$SYMBOL" ]] && CMD="$CMD -p symbol=$SYMBOL"
[[ -n "$CAPITAL" ]] && CMD="$CMD -p initial-capital=$CAPITAL"
[[ "$SANITIZER" != "none" ]] && CMD="$CMD -p sanitizer=$SANITIZER"
[[ -n "$IMBALANCE_BAR_THRESHOLD" ]] && CMD="$CMD -p imbalance-bar-threshold=$IMBALANCE_BAR_THRESHOLD"
[[ -n "$IMBALANCE_BAR_EWMA_ALPHA" ]] && CMD="$CMD -p imbalance-bar-ewma-alpha=$IMBALANCE_BAR_EWMA_ALPHA"
[[ -n "$VOLUME_BAR_SIZE" ]] && CMD="$CMD -p volume-bar-size=$VOLUME_BAR_SIZE"
[[ -n "$DATA_SOURCE" ]] && CMD="$CMD -p data-source=$DATA_SOURCE"
$BASELINE && CMD="$CMD -p hyperopt-trials=0"
$WATCH && CMD="$CMD --watch"
if [[ "$DRY_RUN" == "true" ]]; then
# `argo submit --dry-run -o yaml` renders the resolved Workflow without
# contacting the cluster. Single-template path emits one Workflow object
# (no `kind: WorkflowTask` lines — that string only appears in the
# multi-seed DAG matrix).
eval "$CMD --dry-run -o yaml"
exit 0
fi
echo "Submitting $MODEL training workflow..."
echo " sha: $SHA"
echo " branch: $BRANCH"
echo " model: $MODEL"
$BASELINE && echo " mode: baseline (no hyperopt)"
[[ -n "$TRIALS" ]] && echo " trials: $TRIALS"
[[ -n "$EPOCHS" ]] && echo " epochs: $EPOCHS"
[[ -n "$GPU_POOL" ]] && echo " gpu: $GPU_POOL"
echo " sm: $CUDA_COMPUTE_CAP"
echo ""
eval "$CMD"
exit 0
fi
# ── Multi-seed path (one job per seed; folds run inside the binary) ──
# Render the train-multi-seed-template.yaml with the per-seed matrix
# expanded inline. Plan 5 Task 5 Phase B pivot: was N*K (seed,fold) jobs;
# is now N (seed-only) jobs because `train_baseline_rl` is a multi-fold
# walk-forward executor — it accepts `--max-folds K`, NOT `--fold K`. Each
# rendered task invokes the binary with `--seed N --max-folds K` so the
# walk-forward sweep happens inside the single training process.
#
# The base template ships with a placeholder marker (`# __MATRIX_TASKS__`)
# which we replace with N generated WorkflowTask stanzas.
TEMPLATE_SRC="infra/k8s/argo/train-multi-seed-template.yaml"
if [[ ! -f "$TEMPLATE_SRC" ]]; then
echo "Error: multi-seed template not found at $TEMPLATE_SRC"
exit 1
fi
# Build matrix YAML. Each task is a dag.tasks[] entry that targets the
# `train-single` template with the `seed` parameter bound from inputs.
# The template forwards `seed` as `--seed` and reads `--max-folds` from the
# workflow-scoped `folds` parameter, so each task internally runs all K folds.
build_matrix_tasks() {
local seeds="$1"
# 10 spaces — matches the sibling `- name: ensure-binary` list-item indent
# under `dag.tasks:` (which is itself at 8 spaces). Wrong indent here
# produces a YAML parse error in the rendered template.
local indent=" "
local s
local deps
if [ "$MODEL" = "alpha-rl" ]; then
deps="[ensure-binary, gpu-warmup]"
else
deps="[ensure-fxcache, gpu-warmup]"
fi
for ((s=0; s<seeds; s++)); do
cat <<EOF
${indent}- name: train-s${s}
${indent} template: train-single
${indent} dependencies: ${deps}
${indent} arguments:
${indent} parameters:
${indent} - name: seed
${indent} value: "${s}"
EOF
done
}
MATRIX_TASKS=$(build_matrix_tasks "$MULTI_SEED")
# Substitute the placeholder. Match the *exact* placeholder line (leading
# whitespace + the marker as the only content on the line) — the marker
# string also appears in doc comments above and we must not replace those.
# Use awk (not sed) — multi-line replacement with sed is fragile across
# platforms.
RENDERED=$(awk -v repl="$MATRIX_TASKS" '
/^[[:space:]]*# __MATRIX_TASKS__[[:space:]]*$/ { print repl; next }
{ print }
' "$TEMPLATE_SRC")
if [[ "$DRY_RUN" == "true" ]]; then
# Emit the rendered template + a synthetic per-task marker line so test
# harnesses can count generated jobs without piping through `argo submit`
# (which would require a live cluster). The `kind: WorkflowTask` marker
# is what the test_multi_seed_harness.sh asserts against. Plan 5 Task 5
# Phase B: one marker per seed (not per (seed, fold) pair) because each
# job now sweeps all K folds via `--max-folds`.
echo "$RENDERED"
for ((s=0; s<MULTI_SEED; s++)); do
echo "# kind: WorkflowTask seed=${s} max_folds=${FOLDS}"
done
exit 0
fi
# Submit: write the rendered template to a temp file, apply it as a
# WorkflowTemplate, then submit a Workflow that references it.
TMP_TEMPLATE=$(mktemp -t train-multi-seed.XXXXXX.yaml)
trap 'rm -f "$TMP_TEMPLATE"' EXIT
echo "$RENDERED" > "$TMP_TEMPLATE"
echo "Submitting multi-seed $MODEL training workflow..."
echo " sha: $SHA"
echo " branch: $BRANCH"
echo " model: $MODEL"
echo " multi-seed: $MULTI_SEED"
echo " folds: $FOLDS (each job runs all folds via --max-folds)"
echo " total jobs: $MULTI_SEED"
[[ "$PROFILE" == "true" ]] && echo " profile: nsys (per-job .nsys-rep upload)"
[[ -n "$TAG" ]] && echo " tag: $TAG"
[[ -n "$EPOCHS" ]] && echo " epochs: $EPOCHS"
[[ -n "$GPU_POOL" ]] && echo " gpu: $GPU_POOL"
echo " sm: $CUDA_COMPUTE_CAP"
echo ""
# Apply the rendered WorkflowTemplate, then submit a workflow from it.
kubectl apply -n foxhunt -f "$TMP_TEMPLATE"
CMD="argo submit -n foxhunt --from=wftmpl/train-multi-seed"
CMD="$CMD -p commit-sha=$SHA"
CMD="$CMD -p git-branch=$BRANCH"
CMD="$CMD -p model=$MODEL"
CMD="$CMD -p cuda-compute-cap=$CUDA_COMPUTE_CAP"
CMD="$CMD -p multi-seed=$MULTI_SEED"
CMD="$CMD -p folds=$FOLDS"
CMD="$CMD -p profile=$PROFILE"
[[ -n "$TRIALS" ]] && CMD="$CMD -p hyperopt-trials=$TRIALS"
[[ -n "$EPOCHS" ]] && CMD="$CMD -p train-epochs=$EPOCHS"
[[ -n "$GPU_POOL" ]] && CMD="$CMD -p gpu-pool=$GPU_POOL"
[[ -n "$SYMBOL" ]] && CMD="$CMD -p symbol=$SYMBOL"
[[ -n "$CAPITAL" ]] && CMD="$CMD -p initial-capital=$CAPITAL"
[[ -n "$TAG" ]] && CMD="$CMD --labels foxhunt-tag=$TAG"
[[ -n "$IMBALANCE_BAR_THRESHOLD" ]] && CMD="$CMD -p imbalance-bar-threshold=$IMBALANCE_BAR_THRESHOLD"
[[ -n "$IMBALANCE_BAR_EWMA_ALPHA" ]] && CMD="$CMD -p imbalance-bar-ewma-alpha=$IMBALANCE_BAR_EWMA_ALPHA"
[[ -n "$VOLUME_BAR_SIZE" ]] && CMD="$CMD -p volume-bar-size=$VOLUME_BAR_SIZE"
[[ -n "$DATA_SOURCE" ]] && CMD="$CMD -p data-source=$DATA_SOURCE"
[[ "$SANITIZER" != "none" ]] && CMD="$CMD -p sanitizer=$SANITIZER"
$BASELINE && CMD="$CMD -p hyperopt-trials=0"
$WATCH && CMD="$CMD --watch"
eval "$CMD"