infra(argo): alpha-perception workflow + submission script

Argo WorkflowTemplate at infra/k8s/argo/alpha-perception-template.yaml
runs the stacked Mamba2 -> CfC -> heads PerceptionTrainer on a single
L40S in fr-par-2. Two-stage DAG:
  ensure-binary  (ci-compile-cpu pool, sccache-backed cargo build of
                  alpha_train example, SHA-keyed binary cache under
                  /data/bin/$SHORT_SHA/)
  train          (ci-training-l40s pool, runs the cached binary against
                  /data/futures-baseline/mbp10 with predecoded sidecar
                  cache at /feature-cache/predecoded, writes
                  alpha_train_summary.json to
                  /feature-cache/alpha-perception-runs/$SHA/)

Defaults mirror the validated synthetic-overfit smoke config:
  epochs=5, seq_len=32, mamba2_state_dim=16, lr_cfc=3e-3,
  lr_mamba2=1e-3, n_train_seqs=8000, n_val_seqs=1000, seed=0x4242

Submission script scripts/argo-alpha-perception.sh wraps argo submit
with the standard L40S/H100 cuda-compute-cap mapping. --watch
follows logs.

Workflow nodeSelector pinned to fr-par-2 (consistent with the cluster
topology constraint). ttlStrategy 1h after completion;
activeDeadlineSeconds 4h cap (well above expected ~30-90 min wall).

This is the cluster entrypoint for the stacked perception design.
Once it lands a summary on MinIO, the gate runner (alpha_gate, Task
17) can consume it.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
jgrusewski
2026-05-16 23:10:37 +02:00
parent 1dc1f3563c
commit 522178b2a7
2 changed files with 400 additions and 0 deletions

110
scripts/argo-alpha-perception.sh Executable file
View File

@@ -0,0 +1,110 @@
#!/usr/bin/env bash
# Submit the alpha-perception workflow.
#
# Trains the stacked Mamba2 -> CfC -> heads perception model on MBP-10
# from the training-data PVC. Emits alpha_train_summary.json with
# per-horizon validation AUC to the feature-cache PVC.
#
# Defaults match the validated synthetic-overfit smoke config; cluster
# runs can override for sweep work.
#
# Usage:
# ./scripts/argo-alpha-perception.sh # current HEAD on ml-alpha-phase-a
# ./scripts/argo-alpha-perception.sh --branch main
# ./scripts/argo-alpha-perception.sh --epochs 1 --n-train-seqs 1000 # quick smoke
# ./scripts/argo-alpha-perception.sh --watch
set -euo pipefail
SHA=HEAD
BRANCH=$(git symbolic-ref --short HEAD 2>/dev/null || echo "ml-alpha-phase-a")
GPU_POOL=ci-training-l40s
EPOCHS=5
SEQ_LEN=32
MAMBA2_STATE_DIM=16
LR_CFC=3e-3
LR_MAMBA2=1e-3
N_TRAIN_SEQS=8000
N_VAL_SEQS=1000
SEED=0x4242
WATCH=false
usage() {
cat <<EOF
Usage: $0 [OPTIONS]
--sha <commit> Git SHA (default: HEAD on the current branch)
--branch <branch> Git branch (default: $BRANCH)
--gpu-pool <pool> GPU pool (default: $GPU_POOL)
--epochs <n> Training epochs (default: $EPOCHS)
--seq-len <n> Snapshots per sequence window (default: $SEQ_LEN)
--mamba2-state-dim <n> Mamba2 SSM state dim (default: $MAMBA2_STATE_DIM)
--lr-cfc <f> CfC learning rate (default: $LR_CFC)
--lr-mamba2 <f> Mamba2 learning rate (default: $LR_MAMBA2)
--n-train-seqs <n> Train sequences per epoch (default: $N_TRAIN_SEQS)
--n-val-seqs <n> Val sequences per epoch (default: $N_VAL_SEQS)
--seed <n> Random seed (default: $SEED)
--watch Follow logs via argo watch
EOF
}
while [[ $# -gt 0 ]]; do
case "$1" in
--sha) SHA="$2"; shift 2 ;;
--branch) BRANCH="$2"; shift 2 ;;
--gpu-pool) GPU_POOL="$2"; shift 2 ;;
--epochs) EPOCHS="$2"; shift 2 ;;
--seq-len) SEQ_LEN="$2"; shift 2 ;;
--mamba2-state-dim) MAMBA2_STATE_DIM="$2"; shift 2 ;;
--lr-cfc) LR_CFC="$2"; shift 2 ;;
--lr-mamba2) LR_MAMBA2="$2"; shift 2 ;;
--n-train-seqs) N_TRAIN_SEQS="$2"; shift 2 ;;
--n-val-seqs) N_VAL_SEQS="$2"; shift 2 ;;
--seed) SEED="$2"; shift 2 ;;
--watch) WATCH=true; shift ;;
-h|--help) usage; exit 0 ;;
*) echo "Unknown option: $1"; usage; exit 1 ;;
esac
done
case "$GPU_POOL" in
ci-training-l40s)
SM=89 ;;
ci-training-h100*|ci-training-h100x2|ci-training-h100-sxm)
SM=90 ;;
*)
echo "Unknown gpu-pool: $GPU_POOL"
exit 1 ;;
esac
echo "Submitting alpha-perception workflow..."
echo " branch: $BRANCH"
echo " sha: $SHA"
echo " gpu-pool: $GPU_POOL (sm_$SM)"
echo " epochs: $EPOCHS"
echo " seq-len: $SEQ_LEN"
echo " mamba2-state-dim: $MAMBA2_STATE_DIM"
echo " lr-cfc: $LR_CFC"
echo " lr-mamba2: $LR_MAMBA2"
echo " n-train-seqs: $N_TRAIN_SEQS"
echo " n-val-seqs: $N_VAL_SEQS"
echo " seed: $SEED"
WATCH_FLAG=""
if [[ "$WATCH" == "true" ]]; then
WATCH_FLAG="--watch"
fi
argo submit -n foxhunt --from=wftmpl/alpha-perception \
-p commit-sha="$SHA" \
-p git-branch="$BRANCH" \
-p cuda-compute-cap="$SM" \
-p gpu-pool="$GPU_POOL" \
-p epochs="$EPOCHS" \
-p seq-len="$SEQ_LEN" \
-p mamba2-state-dim="$MAMBA2_STATE_DIM" \
-p lr-cfc="$LR_CFC" \
-p lr-mamba2="$LR_MAMBA2" \
-p n-train-seqs="$N_TRAIN_SEQS" \
-p n-val-seqs="$N_VAL_SEQS" \
-p seed="$SEED" \
$WATCH_FLAG