# alpha-rl: single-pod compile + train on GPU. # # Compiles alpha_rl_train incrementally on the GPU node (~3s warm, # ~90s cold) using the cargo-target-cuda PVC, then runs training # immediately. No separate compile node, no binary transfer, no # fxcache step. Predecoded MBP-10 sidecars live on feature-cache PVC. # # Usage: # argo submit -n foxhunt --from=wftmpl/alpha-rl \ # -p git-branch=ml-alpha-phase-a # argo submit -n foxhunt --from=wftmpl/alpha-rl \ # -p git-branch=ml-alpha-phase-a -p n-steps=50000 -p n-backtests=128 apiVersion: argoproj.io/v1alpha1 kind: WorkflowTemplate metadata: name: alpha-rl namespace: foxhunt labels: app.kubernetes.io/name: alpha-rl app.kubernetes.io/part-of: foxhunt app.kubernetes.io/component: train spec: entrypoint: compile-and-train serviceAccountName: argo-workflow archiveLogs: true podMetadata: labels: app.kubernetes.io/part-of: foxhunt app.kubernetes.io/component: train securityContext: fsGroup: 0 ttlStrategy: secondsAfterCompletion: 3600 activeDeadlineSeconds: 14400 arguments: parameters: - name: git-branch value: ml-alpha-phase-a - name: gpu-pool value: ci-training-l40s - name: n-steps value: "50000" - name: n-backtests value: "128" - name: seed value: "16962" - name: seq-len value: "32" - name: per-capacity value: "32768" - name: instrument-mode value: "all" - name: log-every value: "5000" - name: fold-idx value: "0" - name: n-folds value: "1" - name: n-eval-steps value: "0" - name: nsys-profile value: "false" volumes: - name: git-ssh-key secret: secretName: argo-git-ssh-key defaultMode: 256 - name: training-data persistentVolumeClaim: claimName: training-data-pvc - name: cargo-target-cuda persistentVolumeClaim: claimName: cargo-target-cuda - name: feature-cache persistentVolumeClaim: claimName: feature-cache-pvc templates: - name: compile-and-train nodeSelector: k8s.scaleway.com/pool-name: "{{workflow.parameters.gpu-pool}}" topology.kubernetes.io/zone: fr-par-2 tolerations: - key: nvidia.com/gpu operator: Exists effect: NoSchedule - key: node.cilium.io/agent-not-ready operator: Exists effect: NoSchedule container: image: gitlab-registry.foxhunt.svc.cluster.local:5000/root/foxhunt/ci-builder:latest imagePullPolicy: Always command: ["/bin/bash", "-c"] env: - name: SQLX_OFFLINE value: "true" - name: CARGO_TARGET_DIR value: /cargo-target - name: CARGO_HOME value: /cargo-target/cargo-home - name: CARGO_INCREMENTAL value: "1" resources: requests: cpu: "4" memory: 16Gi nvidia.com/gpu: "1" limits: cpu: "7" memory: 40Gi nvidia.com/gpu: "1" volumeMounts: - name: git-ssh-key mountPath: /etc/git-ssh readOnly: true - name: training-data mountPath: /data readOnly: true - name: feature-cache mountPath: /feature-cache - name: cargo-target-cuda mountPath: /cargo-target args: - | set -e export PATH="${CARGO_HOME}/bin:${PATH}" export LD_LIBRARY_PATH=$(echo "$LD_LIBRARY_PATH" | tr ':' '\n' | grep -v stubs | tr '\n' ':' | sed 's/:$//') # SSH mkdir -p ~/.ssh cp /etc/git-ssh/ssh-privatekey ~/.ssh/id_ed25519 chmod 600 ~/.ssh/id_ed25519 printf 'Host *\n StrictHostKeyChecking no\n UserKnownHostsFile /dev/null\n' > ~/.ssh/config chmod 600 ~/.ssh/config nvidia-smi # Git BRANCH="{{workflow.parameters.git-branch}}" BUILD="/cargo-target/src" REPO="ssh://git@gitlab-gitlab-shell.foxhunt.svc.cluster.local:2222/root/foxhunt.git" git config --global --add safe.directory "$BUILD" if [ -d "$BUILD/.git" ]; then cd "$BUILD" git fetch origin git checkout --force "origin/$BRANCH" git clean -fd else git clone --filter=blob:none "$REPO" "$BUILD" cd "$BUILD" git checkout "origin/$BRANCH" fi SHA=$(git rev-parse --short=9 HEAD) echo "=== Branch: $BRANCH SHA: $SHA ===" # Auto-detect GPU compute capability for all crates' build.rs export CUDA_COMPUTE_CAP=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader | head -1 | tr -d '.') export FOXHUNT_CUDA_ARCH="sm_${CUDA_COMPUTE_CAP}" echo "=== GPU arch: sm_${CUDA_COMPUTE_CAP} ===" # Clear stale build artifacts when arch changes rm -rf ${CARGO_TARGET_DIR}/build/ml-alpha-* ${CARGO_TARGET_DIR}/build/ml-core-* # Compile (~90s cold, ~3s warm after first run) echo "=== Compile ===" time cargo build --release -p ml-alpha --example alpha_rl_train 2>&1 # Train OUT="/feature-cache/alpha-rl-runs/$SHA/fold{{workflow.parameters.fold-idx}}" mkdir -p "$OUT" echo "=== Train on $(nvidia-smi --query-gpu=name --format=csv,noheader) ===" TRAIN_BIN="${CARGO_TARGET_DIR}/release/examples/alpha_rl_train" if [ "{{workflow.parameters.nsys-profile}}" = "true" ]; then TRAIN_CMD="nsys profile -o $OUT/nsys_trace --stats=true --force-overwrite=true $TRAIN_BIN" else TRAIN_CMD="stdbuf -oL $TRAIN_BIN" fi $TRAIN_CMD \ --mbp10-data-dir /data/futures-baseline-mbp10/ES.FUT \ --predecoded-dir /feature-cache/predecoded \ --out "$OUT" \ --n-steps {{workflow.parameters.n-steps}} \ --seq-len {{workflow.parameters.seq-len}} \ --n-backtests {{workflow.parameters.n-backtests}} \ --per-capacity {{workflow.parameters.per-capacity}} \ --seed {{workflow.parameters.seed}} \ --instrument-mode "{{workflow.parameters.instrument-mode}}" \ --fold-idx {{workflow.parameters.fold-idx}} \ --n-folds {{workflow.parameters.n-folds}} \ --n-eval-steps {{workflow.parameters.n-eval-steps}} \ --log-every {{workflow.parameters.log-every}} echo "=== Complete: $OUT ===" ls -lh "$OUT/" if [ -f "$OUT/alpha_rl_train_summary.json" ]; then cat "$OUT/alpha_rl_train_summary.json" fi