Files
foxhunt/infra/k8s/argo/training-workflow-template.yaml
jgrusewski a7c22ca5aa fix: CI OOMKilled on submit-gpu-test + YAML indentation fixes
submit-gpu-test wait sidecar had 64Mi limit — OOMKilled when tracking
large child workflow status JSON. Bumped to 256Mi.

Fixed YAML indentation errors in compile-and-train and training-pipeline
templates (misaligned labels, duplicate component keys).

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-03-22 10:45:43 +01:00

488 lines
18 KiB
YAML

apiVersion: argoproj.io/v1alpha1
kind: WorkflowTemplate
metadata:
name: training-pipeline
namespace: foxhunt
labels:
app.kubernetes.io/name: training-pipeline
app.kubernetes.io/part-of: foxhunt
app.kubernetes.io/component: gpu-test
spec:
entrypoint: train-model
serviceAccountName: argo-workflow
# Label all workflow pods so network policies (MinIO, DNS) allow traffic
podMetadata:
labels:
app.kubernetes.io/part-of: foxhunt
app.kubernetes.io/component: training-workflow
securityContext:
runAsUser: 1000
runAsGroup: 1000
fsGroup: 1000
ttlStrategy:
secondsAfterCompletion: 3600
activeDeadlineSeconds: 21600
arguments:
parameters:
- name: model
- name: binary-tag
# Required: full commit SHA, e.g. "dev-6ab9e184abcd1234..."
# No default — caller must specify which compiled binaries to use.
- name: gpu-pool
value: ci-training-h100 # Pools: ci-training-h100, ci-training-l40s, ci-training-h100x2
- name: hyperopt-trials
value: "20"
- name: hyperopt-epochs
value: "8"
- name: train-epochs
value: "50"
- name: symbol
value: ES.FUT
- name: data-dir
value: /data/futures-baseline
- name: mbp10-data-dir
value: /data/futures-baseline-mbp10
- name: trades-data-dir
value: /data/futures-baseline-trades
- name: tx-cost-bps
value: "0.1"
- name: tick-size
value: "0.25"
- name: spread-ticks
value: "1.0"
- name: initial-capital
value: "35000"
- name: ensemble-top-k
value: "5"
- name: cuda-compute-cap
value: "90"
# Training data PVC — already populated by download jobs (OHLCV, MBP-10, trades).
# RWO: only one pod at a time, but steps run sequentially so no conflict.
volumes:
- name: training-data
persistentVolumeClaim:
claimName: training-data-pvc
readOnly: true
# Ephemeral shared storage for passing artifacts between DAG steps.
# Auto-deleted when workflow completes (all artifacts uploaded to MinIO).
volumeClaimTemplates:
- metadata:
name: workspace
spec:
accessModes: ["ReadWriteOnce"]
storageClassName: scw-bssd
resources:
requests:
storage: 5Gi
templates:
- name: train-model
dag:
tasks:
- name: fetch-binary
template: fetch-binary
# gpu-warmup disabled — re-enable when GPU nodes stay warm
# - name: gpu-warmup
# template: gpu-warmup
- name: hyperopt
template: hyperopt
dependencies: [fetch-binary]
- name: train-best
template: train-best
dependencies: [hyperopt]
- name: evaluate
template: evaluate
dependencies: [train-best]
- name: upload-results
template: upload-results
dependencies: [evaluate]
# ── gpu-warmup: disabled — re-enable when GPU nodes stay warm ──
# - name: gpu-warmup
# nodeSelector:
# k8s.scaleway.com/pool-name: "{{workflow.parameters.gpu-pool}}"
# tolerations:
# - key: nvidia.com/gpu
# operator: Exists
# effect: NoSchedule
# - key: node.cilium.io/agent-not-ready
# operator: Exists
# effect: NoSchedule
# container:
# image: busybox:1.37
# command: ["/bin/sh", "-c"]
# args:
# - |
# echo "GPU warmup: triggering node autoscale..."
# echo "GPU node scheduled, exiting to free resources"
# resources:
# requests:
# nvidia.com/gpu: "1"
# cpu: 100m
# memory: 64Mi
# limits:
# nvidia.com/gpu: "1"
# cpu: 200m
# memory: 128Mi
- name: fetch-binary
nodeSelector:
k8s.scaleway.com/pool-name: platform
container:
image: curlimages/curl:8.12.1
command: ["/bin/sh", "-c"]
args:
- |
set -e
mkdir -p /workspace/bin
GITLAB_API="http://gitlab-webservice-default.foxhunt.svc.cluster.local:8181/api/v4"
TAG="{{workflow.parameters.binary-tag}}"
if [ -z "$TAG" ]; then
echo "ERROR: binary-tag parameter is required (full commit SHA, e.g. dev-6ab9e184...)"
exit 1
fi
echo "Fetching training binaries from GitLab package: foxhunt-training/${TAG}"
BINARIES="hyperopt_baseline_rl hyperopt_baseline_supervised train_baseline_rl train_baseline_supervised evaluate_baseline evaluate_supervised training_uploader"
for bin in $BINARIES; do
echo " Downloading ${bin}..."
curl -fSL -o "/workspace/bin/${bin}" \
-H "PRIVATE-TOKEN: ${GITLAB_PAT}" \
"${GITLAB_API}/projects/1/packages/generic/foxhunt-training/${TAG}/${bin}" || {
echo "WARN: ${bin} not found in package foxhunt-training/${TAG}, skipping"
continue
}
done
chmod +x /workspace/bin/*
echo "Fetched binaries:"
ls -lh /workspace/bin/
env:
- name: GITLAB_PAT
valueFrom:
secretKeyRef:
name: gitlab-pat
key: token
volumeMounts:
- name: workspace
mountPath: /workspace
resources:
requests:
cpu: 100m
memory: 128Mi
limits:
cpu: 500m
memory: 256Mi
- name: hyperopt
nodeSelector:
k8s.scaleway.com/pool-name: "{{workflow.parameters.gpu-pool}}"
tolerations:
- key: nvidia.com/gpu
operator: Exists
effect: NoSchedule
- key: node.cilium.io/agent-not-ready
operator: Exists
effect: NoSchedule
container:
image: gitlab-registry.foxhunt.svc.cluster.local:5000/root/foxhunt/ci-builder:latest
imagePullPolicy: IfNotPresent
command: ["/bin/sh", "-c"]
args:
- |
set -e
export PATH="/workspace/bin:$PATH"
export LD_LIBRARY_PATH=$(echo "$LD_LIBRARY_PATH" | tr ':' '\n' | grep -v stubs | tr '\n' ':' | sed 's/:$//')
nvidia-smi
# Training data available via training-data-pvc mounted at /data
echo "=== Training data (PVC) ==="
echo "OHLCV: $(ls {{workflow.parameters.data-dir}}/ES.FUT/*.dbn.zst 2>/dev/null | wc -l) files"
echo "MBP-10: $(ls {{workflow.parameters.mbp10-data-dir}}/ES.FUT/*.dbn.zst 2>/dev/null | wc -l) files"
echo "Trades: $(ls {{workflow.parameters.trades-data-dir}}/ES.FUT/*.dbn.zst 2>/dev/null | wc -l) files"
mkdir -p /workspace/output/hyperopt
MODEL="{{workflow.parameters.model}}"
case "$MODEL" in
dqn|ppo) BINARY=hyperopt_baseline_rl ;;
*) BINARY=hyperopt_baseline_supervised ;;
esac
TRIALS={{workflow.parameters.hyperopt-trials}}
N_INITIAL=5
if [ "$TRIALS" -le "$N_INITIAL" ]; then
N_INITIAL=$((TRIALS > 1 ? TRIALS - 1 : 1))
fi
echo "Running $BINARY --model $MODEL ($TRIALS trials, $N_INITIAL initial x {{workflow.parameters.hyperopt-epochs}} epochs)"
# Phase 1 (fast): fix architecture, search learning dynamics
$BINARY \
--model "$MODEL" \
--phase fast \
--trials "$TRIALS" \
--n-initial "$N_INITIAL" \
--epochs {{workflow.parameters.hyperopt-epochs}} \
--parallel 0 \
--symbol {{workflow.parameters.symbol}} \
--tx-cost-bps {{workflow.parameters.tx-cost-bps}} \
--tick-size {{workflow.parameters.tick-size}} \
--spread-ticks {{workflow.parameters.spread-ticks}} \
--initial-capital {{workflow.parameters.initial-capital}} \
--data-dir {{workflow.parameters.data-dir}} \
--mbp10-data-dir {{workflow.parameters.mbp10-data-dir}} \
--trades-data-dir {{workflow.parameters.trades-data-dir}} \
--base-dir /workspace/output/hyperopt \
--output /workspace/output/${MODEL}_phase1_results.json
echo "=== Phase 1 complete ==="
cat /workspace/output/${MODEL}_phase1_results.json 2>/dev/null || echo "No Phase 1 results"
# Phase 2 (full): fix dynamics from Phase 1, search architecture
PHASE2_TRIALS=$((TRIALS / 2))
[ "$PHASE2_TRIALS" -lt 5 ] && PHASE2_TRIALS=5
PHASE2_EPOCHS=$(({{workflow.parameters.hyperopt-epochs}} * 2))
$BINARY \
--model "$MODEL" \
--phase full \
--hyperopt-params /workspace/output/${MODEL}_phase1_results.json \
--trials $PHASE2_TRIALS \
--epochs $PHASE2_EPOCHS \
--parallel 0 \
--symbol {{workflow.parameters.symbol}} \
--tx-cost-bps {{workflow.parameters.tx-cost-bps}} \
--tick-size {{workflow.parameters.tick-size}} \
--spread-ticks {{workflow.parameters.spread-ticks}} \
--initial-capital {{workflow.parameters.initial-capital}} \
--data-dir {{workflow.parameters.data-dir}} \
--mbp10-data-dir {{workflow.parameters.mbp10-data-dir}} \
--trades-data-dir {{workflow.parameters.trades-data-dir}} \
--base-dir /workspace/output/hyperopt \
--output /workspace/output/${MODEL}_hyperopt_results.json
echo "=== Two-phase hyperopt complete ==="
cat /workspace/output/${MODEL}_hyperopt_results.json 2>/dev/null || echo "No results file"
env:
- name: RUST_LOG
value: info
- name: SQLX_OFFLINE
value: "true"
- name: OTEL_EXPORTER_OTLP_ENDPOINT
value: "http://tempo.foxhunt.svc.cluster.local:4317"
- name: CUBLAS_WORKSPACE_CONFIG
value: ":4096:8"
volumeMounts:
- name: workspace
mountPath: /workspace
- name: training-data
mountPath: /data
readOnly: true
resources:
requests:
nvidia.com/gpu: "1"
cpu: "7"
memory: 16Gi
limits:
nvidia.com/gpu: "1"
cpu: "8"
memory: 48Gi
- name: train-best
nodeSelector:
k8s.scaleway.com/pool-name: "{{workflow.parameters.gpu-pool}}"
tolerations:
- key: nvidia.com/gpu
operator: Exists
effect: NoSchedule
- key: node.cilium.io/agent-not-ready
operator: Exists
effect: NoSchedule
container:
image: gitlab-registry.foxhunt.svc.cluster.local:5000/root/foxhunt/ci-builder:latest
imagePullPolicy: IfNotPresent
command: ["/bin/sh", "-c"]
args:
- |
set -e
export PATH="/workspace/bin:$PATH"
export LD_LIBRARY_PATH=$(echo "$LD_LIBRARY_PATH" | tr ':' '\n' | grep -v stubs | tr '\n' ':' | sed 's/:$//')
nvidia-smi
# Training data available via training-data-pvc mounted at /data
MODEL="{{workflow.parameters.model}}"
case "$MODEL" in
dqn|ppo) BINARY=train_baseline_rl ;;
*) BINARY=train_baseline_supervised ;;
esac
HYPEROPT_FILE="/workspace/output/${MODEL}_hyperopt_results.json"
HYPEROPT_FLAG=""
if [ -f "$HYPEROPT_FILE" ]; then
HYPEROPT_FLAG="--hyperopt-params $HYPEROPT_FILE"
fi
echo "Training $MODEL with best params ({{workflow.parameters.train-epochs}} epochs)"
$BINARY \
--model "$MODEL" \
--symbol {{workflow.parameters.symbol}} \
--tx-cost-bps {{workflow.parameters.tx-cost-bps}} \
--tick-size {{workflow.parameters.tick-size}} \
--spread-ticks {{workflow.parameters.spread-ticks}} \
--data-dir {{workflow.parameters.data-dir}} \
--mbp10-data-dir {{workflow.parameters.mbp10-data-dir}} \
--trades-data-dir {{workflow.parameters.trades-data-dir}} \
--output-dir /workspace/output \
--max-steps-per-epoch {{workflow.parameters.train-epochs}} \
$HYPEROPT_FLAG \
--ensemble-top-k {{workflow.parameters.ensemble-top-k}}
echo "=== Training complete ==="
ls -lh /workspace/output/
env:
- name: RUST_LOG
value: info
- name: SQLX_OFFLINE
value: "true"
- name: OTEL_EXPORTER_OTLP_ENDPOINT
value: "http://tempo.foxhunt.svc.cluster.local:4317"
- name: CUBLAS_WORKSPACE_CONFIG
value: ":4096:8"
volumeMounts:
- name: workspace
mountPath: /workspace
- name: training-data
mountPath: /data
readOnly: true
resources:
requests:
nvidia.com/gpu: "1"
cpu: "7"
memory: 16Gi
limits:
nvidia.com/gpu: "1"
cpu: "8"
memory: 48Gi
- name: evaluate
nodeSelector:
k8s.scaleway.com/pool-name: "{{workflow.parameters.gpu-pool}}"
tolerations:
- key: nvidia.com/gpu
operator: Exists
effect: NoSchedule
- key: node.cilium.io/agent-not-ready
operator: Exists
effect: NoSchedule
container:
image: gitlab-registry.foxhunt.svc.cluster.local:5000/root/foxhunt/ci-builder:latest
imagePullPolicy: IfNotPresent
command: ["/bin/sh", "-c"]
args:
- |
set -e
export PATH="/workspace/bin:$PATH"
export LD_LIBRARY_PATH=$(echo "$LD_LIBRARY_PATH" | tr ':' '\n' | grep -v stubs | tr '\n' ':' | sed 's/:$//')
# Training data available via training-data-pvc mounted at /data
MODEL="{{workflow.parameters.model}}"
HYPEROPT_FILE="/workspace/output/${MODEL}_hyperopt_results.json"
HYPEROPT_FLAG=""
if [ -f "$HYPEROPT_FILE" ]; then
HYPEROPT_FLAG="--hyperopt-params $HYPEROPT_FILE"
fi
# Evaluate all ensemble members (evaluator auto-discovers *_ensemble_*.safetensors)
echo "Evaluating $MODEL"
evaluate_baseline \
--model "$MODEL" \
--models-dir /workspace/output \
--data-dir {{workflow.parameters.data-dir}} \
--output /workspace/output/${MODEL}_eval_report.json \
--symbol {{workflow.parameters.symbol}} \
--tx-cost-bps {{workflow.parameters.tx-cost-bps}} \
--tick-size {{workflow.parameters.tick-size}} \
--spread-ticks {{workflow.parameters.spread-ticks}} \
$HYPEROPT_FLAG \
|| true
echo "=== Eval report ==="
cat /workspace/output/${MODEL}_eval_report.json 2>/dev/null || echo "No eval report"
env:
- name: RUST_LOG
value: info
- name: SQLX_OFFLINE
value: "true"
volumeMounts:
- name: workspace
mountPath: /workspace
- name: training-data
mountPath: /data
readOnly: true
resources:
requests:
nvidia.com/gpu: "1"
cpu: "4"
memory: 16Gi
limits:
nvidia.com/gpu: "1"
cpu: "8"
memory: 32Gi
- name: upload-results
nodeSelector:
k8s.scaleway.com/pool-name: platform
container:
image: curlimages/curl:8.12.1
command: ["/bin/sh", "-c"]
args:
- |
set -e
MODEL="{{workflow.parameters.model}}"
SYMBOL="{{workflow.parameters.symbol}}"
TIMESTAMP=$(date +%Y%m%d-%H%M%S)
GITLAB_API="http://gitlab-webservice-default.foxhunt.svc.cluster.local:8181/api/v4"
PKG_NAME="foxhunt-training-results"
PKG_VERSION="${MODEL}-${SYMBOL}-${TIMESTAMP}"
echo "Uploading training artifacts to GitLab packages: ${PKG_NAME}/${PKG_VERSION}"
echo " checkpoints (*.safetensors), norm stats, hyperopt results, eval reports"
UPLOADED=0
find /workspace/output -type f | while read -r file; do
REL_PATH="${file#/workspace/output/}"
# Replace / with -- for flat package file naming
SAFE_NAME=$(echo "$REL_PATH" | tr '/' '--')
echo " Uploading ${REL_PATH} as ${SAFE_NAME}..."
curl -f --upload-file "$file" \
-H "PRIVATE-TOKEN: ${GITLAB_PAT}" \
"${GITLAB_API}/projects/1/packages/generic/${PKG_NAME}/${PKG_VERSION}/${SAFE_NAME}" && \
UPLOADED=$((UPLOADED + 1)) || \
echo " WARN: Failed to upload ${REL_PATH}"
done
echo "=== Upload complete (${UPLOADED} files) ==="
echo "Package: ${PKG_NAME}/${PKG_VERSION}"
env:
- name: GITLAB_PAT
valueFrom:
secretKeyRef:
name: gitlab-pat
key: token
volumeMounts:
- name: workspace
mountPath: /workspace
resources:
requests:
cpu: 100m
memory: 128Mi
limits:
cpu: 500m
memory: 256Mi