diff --git a/infra/k8s/argo/alpha-perception-template.yaml b/infra/k8s/argo/alpha-perception-template.yaml index 4ee38c9f9..09579b805 100644 --- a/infra/k8s/argo/alpha-perception-template.yaml +++ b/infra/k8s/argo/alpha-perception-template.yaml @@ -80,11 +80,20 @@ spec: templates: # ── DAG ── + # + # ensure-binary (CPU compile) and warmup-gpu (autoscaler trigger + # for the L40S pool) run in parallel. train only depends on + # ensure-binary — warmup-gpu is fire-and-forget; the autoscaler + # keeps the node warm for its scaledown grace window after the + # warmup pod exits, so train lands on a hot node without waiting + # for autoscaler provisioning. - name: pipeline dag: tasks: - name: ensure-binary template: ensure-binary + - name: warmup-gpu + template: warmup-gpu - name: train template: train dependencies: [ensure-binary] @@ -205,6 +214,47 @@ spec: done echo "$SHORT_SHA" > /tmp/sha + # ── warmup-gpu: trigger L40S autoscaler scale-up in parallel ── + # + # Schedules a tiny CPU-only pod on the gpu-pool's nodeSelector. + # Kubernetes sees an unschedulable pod (no node currently in the + # pool), autoscaler scales the pool from 0 → 1. The pod sleeps + # briefly, exits; the node enters "empty" state with the + # autoscaler's scaledown grace window (~10 min on Scaleway), so + # train can land on it without provisioning latency. + # + # No GPU resource request — that would compete with `train` for + # the single GPU and serialise the steps. The nodeSelector + + # nvidia.com/gpu toleration are enough to force placement on the + # right pool. + - name: warmup-gpu + nodeSelector: + k8s.scaleway.com/pool-name: "{{workflow.parameters.gpu-pool}}" + topology.kubernetes.io/zone: fr-par-2 + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + - key: node.cilium.io/agent-not-ready + operator: Exists + effect: NoSchedule + container: + image: alpine:3.20 + command: ["/bin/sh", "-c"] + resources: + requests: + cpu: "100m" + memory: 64Mi + limits: + cpu: "200m" + memory: 128Mi + args: + - | + echo "GPU warmup pod scheduled on $(hostname) — autoscaler scale-up triggered." + echo "Sleeping 30s to keep the node alive past provisioning; train will land here." + sleep 30 + echo "Warmup complete; node remains in autoscaler scaledown grace window." + # ── train: run alpha_train on L40S ── - name: train inputs: