diff --git a/infra/k8s/argo/alpha-perception-template.yaml b/infra/k8s/argo/alpha-perception-template.yaml index 09579b805..5c61bd52c 100644 --- a/infra/k8s/argo/alpha-perception-template.yaml +++ b/infra/k8s/argo/alpha-perception-template.yaml @@ -218,10 +218,11 @@ spec: # # Schedules a tiny CPU-only pod on the gpu-pool's nodeSelector. # Kubernetes sees an unschedulable pod (no node currently in the - # pool), autoscaler scales the pool from 0 → 1. The pod sleeps - # briefly, exits; the node enters "empty" state with the - # autoscaler's scaledown grace window (~10 min on Scaleway), so - # train can land on it without provisioning latency. + # pool), autoscaler scales the pool 0 → 1. The pod exits the + # instant it lands; the node enters scaledown-grace state for + # ~15 min (Scaleway Kapsule default), which spans the entire + # ensure-binary compile window so train lands on a hot node with + # zero provisioning latency. # # No GPU resource request — that would compete with `train` for # the single GPU and serialise the steps. The nodeSelector + @@ -243,17 +244,14 @@ spec: command: ["/bin/sh", "-c"] resources: requests: + cpu: "50m" + memory: 32Mi + limits: cpu: "100m" memory: 64Mi - limits: - cpu: "200m" - memory: 128Mi args: - | - echo "GPU warmup pod scheduled on $(hostname) — autoscaler scale-up triggered." - echo "Sleeping 30s to keep the node alive past provisioning; train will land here." - sleep 30 - echo "Warmup complete; node remains in autoscaler scaledown grace window." + echo "GPU warmup pod scheduled on $(hostname) — autoscaler scale-up triggered, node now in scaledown-grace window." # ── train: run alpha_train on L40S ── - name: train