Files
foxhunt/infra/k8s/gpu-overlays/ml-training-service-gpu.yaml
jgrusewski b31329931f fix(infra): remove MinIO TLS, fix sccache 0% cache hits, update pool selectors
- Remove all HTTPS/TLS from MinIO (plain HTTP for internal cluster traffic)
- Fix sccache 0% cache hit rate (rustls rejected self-signed MinIO cert)
- Remove hardcoded URLs from k8s_dispatcher.rs (S3_ENDPOINT, TRAINING_RUNTIME_IMAGE,
  CALLBACK_ENDPOINT now required env vars)
- Update GitLab registry S3 credentials to HTTP endpoint
- Fix PVC manifest (20Gi → 100Gi to match cluster)
- Fix nodeSelector: infra/foxhunt → platform (match actual node pool)
- Fix rclone trailing backslash causing chmod to be parsed as rclone args
- Remove minio-ca-cert ConfigMap references from all manifests
- Update trading-service GPU overlay to l40s pool

20 files changed, -118 lines net

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
2026-03-06 23:53:05 +01:00

170 lines
5.4 KiB
YAML

# GPU-enabled overlay for ml-training-service
# Apply manually: kubectl apply -f infra/k8s/gpu-overlays/ml-training-service-gpu.yaml
# Revert to CPU: kubectl apply -f infra/k8s/services/ml-training-service.yaml
#
# Binary fetched from MinIO at pod startup — works on any node pool.
apiVersion: apps/v1
kind: Deployment
metadata:
name: ml-training-service
namespace: foxhunt
labels:
app.kubernetes.io/name: ml-training-service
app.kubernetes.io/part-of: foxhunt
spec:
replicas: 1
strategy:
type: RollingUpdate
rollingUpdate:
maxSurge: 0
maxUnavailable: 1
selector:
matchLabels:
app.kubernetes.io/name: ml-training-service
template:
metadata:
annotations:
gitlab.com/prometheus_scrape: "true"
gitlab.com/prometheus_port: "9094"
gitlab.com/prometheus_path: "/metrics"
labels:
app.kubernetes.io/name: ml-training-service
app.kubernetes.io/part-of: foxhunt
spec:
serviceAccountName: ml-training-service
securityContext:
runAsNonRoot: true
runAsUser: 1000
runAsGroup: 1000
fsGroup: 1000
seccompProfile:
type: RuntimeDefault
nodeSelector:
k8s.scaleway.com/pool-name: gpu-inference
tolerations:
- key: nvidia.com/gpu
operator: Exists
effect: NoSchedule
imagePullSecrets:
- name: gitlab-registry
initContainers:
- name: fetch-binary
image: gitlab-registry.foxhunt.svc.cluster.local:5000/root/foxhunt/foxhunt-runtime:latest
command: ["/bin/sh", "-c"]
args:
- |
set -e
rclone copyto \
":s3:foxhunt-binaries/services/ml-training-service" \
"/binaries/ml-training-service" \
--s3-provider=Minio \
--s3-endpoint=http://minio.foxhunt.svc.cluster.local:9000 \
--s3-access-key-id="${MINIO_ACCESS_KEY}" \
--s3-secret-access-key="${MINIO_SECRET_KEY}" \
--s3-no-check-bucket
chmod +x /binaries/ml-training-service
echo "Fetched ml-training-service ($(stat -c%s /binaries/ml-training-service) bytes)"
env:
- name: MINIO_ACCESS_KEY
valueFrom:
secretKeyRef:
name: minio-credentials
key: access-key
- name: MINIO_SECRET_KEY
valueFrom:
secretKeyRef:
name: minio-credentials
key: secret-key
volumeMounts:
- name: binaries
mountPath: /binaries
resources:
requests:
cpu: 100m
memory: 64Mi
limits:
cpu: 500m
memory: 128Mi
containers:
- name: ml-training-service
image: gitlab-registry.foxhunt.svc.cluster.local:5000/root/foxhunt/foxhunt-runtime:latest
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: ["ALL"]
command: ["/binaries/ml-training-service", "serve"]
ports:
- containerPort: 50053
name: grpc
- containerPort: 9094
name: metrics
env:
- name: DATABASE_PASSWORD
valueFrom:
secretKeyRef:
name: db-credentials
key: password
- name: DATABASE_URL
value: "postgresql://foxhunt:$(DATABASE_PASSWORD)@postgres:5432/foxhunt"
- name: REDIS_URL
value: "redis://redis:6379"
- name: JWT_SECRET
valueFrom:
secretKeyRef:
name: jwt-secret
key: secret
- name: JWT_ISSUER
value: foxhunt-api
- name: JWT_AUDIENCE
value: foxhunt-services
- name: S3_ENDPOINT
value: "http://minio.foxhunt.svc.cluster.local:9000"
- name: S3_BUCKET
value: foxhunt-models
- name: ENABLE_GPU
value: "true"
- name: RUST_LOG
value: info
- name: OTEL_EXPORTER_OTLP_ENDPOINT
value: "http://tempo.foxhunt.svc.cluster.local:4317"
volumeMounts:
- name: binaries
mountPath: /binaries
readOnly: true
- name: tls-certs
mountPath: /app/certs/ml-training-service
readOnly: true
- name: tmp
mountPath: /tmp
readinessProbe:
tcpSocket:
port: 50053
initialDelaySeconds: 15
periodSeconds: 10
livenessProbe:
tcpSocket:
port: 50053
initialDelaySeconds: 30
periodSeconds: 15
failureThreshold: 5
resources:
requests:
nvidia.com/gpu: "1"
cpu: "1"
memory: 2Gi
limits:
nvidia.com/gpu: "1"
cpu: "4"
memory: 8Gi
volumes:
- name: binaries
emptyDir:
sizeLimit: 200Mi
- name: tls-certs
secret:
secretName: ml-training-tls
- name: tmp
emptyDir:
sizeLimit: 50Mi