diff --git a/.gitlab-ci.yml b/.gitlab-ci.yml index f25d4687c..a8976a20f 100644 --- a/.gitlab-ci.yml +++ b/.gitlab-ci.yml @@ -1,5 +1,5 @@ # GitLab CI/CD — Foxhunt -# Runs on ci-build pool (GP1-M, scale-to-zero) +# Runs on gpu-training pool (H100-1-80G: 24 vCPU, 240GB, scale-to-zero) # CI builder image: CUDA 12.4 + Rust 1.89 + protoc + sccache (hosted on Scaleway CR) # Service images: pushed to internal GitLab registry (Kaniko) @@ -19,8 +19,6 @@ variables: AWS_ACCESS_KEY_ID: $SCW_ACCESS_KEY AWS_SECRET_ACCESS_KEY: $SCW_SECRET_KEY RUSTC_WRAPPER: /usr/local/bin/sccache - # candle-kernels needs this to skip nvidia-smi on GPU-less CI nodes - CUDA_COMPUTE_CAP: "86" # CI builder image on Scaleway Container Registry (reachable by Kapsule nodes) CI_BUILDER_IMAGE: rg.fr-par.scw.cloud/foxhunt-ci/ci-builder:latest # Internal GitLab registry (HTTP, reachable from pods only — for service images) @@ -81,24 +79,8 @@ test: stage: test needs: [check] script: - # Stub libcuda.so.1 lets CUDA-linked binaries load on GPU-less nodes - - ln -sf /usr/local/cuda/lib64/stubs/libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1 || true - - export LD_LIBRARY_PATH="/usr/local/cuda/lib64/stubs:${LD_LIBRARY_PATH}" - # Run all workspace tests EXCEPT ml crate (52 CUDA-dependent tests need real GPU) - - cargo test --workspace --exclude ml --lib -- --skip model_loader::tests - -# GPU test job — runs ml crate tests on real CUDA hardware (gpu-inference pool, L4) -test-gpu: - extends: .rust-base - stage: test - needs: [check] - tags: - - kapsule - - gpu - when: manual - allow_failure: true - script: - - cargo test -p ml --lib + # All tests run on H100 node — real CUDA available, no stubs needed + - cargo test --workspace --lib -- --skip model_loader::tests # -------------------------------------------------------------------------- # Stage 3: Build + push service images (main only, Kaniko → GitLab registry) diff --git a/infra/k8s/gitlab/runner-values.yaml b/infra/k8s/gitlab/runner-values.yaml index c288b5bb9..77ca9d4d5 100644 --- a/infra/k8s/gitlab/runner-values.yaml +++ b/infra/k8s/gitlab/runner-values.yaml @@ -1,6 +1,6 @@ -# GitLab Runner — Kubernetes executor targeting ci-build pool +# GitLab Runner — Kubernetes executor targeting gpu-training pool # Runner manager pod lives on gitlab node pool -# Build pods spawn on ci-build pool (scale-to-zero) +# Build pods spawn on gpu-training pool (H100, scale-to-zero — fast builds + GPU tests) gitlabUrl: http://gitlab-webservice-default.foxhunt.svc.cluster.local:8181 # runnerToken set via --set at install time @@ -26,20 +26,19 @@ runners: image = "rust:1.89-slim" privileged = false - # Build pods on ci-build pool + # Build pods on gpu-training pool (H100: 24 vCPU, 240GB, CUDA) [runners.kubernetes.node_selector] - "k8s.scaleway.com/pool-name" = "ci-build" + "k8s.scaleway.com/pool-name" = "gpu-training" [[runners.kubernetes.node_tolerations]] - key = "ci-build" - operator = "Equal" - value = "true" + key = "nvidia.com/gpu" + operator = "Exists" effect = "NoSchedule" - # Resource limits for build pods - cpu_request = "2000m" - cpu_limit = "6000m" - memory_request = "4Gi" - memory_limit = "12Gi" + # Resource limits for build pods (H100 node: 24 vCPU, 240GB) + cpu_request = "4000m" + cpu_limit = "16000m" + memory_request = "16Gi" + memory_limit = "64Gi" # Helper container helper_cpu_request = "100m" @@ -55,8 +54,8 @@ runners: "app.kubernetes.io/part-of" = "foxhunt-ci" - # Runner tags for job matching - tags: "kapsule,rust,docker" + # Runner tags for job matching (gpu tag enables test-gpu job) + tags: "kapsule,rust,docker,gpu" # Concurrency concurrent: 4 diff --git a/infra/live/production/kapsule/terragrunt.hcl b/infra/live/production/kapsule/terragrunt.hcl index c389b2e74..7baa95a7b 100644 --- a/infra/live/production/kapsule/terragrunt.hcl +++ b/infra/live/production/kapsule/terragrunt.hcl @@ -25,8 +25,5 @@ inputs = { enable_gitlab_pool = true gitlab_type = "GP1-XS" - # CI build pool (ephemeral runner pods, scale-to-zero) - enable_ci_build_pool = true - ci_build_type = "GP1-M" - ci_build_max_size = 2 + # CI builds run on gpu-training pool (H100: 24 vCPU, 240GB, fast builds + GPU tests) } diff --git a/infra/modules/kapsule/main.tf b/infra/modules/kapsule/main.tf index e1849af30..171dcdda7 100644 --- a/infra/modules/kapsule/main.tf +++ b/infra/modules/kapsule/main.tf @@ -38,22 +38,6 @@ resource "scaleway_k8s_pool" "always_on" { region = var.region } -resource "scaleway_k8s_pool" "ci" { - cluster_id = scaleway_k8s_cluster.foxhunt.id - name = "ci" - node_type = var.ci_type - size = 1 - min_size = 0 - max_size = var.ci_max_size - autoscaling = true - autohealing = true - region = var.region - - lifecycle { - ignore_changes = [size] - } -} - # GPU pool for ML training (H100 — large VRAM for ensemble training) resource "scaleway_k8s_pool" "gpu_training" { count = var.enable_gpu_training_pool ? 1 : 0 @@ -105,20 +89,3 @@ resource "scaleway_k8s_pool" "gitlab" { region = var.region } -# CI build pod pool (scale-to-zero for ephemeral GitLab Runner build pods) -resource "scaleway_k8s_pool" "ci_build" { - count = var.enable_ci_build_pool ? 1 : 0 - cluster_id = scaleway_k8s_cluster.foxhunt.id - name = "ci-build" - node_type = var.ci_build_type - size = 0 - min_size = 0 - max_size = var.ci_build_max_size - autoscaling = true - autohealing = true - region = var.region - - lifecycle { - ignore_changes = [size] - } -} diff --git a/infra/modules/kapsule/outputs.tf b/infra/modules/kapsule/outputs.tf index 312da09e3..b1226759d 100644 --- a/infra/modules/kapsule/outputs.tf +++ b/infra/modules/kapsule/outputs.tf @@ -19,11 +19,6 @@ output "always_on_pool_id" { value = scaleway_k8s_pool.always_on.id } -output "ci_pool_id" { - description = "ID of the CI node pool" - value = scaleway_k8s_pool.ci.id -} - output "gpu_training_pool_id" { description = "ID of the GPU training node pool" value = var.enable_gpu_training_pool ? scaleway_k8s_pool.gpu_training[0].id : "" @@ -39,7 +34,3 @@ output "gitlab_pool_id" { value = var.enable_gitlab_pool ? scaleway_k8s_pool.gitlab[0].id : "" } -output "ci_build_pool_id" { - description = "ID of the CI build node pool" - value = var.enable_ci_build_pool ? scaleway_k8s_pool.ci_build[0].id : "" -} diff --git a/infra/modules/kapsule/variables.tf b/infra/modules/kapsule/variables.tf index 88ec23e6b..c4fd9e2cb 100644 --- a/infra/modules/kapsule/variables.tf +++ b/infra/modules/kapsule/variables.tf @@ -21,18 +21,6 @@ variable "always_on_type" { default = "DEV1-M" } -variable "ci_type" { - description = "Instance type for the CI node pool" - type = string - default = "GP1-XS" -} - -variable "ci_max_size" { - description = "Maximum number of nodes in the CI pool" - type = number - default = 1 -} - variable "enable_gpu_training_pool" { description = "Create GPU node pool for ML training (H100)" type = bool @@ -81,20 +69,3 @@ variable "gitlab_type" { default = "DEV1-L" } -variable "enable_ci_build_pool" { - description = "Create dedicated node pool for CI build pods" - type = bool - default = false -} - -variable "ci_build_type" { - description = "Instance type for the CI build node pool" - type = string - default = "GP1-XS" -} - -variable "ci_build_max_size" { - description = "Maximum number of nodes in the CI build pool" - type = number - default = 2 -} diff --git a/infra/scripts/harden.sh b/infra/scripts/harden.sh index 2e038a5f7..eabbd1e24 100755 --- a/infra/scripts/harden.sh +++ b/infra/scripts/harden.sh @@ -97,7 +97,7 @@ done # --- 6. Scale CI/GPU pools to 0 if idle --------------------------------- echo "" echo "--- Pool Scale-Down ---" -for pool_name in ci gpu; do +for pool_name in gpu-training gpu-inference; do POOL_ID=$(scw k8s pool list "cluster-id=${CLUSTER_ID}" "region=${SCW_REGION}" -o json 2>/dev/null \ | python3 -c " import sys, json