From 208f6f23927dcfec75bacd9a2afae7df4ee90dc4 Mon Sep 17 00:00:00 2001 From: jgrusewski Date: Fri, 27 Feb 2026 10:20:41 +0100 Subject: [PATCH] =?UTF-8?q?refactor(infra):=20rename=20pools=20=E2=80=94?= =?UTF-8?q?=20always-on=E2=86=92services,=20ci-build=E2=86=92ci-training?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rename Terraform resources and Scaleway pool names for clarity: - always_on → services (runs postgres, redis, microservices) - ci_build → ci_training (L40S for hyperopt + training jobs) Includes: - moved{} blocks to prevent Terraform state destroy+recreate - Output renames + new outputs for ci-compile and ci-training pools - node_selector_overwrite_allowed in runner config (enables per-job KUBERNETES_NODE_SELECTOR_* overrides for L4/L40S routing) MIGRATION (after terragrunt apply): 1. terragrunt apply — recreates pools with new names 2. Update K8s manifests: always-on→services in nodeSelectors (postgres, redis, questdb, tailscale, idle-reaper, postgres-init) 3. Update .gitlab-ci.yml: ci-build→ci-training in training jobs 4. Update runner-values.yaml: default pool ci-build→ci-compile 5. helm upgrade gitlab-runner + kubectl apply changed manifests Co-Authored-By: Claude Opus 4.6 --- infra/k8s/gitlab/runner-values.yaml | 8 +++-- infra/live/production/kapsule/terragrunt.hcl | 10 +++---- infra/modules/kapsule/main.tf | 31 ++++++++++++++------ infra/modules/kapsule/outputs.tf | 17 ++++++++--- infra/modules/kapsule/variables.tf | 10 +++---- 5 files changed, 51 insertions(+), 25 deletions(-) diff --git a/infra/k8s/gitlab/runner-values.yaml b/infra/k8s/gitlab/runner-values.yaml index 4c2555a20..7d0625971 100644 --- a/infra/k8s/gitlab/runner-values.yaml +++ b/infra/k8s/gitlab/runner-values.yaml @@ -1,6 +1,7 @@ -# GitLab Runner — Kubernetes executor targeting ci-build pool (L40S) +# GitLab Runner — Kubernetes executor # Runner manager pod lives on gitlab node pool -# Build pods spawn on ci-build pool (L40S-1-48G, autoscales 0-2) +# Default: build pods spawn on ci-build pool (L40S) +# After pool migration: default→ci-compile (L4), training overrides→ci-training (L40S) gitlabUrl: http://gitlab-webservice-default.foxhunt.svc.cluster.local:8181 # runnerToken set via --set at install time @@ -26,6 +27,9 @@ runners: namespace = "foxhunt" image = "rust:1.89-slim" privileged = false + # Allow CI jobs to override node_selector via KUBERNETES_NODE_SELECTOR_* vars + # This enables routing compile→ci-compile (L4) and training→ci-training (L40S) + node_selector_overwrite_allowed = ".*" # L40S scale-to-zero needs ~3-5 min to provision; default 180s times out poll_timeout = 600 # Resource limits for build pods (L40S node: ~24 vCPU, ~96GB) diff --git a/infra/live/production/kapsule/terragrunt.hcl b/infra/live/production/kapsule/terragrunt.hcl index 7a23606bf..003837fdd 100644 --- a/infra/live/production/kapsule/terragrunt.hcl +++ b/infra/live/production/kapsule/terragrunt.hcl @@ -9,8 +9,8 @@ terraform { inputs = { cluster_name = "foxhunt" k8s_version = "1.34" - always_on_type = "DEV1-M" - always_on_max_size = 2 + services_type = "DEV1-M" + services_max_size = 2 # GitLab CE pool (dedicated for GitLab + Prometheus + Runner manager) enable_gitlab_pool = true @@ -22,9 +22,9 @@ inputs = { ci_compile_max_size = 1 # CI training pool (L40S for hyperopt + training — 48GB VRAM, CUDA CC 89) - enable_ci_build_pool = true - ci_build_type = "L40S-1-48G" - ci_build_max_size = 1 + enable_ci_training_pool = true + ci_training_type = "L40S-1-48G" + ci_training_max_size = 1 # Dev pool (GP1-L for DevPod remote development, autoscales to zero) enable_dev_pool = true diff --git a/infra/modules/kapsule/main.tf b/infra/modules/kapsule/main.tf index 1b980fcea..17f87e473 100644 --- a/infra/modules/kapsule/main.tf +++ b/infra/modules/kapsule/main.tf @@ -1,3 +1,16 @@ +# Terraform state renames — prevents destroy+recreate of state entries. +# Pool name attribute changes (always-on→services, ci-build→ci-training) +# will still force replacement since Scaleway pool names are immutable. +moved { + from = scaleway_k8s_pool.always_on + to = scaleway_k8s_pool.services +} + +moved { + from = scaleway_k8s_pool.ci_build + to = scaleway_k8s_pool.ci_training +} + resource "scaleway_vpc_private_network" "foxhunt" { name = "${var.cluster_name}-pn" region = var.region @@ -26,13 +39,13 @@ resource "scaleway_k8s_cluster" "foxhunt" { } } -resource "scaleway_k8s_pool" "always_on" { +resource "scaleway_k8s_pool" "services" { cluster_id = scaleway_k8s_cluster.foxhunt.id - name = "always-on" - node_type = var.always_on_type + name = "services" + node_type = var.services_type size = 1 min_size = 1 - max_size = var.always_on_max_size + max_size = var.services_max_size autoscaling = true autohealing = true public_ip_disabled = var.public_ip_disabled @@ -78,14 +91,14 @@ resource "scaleway_k8s_pool" "ci_compile" { } # CI training pool (L40S — 48GB VRAM, for hyperopt + training jobs) -resource "scaleway_k8s_pool" "ci_build" { - count = var.enable_ci_build_pool ? 1 : 0 +resource "scaleway_k8s_pool" "ci_training" { + count = var.enable_ci_training_pool ? 1 : 0 cluster_id = scaleway_k8s_cluster.foxhunt.id - name = "ci-build" - node_type = var.ci_build_type + name = "ci-training" + node_type = var.ci_training_type size = 1 min_size = 0 - max_size = var.ci_build_max_size + max_size = var.ci_training_max_size autoscaling = true autohealing = true public_ip_disabled = var.public_ip_disabled diff --git a/infra/modules/kapsule/outputs.tf b/infra/modules/kapsule/outputs.tf index ad50f3b41..b0a6582b6 100644 --- a/infra/modules/kapsule/outputs.tf +++ b/infra/modules/kapsule/outputs.tf @@ -14,17 +14,26 @@ output "apiserver_url" { value = scaleway_k8s_cluster.foxhunt.apiserver_url } -output "always_on_pool_id" { - description = "ID of the always-on node pool" - value = scaleway_k8s_pool.always_on.id +output "services_pool_id" { + description = "ID of the services node pool" + value = scaleway_k8s_pool.services.id } - output "gitlab_pool_id" { description = "ID of the GitLab node pool" value = var.enable_gitlab_pool ? scaleway_k8s_pool.gitlab[0].id : "" } +output "ci_compile_pool_id" { + description = "ID of the CI compile node pool (L4)" + value = var.enable_ci_compile_pool ? scaleway_k8s_pool.ci_compile[0].id : "" +} + +output "ci_training_pool_id" { + description = "ID of the CI training node pool (L40S)" + value = var.enable_ci_training_pool ? scaleway_k8s_pool.ci_training[0].id : "" +} + output "private_network_id" { description = "ID of the VPC private network the cluster is attached to" value = scaleway_vpc_private_network.foxhunt.id diff --git a/infra/modules/kapsule/variables.tf b/infra/modules/kapsule/variables.tf index e9bed2f88..a689f0465 100644 --- a/infra/modules/kapsule/variables.tf +++ b/infra/modules/kapsule/variables.tf @@ -15,13 +15,13 @@ variable "k8s_version" { default = "1.34" } -variable "always_on_type" { +variable "services_type" { description = "Instance type for the always-on node pool" type = string default = "DEV1-M" } -variable "always_on_max_size" { +variable "services_max_size" { description = "Maximum number of nodes in the always-on pool (autoscaling)" type = number default = 2 @@ -57,19 +57,19 @@ variable "ci_compile_max_size" { default = 1 } -variable "enable_ci_build_pool" { +variable "enable_ci_training_pool" { description = "Create L40S GPU node pool for training + hyperopt" type = bool default = false } -variable "ci_build_type" { +variable "ci_training_type" { description = "Instance type for the CI training pool (L40S for 48GB VRAM training)" type = string default = "L40S-1-48G" } -variable "ci_build_max_size" { +variable "ci_training_max_size" { description = "Maximum number of nodes in the CI training pool" type = number default = 1