Files
foxhunt/infra/modules/kapsule/main.tf
jgrusewski a252119fd4 infra(kapsule): remove h100x2 + h100-sxm pools
Reasons:
- h100-sxm: Scaleway account quota is 0/0 for the SXM instance type
  (cp_servers_type_H100_SXM_2_80G). The pool was perpetually trying
  to maintain size=1 with a creation_error node, which JAMMED the
  cluster autoscaler. That blocked L40S scale-up entirely during
  today's alpha-perception cluster run.
- h100x2: more expensive than current workloads justify. Every
  training path (alpha perception, future PPO) fits on either the
  single-GPU H100 or L40S.

Removed:
- Two `scaleway_k8s_pool` resources from infra/modules/kapsule/main.tf
- Six variables (enable + type + max_size for each pool)
- Two outputs (pool_id for each)
- Corresponding inputs in infra/live/production/kapsule/terragrunt.hcl

The live cluster has the SXM pool stuck node manually deleted via
`scw k8s node delete` (this commit-session); the pool resource itself
will be destroyed on next `terragrunt apply`.

Post-cleanup pool inventory:
- platform (DEV1-L × 3)
- ci-training-h100  (H100-1-80G, max 1)  <- regular single-GPU
- ci-training-l40s  (L40S-1-48G, max 1)  <- primary training target
- ci-compile-cpu    (POP2-HC, max 4)
- ci-compile-cpu-hm (POP2-HM, max 1)

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-05-16 23:43:12 +02:00

159 lines
5.1 KiB
HCL
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
resource "scaleway_vpc_private_network" "foxhunt" {
name = "${var.cluster_name}-pn"
region = var.region
}
# VPC Public Gateway — provides NAT (masquerade) for private nodes
# and DHCP with default route propagation to prevent DNS deadlock.
# See incident_dns_deadlock.md for why this is critical.
resource "scaleway_vpc_public_gateway" "foxhunt" {
name = "${var.cluster_name}-gw"
type = "VPC-GW-S"
zone = "${var.region}-2"
bastion_enabled = true
bastion_port = 61000
}
resource "scaleway_vpc_public_gateway_dhcp" "foxhunt" {
subnet = "172.16.0.0/22"
push_default_route = true
push_dns_server = true
zone = "${var.region}-2"
}
resource "scaleway_vpc_gateway_network" "foxhunt" {
gateway_id = scaleway_vpc_public_gateway.foxhunt.id
private_network_id = scaleway_vpc_private_network.foxhunt.id
dhcp_id = scaleway_vpc_public_gateway_dhcp.foxhunt.id
enable_masquerade = true
zone = "${var.region}-2"
}
resource "scaleway_k8s_cluster" "foxhunt" {
name = var.cluster_name
version = var.k8s_version
cni = "cilium"
region = var.region
delete_additional_resources = true
private_network_id = scaleway_vpc_private_network.foxhunt.id
auto_upgrade {
enable = true
maintenance_window_start_hour = 4
maintenance_window_day = "sunday"
}
autoscaler_config {
disable_scale_down = false
scale_down_delay_after_add = "10m"
scale_down_unneeded_time = "10m"
estimator = "binpacking"
ignore_daemonsets_utilization = true
}
}
# Platform pool — runs everything: app services, databases, GitLab, monitoring
resource "scaleway_k8s_pool" "platform" {
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "platform"
node_type = var.platform_type
size = 1
min_size = 1
max_size = var.platform_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}
# L40S training pool (48GB VRAM, CUDA CC 89 — hyperopt + supervised training)
resource "scaleway_k8s_pool" "ci_training_l40s" {
count = var.enable_ci_training_l40s_pool ? 1 : 0
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "ci-training-l40s"
node_type = var.ci_training_l40s_type
size = 1
min_size = 0
max_size = var.ci_training_l40s_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}
# H100 training pool (80GB VRAM, CUDA CC 90 — RL hyperopt)
resource "scaleway_k8s_pool" "ci_training_h100" {
count = var.enable_ci_training_h100_pool ? 1 : 0
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "ci-training-h100"
node_type = var.ci_training_h100_type
size = 1
min_size = 0 # Scale to zero when idle. Cold start: ~15 min NVIDIA driver install.
max_size = var.ci_training_h100_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}
# NOTE: ci-training-h100x2 and ci-training-h100-sxm pools removed
# 2026-05-16 — Scaleway account doesn't have quota for the SXM variant
# (cp_servers_type_H100_SXM_2_80G 0/0), and the H100x2 variant is
# more expensive than our actual usage justifies. Keeping the regular
# single-GPU H100 pool (ci_training_h100) + the L40S pool covers
# every current training workload.
# CPU compile pool (POP2-HC — high CPU, no GPU, for cargo check/build)
resource "scaleway_k8s_pool" "ci_compile_cpu" {
count = var.enable_ci_compile_cpu_pool ? 1 : 0
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "ci-compile-cpu"
node_type = var.ci_compile_cpu_type
size = 1
min_size = 0
max_size = var.ci_compile_cpu_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}
# High-memory CPU compile pool (POP2-HM — same CPU count as ci-compile-cpu
# but 4× the RAM; reserved for rare full-rebuild precompute_features runs
# that can't fit on the standard 64GB nodes). min_size=0 + size=0 initial
# so the pool costs nothing when idle — autoscaler provisions one node
# only when a pod targets `nodeSelector: ci-compile-cpu-hm`.
resource "scaleway_k8s_pool" "ci_compile_cpu_hm" {
count = var.enable_ci_compile_cpu_hm_pool ? 1 : 0
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "ci-compile-cpu-hm"
node_type = var.ci_compile_cpu_hm_type
size = 0
min_size = 0
max_size = var.ci_compile_cpu_hm_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}