Files
foxhunt/infra/modules/kapsule/main.tf
jgrusewski 70fca46227 revert: H100 pool back to min_size=0 — too expensive to keep warm
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-17 09:13:46 +02:00

167 lines
5.1 KiB
HCL

resource "scaleway_vpc_private_network" "foxhunt" {
name = "${var.cluster_name}-pn"
region = var.region
}
# VPC Public Gateway — provides NAT (masquerade) for private nodes
# and DHCP with default route propagation to prevent DNS deadlock.
# See incident_dns_deadlock.md for why this is critical.
resource "scaleway_vpc_public_gateway" "foxhunt" {
name = "${var.cluster_name}-gw"
type = "VPC-GW-S"
zone = "${var.region}-2"
bastion_enabled = true
bastion_port = 61000
}
resource "scaleway_vpc_public_gateway_dhcp" "foxhunt" {
subnet = "172.16.0.0/22"
push_default_route = true
push_dns_server = true
zone = "${var.region}-2"
}
resource "scaleway_vpc_gateway_network" "foxhunt" {
gateway_id = scaleway_vpc_public_gateway.foxhunt.id
private_network_id = scaleway_vpc_private_network.foxhunt.id
dhcp_id = scaleway_vpc_public_gateway_dhcp.foxhunt.id
enable_masquerade = true
zone = "${var.region}-2"
}
resource "scaleway_k8s_cluster" "foxhunt" {
name = var.cluster_name
version = var.k8s_version
cni = "cilium"
region = var.region
delete_additional_resources = true
private_network_id = scaleway_vpc_private_network.foxhunt.id
auto_upgrade {
enable = true
maintenance_window_start_hour = 4
maintenance_window_day = "sunday"
}
autoscaler_config {
disable_scale_down = false
scale_down_delay_after_add = "10m"
scale_down_unneeded_time = "10m"
estimator = "binpacking"
ignore_daemonsets_utilization = true
}
}
# Platform pool — runs everything: app services, databases, GitLab, monitoring
resource "scaleway_k8s_pool" "platform" {
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "platform"
node_type = var.platform_type
size = 1
min_size = 1
max_size = var.platform_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}
# L40S training pool (48GB VRAM, CUDA CC 89 — hyperopt + supervised training)
resource "scaleway_k8s_pool" "ci_training_l40s" {
count = var.enable_ci_training_l40s_pool ? 1 : 0
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "ci-training-l40s"
node_type = var.ci_training_l40s_type
size = 1
min_size = 0
max_size = var.ci_training_l40s_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}
# H100 training pool (80GB VRAM, CUDA CC 90 — RL hyperopt)
resource "scaleway_k8s_pool" "ci_training_h100" {
count = var.enable_ci_training_h100_pool ? 1 : 0
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "ci-training-h100"
node_type = var.ci_training_h100_type
size = 1
min_size = 0 # Scale to zero when idle. Cold start: ~15 min NVIDIA driver install.
max_size = var.ci_training_h100_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}
# H100x2 training pool (2x 80GB VRAM — multi-GPU parallel hyperopt)
resource "scaleway_k8s_pool" "ci_training_h100x2" {
count = var.enable_ci_training_h100x2_pool ? 1 : 0
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "ci-training-h100x2"
node_type = var.ci_training_h100x2_type
size = 1
min_size = 0
max_size = var.ci_training_h100x2_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}
# H100-SXM training pool (2x 80GB SXM VRAM — NVLink multi-GPU)
resource "scaleway_k8s_pool" "ci_training_h100_sxm" {
count = var.enable_ci_training_h100_sxm_pool ? 1 : 0
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "ci-training-h100-sxm"
node_type = var.ci_training_h100_sxm_type
size = 1
min_size = 0
max_size = var.ci_training_h100_sxm_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}
# CPU compile pool (POP2-HC — high CPU, no GPU, for cargo check/build)
resource "scaleway_k8s_pool" "ci_compile_cpu" {
count = var.enable_ci_compile_cpu_pool ? 1 : 0
cluster_id = scaleway_k8s_cluster.foxhunt.id
name = "ci-compile-cpu"
node_type = var.ci_compile_cpu_type
size = 1
min_size = 0
max_size = var.ci_compile_cpu_max_size
autoscaling = true
autohealing = true
public_ip_disabled = var.public_ip_disabled
region = var.region
lifecycle {
ignore_changes = [size]
}
}