diff --git a/.gitignore b/.gitignore index 41825d45..47546e02 100644 --- a/.gitignore +++ b/.gitignore @@ -15,6 +15,7 @@ crash.*.log # to change depending on the environment. *.tfvars *.tfvars.json +!terraform/gcp_old/tpu-inference/k8s/prod.auto.tfvars # Ignore override files as they are usually used to override resources locally and so # are not checked in diff --git a/terraform/gcp_old/tpu-inference/k8s/.terraform.lock.hcl b/terraform/gcp_old/tpu-inference/k8s/.terraform.lock.hcl new file mode 100644 index 00000000..9e99a180 --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/.terraform.lock.hcl @@ -0,0 +1,82 @@ +# This file is maintained automatically by "terraform init". +# Manual edits may be lost in future updates. + +provider "registry.terraform.io/hashicorp/google" { + version = "7.41.0" + constraints = ">= 6.0.0" + hashes = [ + "h1:+wur8ZrXd5n6e4qXbiSRq9GntW/gmZ7rsWsvJPo3bB4=", + "zh:162ba4a321308dc683f13faaf8d0ec98efda9245f0a07d5abb867da5963d00b2", + "zh:2377a2db9b5a6977fe16f75b107bc47e663630b5a7a8e7a99006cdd820e70ed7", + "zh:24cd310110d08d4e5c74c78da41c4f95573c39e4eaef2c92650ebb71cb91e9a1", + "zh:3315101a286b57ed6a742a8c7c04809f9ef79da3bfa6df5847bbaff207e80d31", + "zh:50f193d70277c5a8c6d8736ac2f4bf4f390b0a265e8d0044c7cb30fbef1f2ee8", + "zh:5a32a8565ff7084bff87ccd56e26cb31b78c98625a94fd49ec78b48f27f1ced6", + "zh:5ec0abe1056adb1f84dc4645583e1ca92b891d694929d849f82cee7221f72bfb", + "zh:99fa3a8a1d42cace4df576b1111fe2a9286c8a40e6c7fab778f13213c9fe858a", + "zh:b715f6779d201c80e8f83a21acd43f474da13dd329721cec6cd4450c42398c5e", + "zh:cc1f3acaaf68f4ff19ef5893e9cefca63e5b93c0800517dfc0a60672911c3f26", + "zh:f569b65999264a9416862bca5cd2a6177d94ccb0424f3a4ef424428912b9cb3c", + "zh:fd3a26b4d131414045af53780aae841e930ec6af432794768549760c7d364765", + ] +} + +provider "registry.terraform.io/hashicorp/helm" { + version = "2.17.0" + constraints = "~> 2.12" + hashes = [ + "h1:kQMkcPVvHOguOqnxoEU2sm1ND9vCHiT8TvZ2x6v/Rsw=", + "zh:06fb4e9932f0afc1904d2279e6e99353c2ddac0d765305ce90519af410706bd4", + "zh:104eccfc781fc868da3c7fec4385ad14ed183eb985c96331a1a937ac79c2d1a7", + "zh:129345c82359837bb3f0070ce4891ec232697052f7d5ccf61d43d818912cf5f3", + "zh:3956187ec239f4045975b35e8c30741f701aa494c386aaa04ebabffe7749f81c", + "zh:66a9686d92a6b3ec43de3ca3fde60ef3d89fb76259ed3313ca4eb9bb8c13b7dd", + "zh:88644260090aa621e7e8083585c468c8dd5e09a3c01a432fb05da5c4623af940", + "zh:a248f650d174a883b32c5b94f9e725f4057e623b00f171936dcdcc840fad0b3e", + "zh:aa498c1f1ab93be5c8fbf6d48af51dc6ef0f10b2ea88d67bcb9f02d1d80d3930", + "zh:bf01e0f2ec2468c53596e027d376532a2d30feb72b0b5b810334d043109ae32f", + "zh:c46fa84cc8388e5ca87eb575a534ebcf68819c5a5724142998b487cb11246654", + "zh:d0c0f15ffc115c0965cbfe5c81f18c2e114113e7a1e6829f6bfd879ce5744fbb", + "zh:f569b65999264a9416862bca5cd2a6177d94ccb0424f3a4ef424428912b9cb3c", + ] +} + +provider "registry.terraform.io/hashicorp/kubernetes" { + version = "3.2.1" + hashes = [ + "h1:mcG69DdvaQvDNQzIo+SLVekECRLiNKavq5jbp/yieOU=", + "zh:067fe16a852d42e0f571712e36cb3e71855f917ea2041415e155f56ebc480d7f", + "zh:2815e174f8f0f032ea3a64f2196740ad000a39f88ae5646e7061bf15ed589f62", + "zh:2f94f6b689c59c43e596e724228f2861095d02c2a2ac2a257a4619667135ac75", + "zh:3e807310c84f11561b9ba06b978f03c46cfeca2e84ad0803d34d5d30a8a637cb", + "zh:5cba6f92202c60cac6898141356420709f5341b80ae4c360725cc647f86188ff", + "zh:72b841b6f0820d8f87c3d7c5a3611c35121ab9a4c1db4ea7a98b0319f209e474", + "zh:74770b892ee9b04829d92318d9e8ca96f8143b0c6c766e4141901908173fd01d", + "zh:7a723c8ebf9e218d0f7a0cfe6c0437f2b5eeb7ae015a14fad16e0f7fd9ef79ab", + "zh:a0f5073b2636a3894d4e9dd1b6853d5f324dd78728313bff79b842e5e9eca96f", + "zh:c13241cba993ef63a537beb6a1caf00e233bb045b50d197349530de0ee3276d5", + "zh:d52826f4b0227b7db99ea4a1d48f49a0bfb440563c92ebd2f8faec273c856c2d", + "zh:dc1cf5505a39a264a650b0830f74150ad02368787e5ead89e4007034f8f47831", + ] +} + +provider "registry.terraform.io/hashicorp/null" { + version = "3.3.0" + constraints = ">= 3.0.0" + hashes = [ + "h1:a14TKo7Xvg4W8+H1VA6p+oLZTLxVQnYUD8LOaOs14A8=", + "zh:021748b5ea3b5f6956f2e75c42c5cdc113b391fb98ac71364a4965d23b37000f", + "zh:3b27956f8541d46704fda234e0d535c2ae2a4b33411848b1ee262a1ec03568b0", + "zh:3de4ed47d6d0f4d8edba4a5092c7c9799950eda63989d8d0d2586e6afcb0aa20", + "zh:57ed8935c7d56dbc91cf2673534582cacfaab7a2f105f51d9f797e99df0c0c47", + "zh:58e176ba1d142827089e30e0711e007309a9f2726e8881986da5026e9778fdf4", + "zh:5949c4a3d4a93f841f155cdb7e991c087e637145c1630572e21948224f8f4923", + "zh:76d60f366b743003c1b085afa769b45b2198ee919927e45807d7d44fb42c067d", + "zh:78d5eefdd9e494defcb3c68d282b8f96630502cac21d1ea161f53cfe9bb483b3", + "zh:79cd1bab1261a07f84e917191d7ddc4340ac5f5524283767256f7ffd7f87caf0", + "zh:8ec9083038cf710b30e319eaa467c9df7fa52bbd9969b61053a35bc2cdd2e0a6", + "zh:a6e502cb579685ab7aeb886c2bb11ddd9cfed74b41008592d57cbc3351a9218b", + "zh:acb74d6b4f66ff6acfcda315df802a7432170ef3955c9b432cb4580767004006", + "zh:f0ce55d8d9ffdb33dab612b1246f9bab060a9d54fc32ce2b4a038646155660af", + ] +} diff --git a/terraform/gcp_old/tpu-inference/k8s/README.md b/terraform/gcp_old/tpu-inference/k8s/README.md new file mode 100644 index 00000000..36e2b695 --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/README.md @@ -0,0 +1,166 @@ +# MultiKueue TPU CI/CD Infrastructure on GKE + +This directory contains the Terraform configuration, Kueue manifest templates, and manifest generator script for managing the MultiKueue-based TPU CI/CD testing infrastructure across GKE clusters. + +--- + +## 1. Overview & Architecture + +The infrastructure uses a **MultiKueue** architecture to distribute TPU CI benchmark workloads submitted via Buildkite across dedicated GKE clusters in Google Cloud Platform: + +``` + +----------------------------------------------+ + | Manager Cluster | + | Project: cloud-ullm-inference-ci-cd | + | Name: tpu-ci-manager | + | Region: us-central1 | + +----------------------+-----------------------+ + | + Buildkite Agent Stack + (Controller) + | + Kueue MultiKueue + (Admission & Dispatch) + | + v + +----------------------------------------+ + | Worker Cluster | + | Project: cloud-tpu-inference-test | + | Name: tpu-ci-southamerica-west1-a | + | Location: southamerica-west1-a | + +-------------------+--------------------+ + | + GKE TPU Node Pools + (v6e-1: 1-chip, v6e-8: 8-chip) +``` + +--- + +## 2. Key Accomplishments & Design Principles + +### A. Centralized Buildkite Controller +To prevent race conditions where worker clusters compete to claim the same Buildkite job, the `agent-stack-k8s` controller runs **exclusively on the manager cluster**, and there is exactly **one** of it, on a single queue. + +The Buildkite queue deliberately carries no TPU information. A step names its shape with `--profile`, and the controller stays shape-agnostic, so adding a TPU profile is a regenerated ConfigMap rather than another Helm release. The controller's own pod is CPU-only and carries no Kueue queue label, so Kueue never queues or evicts it - only the workload it submits. + +### B. MultiKueue Dispatch over Connect Gateway +Kueue on the manager cluster inspects submitted jobs, matches cluster queues (`v6e-1-1x1`, `v6e-8-2x4`), and dispatches the workload object to the corresponding worker cluster (`southamerica-west1-a`) via GKE Connect Gateway using Workload Identity (`roles/gkehub.gatewayEditor`). + +### C. Native Cross-Project Image Pulling (No K8s Secrets Required) +Container images for testing (such as `us-central1-docker.pkg.dev/cloud-ullm-inference-ci-cd/tpu-inference-ci/vllm-tpu`) are hosted in the manager project. + +Terraform codifies cross-project IAM reader permissions (`roles/artifactregistry.reader`) for worker node service accounts (`tpu-ci-wkr-node@cloud-tpu-inference-test.iam.gserviceaccount.com`) on the manager project. GKE node containerd runtimes authenticate natively via GCP metadata tokens without requiring manual Kubernetes `imagePullSecrets` or service account keys. + +### D. TPU is the only resource under quota +Buildkite workloads generate helper containers (e.g. `copy-agent`), initContainers (`imagecheck`) and the gcsfuse sidecar, all of which request CPU and memory. Kueue's default `quotaCheckStrategy: BlockUndeclared` refuses to admit a workload that requests a resource its ClusterQueue does not cover, so an earlier version of the queues carried `cpu` and `memory` in `coveredResources` with quotas of `10000` / `10000Gi` as a stand-in for "unbounded" - a `nominalQuota` of `0` is a real zero, not unlimited. + +Both controller configs now set `resources.quotaCheckStrategy: IgnoreUndeclared` (Kueue 0.19, feature gate on by default), under which only the resources a ClusterQueue declares are checked. `queue_group.yaml.tpl` covers `google.com/tpu` alone; CPU and memory are enforced by the kube scheduler against node capacity, which is where they belong. Manager and workers must carry the same setting, since the worker's Kueue admits the mirrored workload too. + +### E. Standardized Node Pool Configuration +TPU node pools in `clusters.tf` are configured with: +- **Taints**: `google.com/tpu=present:NoSchedule` (allows GKE Cluster Autoscaler to simulate scale-up for pending TPU pods). +- **Reservation Affinity**: `reservation_affinity` targeting designated Cloud TPU reservations. +- **Placement by workload, not by flavor**: the `ResourceFlavor` is per-accelerator (`v6e`) and carries no `nodeLabels`, so every profile in the cohort shares one flavor and can borrow from it. Kueue therefore injects no node selector, and the submitted workload names `accelerator_label` and `topology` itself. That is what makes a 1-chip job and an 8-chip job draw on the same pool of chips. + +### F. JobSet for multi-pod workloads +A `batch/v1` Job cannot span hosts, so multi-host slices and prefill/decode disaggregation need JobSet. The operator is installed on the manager and every worker from a single `jobset_version`, and `jobset.x-k8s.io/jobset` is enabled in `integrations.frameworks` on both sides - MultiKueue mirrors the object across clusters, so the CRD, the operator version and the enabled framework list all have to match. + +A pool with `hosts > 1` is a multi-host TPU slice pool: GKE creates it with a `COMPACT` placement policy carrying the topology, one node per host, and the autoscaler scales it atomically - every host or none. `clusters.tf` checks that such a pool's node counts are whole slices and that one pool is one slice (`max_nodes == hosts`); more slices of a shape are more pools. A four-host `ct6e-standard-4t` 4x4 slice (16 chips, a JobSet with `parallelism 4`) has been run this way; none is declared in the tfvars today. + +### G. Physical shapes constrain borrowing +Node pools are single-shape, so with an 18-chip reservation one 8-chip node and ten 1-chip nodes leave nothing for a 16-chip slice. Quota borrowing across shapes is therefore not free: reclaiming chips means draining and deleting nodes of one shape before nodes of the other can be created. Measured on this cluster, that costs roughly 300s on top of the ~110s node scale-up. Worth knowing before tuning quotas - the cost is physical, not a Kueue setting. + +--- + +## 3. Workflow & Usage + +### Modifying Configuration +All cluster topology and pool limits are declared in `prod.auto.tfvars`. + +Example pool definition: +```hcl +tpu_pools = { + v6e-1-1x1 = { + machine_type = "ct6e-standard-1t" + accelerator = "v6e" # names the cohort and the ResourceFlavor + accelerator_label = "tpu-v6e-slice" + topology = "1x1" + chips_per_node = 1 + min_nodes = 2 # kept warm + nominal_nodes = 10 # Kueue nominalQuota = chips x nominal_nodes + max_nodes = 10 # autoscaling ceiling; anything above nominal is borrowed + reservation_name = "cloudtpu-20250327121505-861300654" + } + # a slice across hosts would add `hosts = 4` on a ct6e-standard-4t 4x4 pool; + # one multi-host pool is one slice and max_nodes must equal hosts. +} +``` + +The pool key is the profile name, and it is used verbatim as the Kueue +ClusterQueue and LocalQueue name and as the `--profile` a pipeline passes. +`nominal_nodes` is what the profile owns; anything between it and `max_nodes` +is borrowed from the cohort and is the first thing reclaimed when another +profile needs its own quota back. + +### Manifest Generation + +```bash +python3 -m pip install -r scripts/requirements.txt # hcl2, once +python3 scripts/generate_manifests.py +``` + +Manifests are rendered into per-cluster directories, numbered in the order +`kubectl` applies them: + +``` +generated/manager/ 01-base 02-multikueue-fleet 03-cohorts + 04-resource-flavors 05-queues [06-launcher] +generated/worker-/ 01-base 02-resource-flavors 03-queues + [04-launcher-rbac] +``` + +Only `01-base` ordering is load-bearing - it carries the Namespace everything +else is created into. The rest is soft: a ClusterQueue naming a ResourceFlavor +that does not exist yet goes inactive and recovers when it appears. + +`generated/` is committed, so regenerating should produce no diff unless you +changed a template or the tfvars. A non-empty diff after an unrelated change +means something drifted. + +### Applying Infrastructure & Manifests + +```bash +terraform fmt && terraform apply +./scripts/deploy_manifests.sh +``` + +`deploy_manifests.sh` applies each cluster's directory in one call. To land +them one at a time when debugging a fresh install: + +```bash +for f in generated/manager/*.yaml; do echo "== $f"; kubectl apply -f "$f" || break; done +``` + +### Verifying a deploy + +Against the **manager**: + +```bash +kubectl -n buildkite get localqueue +kubectl get crd jobsets.jobset.x-k8s.io +kubectl -n buildkite get pods -l app.kubernetes.io/name=agent-stack-k8s +``` + +Expect LocalQueues matching the profile names, the JobSet CRD, and exactly one +agent-stack pod - more than one means an older per-profile controller survived +and will compete for jobs. + +--- + +## 4. Best Practices & Troubleshooting + +### Build Cancellation +Always cancel builds via the **Buildkite UI** or Buildkite CLI (`buildkite-agent build cancel `). Avoid running manual `kubectl delete job` directly out-of-band, as deleting `Job` objects bypasses the controller event watcher and leaves Buildkite builds pending until step timeouts expire. + +### Pipeline Timeouts +In `.buildkite/pipeline_kube.yaml`, ensure pipeline steps include `timeout_in_minutes: N` and `cancel_on_build_failing: true` so stalled steps fail fast automatically. diff --git a/terraform/gcp_old/tpu-inference/k8s/backend.tf b/terraform/gcp_old/tpu-inference/k8s/backend.tf new file mode 100644 index 00000000..ffcdf856 --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/backend.tf @@ -0,0 +1,6 @@ +terraform { + backend "gcs" { + bucket = "cloud-ullm-inference-ci-cd-tf-state" + prefix = "buildkite-tpu-ci/foundation" + } +} diff --git a/terraform/gcp_old/tpu-inference/k8s/buildkite.tf b/terraform/gcp_old/tpu-inference/k8s/buildkite.tf new file mode 100644 index 00000000..9a258511 --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/buildkite.tf @@ -0,0 +1,70 @@ +# Buildkite Agent Stack (manager cluster only, so worker clusters never race +# to claim Buildkite jobs). +# +# One controller, one queue. Every TPU step goes through the launcher, which +# submits the real workload as a Kueue-managed object, so the Buildkite queue +# no longer has to encode a TPU shape - the profile is an argument to `launch`. +# Adding a TPU profile is then a regenerated ConfigMap rather than another Helm +# release. +# +# The agent pod is CPU-only and carries no Kueue queue label, so Kueue never +# queues or evicts it. Two properties follow: +# +# - The agent acquires its Buildkite job in seconds rather than after TPU +# admission and node pool scale-up, so the job is never held reserved long +# enough for the reservation to lapse and be picked up twice. +# - Kueue preemption evicts the workload without killing the agent, so a +# preempted run is a pause in the step log rather than a failed build +# needing `retry: automatic`. +# +# Placement - node selector, TPU resources, Kueue queue label - belongs to the +# submitted workload, not here. + +resource "kubernetes_namespace_v1" "buildkite_manager" { + provider = kubernetes.manager + + metadata { + name = "buildkite" + labels = { + "pod-security.kubernetes.io/enforce" = "baseline" + "pod-security.kubernetes.io/audit" = "restricted" + "pod-security.kubernetes.io/warn" = "restricted" + } + } +} + +resource "helm_release" "buildkite_agent_stack" { + provider = helm.manager + name = "agent-stack-k8s" + repository = "oci://ghcr.io/buildkite/helm" + chart = "agent-stack-k8s" + version = var.buildkite_agent_stack_chart_version + namespace = kubernetes_namespace_v1.buildkite_manager.metadata[0].name + create_namespace = false + force_update = true + cleanup_on_fail = true + replace = true + + values = [ + yamlencode({ + agentStackSecret = "agent-stack-k8s-secret" + + config = { + id = "tpu-ci" + queue = var.buildkite_queue + debug = var.buildkite_agent_stack_debug + + # Applies to the launcher pod only; the workload carries its own + # activeDeadlineSeconds. Sized to outlast queue wait plus the run, + # because the launcher waits for both. + job-active-deadline-seconds = var.tpu_job_max_runtime_seconds + pod-pending-timeout = "180m" + } + }) + ] + + depends_on = [ + kubernetes_namespace_v1.buildkite_manager, + null_resource.manager_external_secrets_helm + ] +} diff --git a/terraform/gcp_old/tpu-inference/k8s/cache.tf b/terraform/gcp_old/tpu-inference/k8s/cache.tf new file mode 100644 index 00000000..deb5cb91 --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/cache.tf @@ -0,0 +1,239 @@ +# A cache bucket per worker cluster, in that cluster's own region. +# +# The caches are the largest single lever on how efficiently the fleet uses its +# chips, and distance is what decides their cost. Measured from a pod against a +# bucket 10,000 km away, a compilation-cache miss took 502ms; against a bucket +# in the cluster's region, 36ms. A compile-heavy step asks whether an entry +# exists far more often than it reads one, so that difference moved a full suite +# from 1.60x bare metal's chip-minutes to 0.80x. +# +# Generated from var.worker_clusters, the same map that creates the clusters, +# their node pools and their queues - so a new region gets a co-located bucket +# without anyone remembering to make one. + +locals { + # Zone to region: southamerica-west1-a -> southamerica-west1. Same expression + # clusters.tf already uses for the NAT routers. + cache_region = { + for k, v in var.worker_clusters : + k => join("-", slice(split("-", v.location), 0, 2)) + } + + # Bucket names are globally unique across all of GCP and cannot be renamed - + # changing one destroys and recreates it - so the name carries a hash of what + # makes it ours. The inputs are things that never change for a given bucket: + # project, cluster key, purpose. Nothing that scales or gets retuned goes in, + # or resizing a pool would silently propose replacing the cache. + # + # Hashed rather than spelled out because names cap at 63 characters and + # "cloud-tpu-inference-test" plus "southamerica-west1" leaves little room. + # The readable facts are labels instead; nothing but terraform and the + # PersistentVolume ever refers to the name. + # One bucket per cache, not one bucket with two prefixes. + # + # The CSI driver identifies a volume by volumeHandle, which is the bucket + # name, so two PersistentVolumes on one bucket are one volume to the kubelet: + # it mounts once and both mountPaths land on the same directory. That is + # exactly what happened - /cache/jax listed the model cache. + # + # Separating them also lets the two have their own retention, which they + # want: compilation output is cheap to recreate and churns constantly, while + # a model is expensive to fetch and rarely changes. + cache_bucket = { + for k, v in var.worker_clusters : + k => format("%s-cache-%s-%s", var.name_prefix, local.cache_region[k], + substr(sha256("${v.project}/${k}/cache"), 0, 6)) + } + + models_bucket = { + for k, v in var.worker_clusters : + k => format("%s-models-%s-%s", var.name_prefix, local.cache_region[k], + substr(sha256("${v.project}/${k}/models"), 0, 6)) + } +} + +resource "google_storage_bucket" "cache" { + for_each = var.worker_clusters + + name = local.cache_bucket[each.key] + project = each.value.project + location = local.cache_region[each.key] + + uniform_bucket_level_access = true + storage_class = "STANDARD" + + # Real folders, chosen now because it cannot be chosen later: hierarchical + # namespace is fixed when a bucket is created and changing it means replacing + # the bucket. + # + # It matters here because of how gcsfuse writes. Every write goes to a + # temporary object and is then renamed, and on a flat bucket a rename is a + # copy followed by a delete. HNS makes it a folder operation - atomic, and + # with up to 8x the initial QPS limit. Requires uniform bucket-level access, + # which is set above. + # + # What it costs: no object versioning, retention lock, bucket lock, + # cross-bucket replication or object-level ACLs. A rebuildable cache uses + # none of those. + hierarchical_namespace { + enabled = true + } + + # No soft delete. It is on by default with a seven-day retention, and + # soft-deleted objects keep accruing storage charges for the whole of it. + # + # That is the wrong default for this bucket specifically: gcsfuse renames + # through copy-and-delete, and the lifecycle rule below deletes on a schedule, + # so a cache generates deletions continuously. Retaining a week of them would + # cost more than the cache itself on a bucket that turns over every 30 days - + # to protect data that is, by construction, recomputable. + soft_delete_policy { + retention_duration_seconds = 0 + } + + # Nothing here is public and nothing should become public by accident. The + # bucket holds model weights and compilation output for a CI fleet. + public_access_prevention = "enforced" + + # The cache is rebuildable by definition - a lost entry costs a recompile, + # not data - but rebuilding all of it costs about 2.7x a suite's chips, so + # do not let a terraform mistake take it. + force_destroy = false + + lifecycle_rule { + condition { + age = var.cache_lifecycle_age_days + } + action { + type = "Delete" + } + } + + # Resumable uploads that never finished. gcsfuse uploads large objects in + # parts, and a pod evicted or killed mid-write leaves those parts behind, + # billed as storage and invisible to an object listing. + lifecycle_rule { + condition { + age = 1 + } + action { + type = "AbortIncompleteMultipartUpload" + } + } + + labels = merge(var.labels, { + purpose = "tpu-ci-cache" + cluster = each.key + region = local.cache_region[each.key] + }) +} + +resource "google_storage_bucket" "models" { + for_each = var.worker_clusters + + name = local.models_bucket[each.key] + project = each.value.project + location = local.cache_region[each.key] + + uniform_bucket_level_access = true + storage_class = "STANDARD" + + # Real folders, chosen now because it cannot be chosen later: hierarchical + # namespace is fixed when a bucket is created and changing it means replacing + # the bucket. + # + # It matters here because of how gcsfuse writes. Every write goes to a + # temporary object and is then renamed, and on a flat bucket a rename is a + # copy followed by a delete. HNS makes it a folder operation - atomic, and + # with up to 8x the initial QPS limit. Requires uniform bucket-level access, + # which is set above. + # + # What it costs: no object versioning, retention lock, bucket lock, + # cross-bucket replication or object-level ACLs. A rebuildable cache uses + # none of those. + hierarchical_namespace { + enabled = true + } + + # No soft delete. It is on by default with a seven-day retention, and + # soft-deleted objects keep accruing storage charges for the whole of it. + # + # That is the wrong default for this bucket specifically: gcsfuse renames + # through copy-and-delete, and the lifecycle rule below deletes on a schedule, + # so a cache generates deletions continuously. Retaining a week of them would + # cost more than the cache itself on a bucket that turns over every 30 days - + # to protect data that is, by construction, recomputable. + soft_delete_policy { + retention_duration_seconds = 0 + } + + # Nothing here is public and nothing should become public by accident. The + # bucket holds model weights and compilation output for a CI fleet. + public_access_prevention = "enforced" + + # The cache is rebuildable by definition - a lost entry costs a recompile, + # not data - but rebuilding all of it costs about 2.7x a suite's chips, so + # do not let a terraform mistake take it. + force_destroy = false + + lifecycle_rule { + condition { + age = var.models_lifecycle_age_days + } + action { + type = "Delete" + } + } + + # Resumable uploads that never finished. gcsfuse uploads large objects in + # parts, and a pod evicted or killed mid-write leaves those parts behind, + # billed as storage and invisible to an object listing. + lifecycle_rule { + condition { + age = 1 + } + action { + type = "AbortIncompleteMultipartUpload" + } + } + + labels = merge(var.labels, { + purpose = "tpu-ci-models" + cluster = each.key + region = local.cache_region[each.key] + }) +} + +# Bucket-scoped and additive on purpose. +# +# _member manages exactly one (bucket, role, member) tuple. _binding would own +# the whole role and _policy the whole bucket, either of which fights anything +# else that manages IAM here - terraform reverting their change, them reverting +# terraform's. Project-level grants are also avoided: they are reconciled away +# by internal tooling, where bucket-level ones persist. +resource "google_storage_bucket_iam_member" "cache_workload_identity_rw" { + for_each = var.worker_clusters + + bucket = google_storage_bucket.cache[each.key].name + role = "roles/storage.objectUser" + member = "serviceAccount:${each.value.project}.svc.id.goog[buildkite/tpu-workload]" +} + +output "cache_buckets" { + description = "Compilation cache and model buckets per worker cluster." + value = { + for k, v in var.worker_clusters : k => { + cache = google_storage_bucket.cache[k].name + models = google_storage_bucket.models[k].name + region = local.cache_region[k] + } + } +} + +resource "google_storage_bucket_iam_member" "models_workload_identity_rw" { + for_each = var.worker_clusters + + bucket = google_storage_bucket.models[each.key].name + role = "roles/storage.objectUser" + member = "serviceAccount:${each.value.project}.svc.id.goog[buildkite/tpu-workload]" +} diff --git a/terraform/gcp_old/tpu-inference/k8s/clusters.tf b/terraform/gcp_old/tpu-inference/k8s/clusters.tf new file mode 100644 index 00000000..0217c993 --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/clusters.tf @@ -0,0 +1,456 @@ +# Cloud NAT for Manager Cluster (outbound image pulling) +resource "google_compute_router" "manager" { + name = "${var.name_prefix}-mgr-router" + project = var.project_id + region = var.manager_region + network = var.network +} + +resource "google_compute_router_nat" "manager" { + name = "${var.name_prefix}-mgr-nat" + project = var.project_id + region = var.manager_region + router = google_compute_router.manager.name + nat_ip_allocate_option = "AUTO_ONLY" + source_subnetwork_ip_ranges_to_nat = "ALL_SUBNETWORKS_ALL_IP_RANGES" +} + +# Cloud NAT for Worker Clusters (outbound image pulling) +resource "google_compute_router" "worker" { + for_each = var.worker_clusters + + name = "${var.name_prefix}-wkr-router-${each.key}" + project = each.value.project + region = join("-", slice(split("-", each.value.location), 0, 2)) + network = each.value.network +} + +resource "google_compute_router_nat" "worker" { + for_each = var.worker_clusters + + name = "${var.name_prefix}-wkr-nat-${each.key}" + project = each.value.project + region = join("-", slice(split("-", each.value.location), 0, 2)) + router = google_compute_router.worker[each.key].name + nat_ip_allocate_option = "AUTO_ONLY" + source_subnetwork_ip_ranges_to_nat = "ALL_SUBNETWORKS_ALL_IP_RANGES" +} + +resource "google_container_cluster" "manager" { + project = var.project_id + name = "${var.name_prefix}-manager" + location = var.manager_region + + network = var.network + subnetwork = var.manager_subnetwork + + deletion_protection = var.deletion_protection + remove_default_node_pool = true + initial_node_count = 1 + node_locations = var.manager_zones + networking_mode = "VPC_NATIVE" + enable_shielded_nodes = true + enable_intranode_visibility = true + + release_channel { + channel = var.release_channel + } + + workload_identity_config { + workload_pool = "${var.project_id}.svc.id.goog" + } + + fleet { + project = var.project_id + } + + # These labels ask GKE Fleet to generate ClusterProfile inventory objects in + # kueue-system. MultiKueue can then use federated credentials instead of a + # stored worker kubeconfig and bearer token. + resource_labels = merge(local.common_labels, { + fleet-clusterinventory-management-cluster = "true" + fleet-clusterinventory-namespace = "kueue-system" + role = "manager" + }) + + private_cluster_config { + enable_private_nodes = true + enable_private_endpoint = var.enable_private_endpoint + master_ipv4_cidr_block = var.manager_master_ipv4_cidr_block + + master_global_access_config { + enabled = true + } + } + + addons_config { + gce_persistent_disk_csi_driver_config { + enabled = true + } + } + + secret_manager_config { + enabled = true + rotation_config { + enabled = true + rotation_interval = "300s" + } + } + + secret_sync_config { + enabled = true + rotation_config { + enabled = true + rotation_interval = "300s" + } + } + + security_posture_config { + mode = "BASIC" + vulnerability_mode = "VULNERABILITY_BASIC" + } + + monitoring_config { + enable_components = [ + "APISERVER", + "CONTROLLER_MANAGER", + "DAEMONSET", + "DEPLOYMENT", + "HPA", + "KUBELET", + "POD", + "SCHEDULER", + "STATEFULSET", + "STORAGE", + "SYSTEM_COMPONENTS", + ] + managed_prometheus { + enabled = true + } + } + + lifecycle { + ignore_changes = [monitoring_config] + } +} + +resource "google_container_node_pool" "manager_system" { + project = var.project_id + name = "system" + location = google_container_cluster.manager.location + cluster = google_container_cluster.manager.name + + # Regional node-pool initial counts are per zone. Start with one in each + # configured zone, then let total autoscaling enforce the aggregate floor. + initial_node_count = 1 + + autoscaling { + total_min_node_count = var.manager_system_min_nodes + total_max_node_count = var.manager_system_max_nodes + location_policy = "BALANCED" + } + + node_config { + machine_type = var.manager_system_machine_type + image_type = "COS_CONTAINERD" + service_account = coalesce( + var.manager_node_service_account, + try(google_service_account.manager_nodes[0].email, null), + "default" + ) + oauth_scopes = ["https://www.googleapis.com/auth/cloud-platform"] + labels = { + "tpu-ci.google.com/role" = "system" + } + + workload_metadata_config { + mode = "GKE_METADATA" + } + + shielded_instance_config { + enable_integrity_monitoring = true + enable_secure_boot = true + } + } + + management { + auto_repair = true + auto_upgrade = true + } + + upgrade_settings { + max_surge = 1 + max_unavailable = 0 + } + + lifecycle { + ignore_changes = [initial_node_count] + + precondition { + condition = var.manager_system_min_nodes >= length(var.manager_zones) + error_message = "manager_system_min_nodes must provide at least one system node per manager zone." + } + } +} + +resource "google_container_cluster" "worker" { + for_each = var.worker_clusters + + name = "${var.name_prefix}-${each.key}" + project = each.value.project + location = each.value.location + + network = each.value.network + subnetwork = each.value.subnetwork + + remove_default_node_pool = true + initial_node_count = 1 + + deletion_protection = var.deletion_protection + + release_channel { + channel = var.release_channel + } + + private_cluster_config { + enable_private_nodes = var.enable_private_nodes + enable_private_endpoint = false + master_ipv4_cidr_block = each.value.master_ipv4_cidr_block + } + + ip_allocation_policy {} + + workload_identity_config { + workload_pool = "${each.value.project}.svc.id.goog" + } + + # Cloud Storage FUSE, so a workload can mount a bucket as a filesystem. Not + # enabled by default in GKE, unlike the Persistent Disk driver, and it + # requires Workload Identity - configured above. + # + # Nothing mounts a bucket yet, and enabling the driver changes nothing on its + # own: a pod has to opt in with the gke-gcsfuse/volumes: "true" annotation + # before the sidecar is injected. This is here so the option can be measured + # against the two we already have - a cloned disk, and gs:// read directly, + # which a workload pod can already do. + addons_config { + gcs_fuse_csi_driver_config { + enabled = true + } + } + + # OPTIMIZE_UTILIZATION, deliberately, while the two TPU shapes share one + # cohort. The pools are physically separate - 1-chip and 8-chip nodes drawing + # on the same reservation - so the single-chip lane can only use the eight + # chips an idle 8-chip node is holding once that node is torn down. Reaping + # promptly is what makes the handover possible at all; BALANCED would let the + # idle node sit and starve the other shape for longer. + # + # This stops being the right trade once each shape has its own nominal + # capacity and no longer needs to borrow across shapes. Revisit with the pool + # sizes, not on its own. + cluster_autoscaling { + autoscaling_profile = "OPTIMIZE_UTILIZATION" + } + + resource_labels = merge(local.common_labels, { + role = "worker" + worker = each.key + }) + + lifecycle { + ignore_changes = [ + private_cluster_config, + secret_sync_config, + secret_manager_config, + resource_labels["asmv2"], + resource_labels["mesh_id"], + monitoring_config + ] + } +} + +resource "google_container_node_pool" "worker_system" { + for_each = var.worker_clusters + + name = "system" + project = each.value.project + cluster = "${var.name_prefix}-${each.key}" + location = each.value.location + node_count = try(each.value.system_min_nodes, 1) + + autoscaling { + min_node_count = try(each.value.system_min_nodes, 1) + max_node_count = try(each.value.system_max_nodes, 3) + } + + management { + auto_repair = true + auto_upgrade = true + } + + node_config { + machine_type = try(each.value.system_machine_type, "e2-standard-4") + service_account = coalesce( + try(each.value.node_service_account, null), + try(google_service_account.worker_nodes[each.key].email, null), + "default" + ) + oauth_scopes = ["https://www.googleapis.com/auth/cloud-platform"] + + labels = merge(local.common_labels, { + profile = "system" + worker = each.key + "tpu-ci.google.com/worker" = each.key + }) + + resource_labels = merge(local.common_labels, { + profile = "system" + worker = each.key + }) + + metadata = { + disable-legacy-endpoints = "true" + } + } + + lifecycle { + ignore_changes = [ + node_config[0].resource_labels["asmv2"], + node_config[0].resource_labels["mesh_id"] + ] + } +} + +resource "google_container_node_pool" "worker_tpu" { + for_each = local.tpu_node_pools + + name = each.value.profile_name + project = each.value.project + cluster = "${var.name_prefix}-${each.value.worker_name}" + location = each.value.location + + initial_node_count = each.value.min_nodes + + autoscaling { + min_node_count = each.value.min_nodes + max_node_count = each.value.max_nodes + location_policy = "ANY" + } + + management { + auto_repair = true + auto_upgrade = true + } + + node_config { + machine_type = each.value.machine_type + service_account = coalesce( + try(each.value.node_service_account, null), + try(google_service_account.worker_nodes[each.value.worker_name].email, null), + "default" + ) + oauth_scopes = ["https://www.googleapis.com/auth/cloud-platform"] + + gcfs_config { + enabled = true + } + + labels = merge(local.common_labels, { + profile = each.value.profile_name + worker = each.value.worker_name + "tpu-ci.google.com/worker" = each.value.worker_name + "tpu-ci.google.com/profile" = each.value.profile_name + "cloud.google.com/gke-tpu-topology" = try(each.value.topology, "") + "google.com/tpu-topology" = try(each.value.kueue_topology, try(each.value.topology, "")) + "google.com/tpu-chips-per-node" = tostring(each.value.chips_per_node) + }) + + taint { + key = "google.com/tpu" + value = "present" + effect = "NO_SCHEDULE" + } + + dynamic "reservation_affinity" { + for_each = try(each.value.reservation_name, "") != "" ? [1] : [] + content { + consume_reservation_type = "SPECIFIC_RESERVATION" + key = "compute.googleapis.com/reservation-name" + values = [each.value.reservation_name] + } + } + + resource_labels = merge(local.common_labels, { + profile = each.value.profile_name + worker = each.value.worker_name + }) + + metadata = { + disable-legacy-endpoints = "true" + } + + + } + + # A multi-host TPU slice pool: GKE needs the topology on a placement policy + # and creates one node per host. The pool is the atomic unit - the + # autoscaler goes from zero to every host of the slice and back, never to a + # part of one - which is why min/max are validated below as whole slices. + dynamic "placement_policy" { + for_each = each.value.is_multi_host && try(each.value.topology, null) != null ? [1] : [] + content { + type = "COMPACT" + tpu_topology = each.value.topology + } + } + + lifecycle { + precondition { + condition = !each.value.is_multi_host || ( + each.value.min_nodes % each.value.hosts == 0 && each.value.max_nodes % each.value.hosts == 0 + ) + error_message = "${each.key}: a multi-host pool scales in whole slices; min_nodes and max_nodes must be multiples of hosts (${each.value.hosts})." + } + precondition { + condition = !each.value.is_multi_host || each.value.max_nodes == each.value.hosts + error_message = "${each.key}: one GKE multi-host node pool is one slice, so max_nodes must equal hosts (${each.value.hosts}); more slices of a shape are more pools." + } + ignore_changes = [ + # A create-time field only: GKE reports whatever the pool has scaled to + # since, so it drifts on its own and no apply can set it. Left tracked, + # any edit to min_nodes reads as a change to it and forces the pool to be + # destroyed and rebuilt, when all that was wanted is a new autoscaling + # floor. The manager's system pool ignores it for the same reason. + initial_node_count, + node_config[0].guest_accelerator, + node_config[0].reservation_affinity, + node_config[0].kubelet_config, + node_config[0].shielded_instance_config, + node_config[0].windows_node_config, + node_config[0].advanced_machine_features, + node_config[0].resource_labels["asmv2"], + node_config[0].resource_labels["mesh_id"], + upgrade_settings, + node_drain_config + ] + } +} + +resource "google_gke_hub_membership" "worker" { + for_each = var.worker_clusters + + project = var.project_id + membership_id = "${var.name_prefix}-${each.key}" + endpoint { + gke_cluster { + resource_link = "//container.googleapis.com/projects/${each.value.project}/locations/${each.value.location}/clusters/${google_container_cluster.worker[each.key].name}" + } + } + + lifecycle { + ignore_changes = [ + authority + ] + } +} diff --git a/terraform/gcp_old/tpu-inference/k8s/external_secrets.tf b/terraform/gcp_old/tpu-inference/k8s/external_secrets.tf new file mode 100644 index 00000000..8947139c --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/external_secrets.tf @@ -0,0 +1,53 @@ +# External Secrets Operator Helm Installation (Manager & Worker Clusters) +# Note: Kubernetes Custom Resources (ClusterSecretStore, ExternalSecret) are generated +# and applied via the manifest generator for inspection before application. + +resource "null_resource" "manager_external_secrets_helm" { + triggers = { + endpoint = google_container_cluster.manager.endpoint + } + + provisioner "local-exec" { + command = < v + if var.create_service_accounts && try(v.node_service_account, null) == null + } + + project = each.value.project + account_id = "${var.name_prefix}-wkr-node" + display_name = "Worker GKE Node SA (${each.key})" +} + +resource "google_project_iam_member" "worker_nodes" { + for_each = { + for k, v in local.worker_node_role_bindings : k => v + if var.create_service_accounts && try(var.worker_clusters[v.worker_name].node_service_account, null) == null + } + + project = each.value.project + role = each.value.role + member = "serviceAccount:${google_service_account.worker_nodes[each.value.worker_name].email}" +} + +# Worker Node Access on Manager Project (Artifact Registry Reader for pulling images) +resource "google_project_iam_member" "worker_nodes_manager_project" { + for_each = var.worker_clusters + + project = var.project_id + role = "roles/artifactregistry.reader" + member = "serviceAccount:${coalesce(try(each.value.node_service_account, null), try(google_service_account.worker_nodes[each.key].email, null))}" +} + +# Manager Node Access on Worker Projects (Artifact Registry Reader for gcp-auth-plugin and tooling) +resource "google_project_iam_member" "manager_nodes_worker_project" { + for_each = { + for worker_key, worker in var.worker_clusters : worker_key => worker + if var.create_service_accounts && var.manager_node_service_account == null + } + + project = each.value.project + role = "roles/artifactregistry.reader" + member = "serviceAccount:${google_service_account.manager_nodes[0].email}" +} + +# Secret Manager Accessor for External Secrets Operator via direct Workload Identity +resource "google_project_iam_member" "external_secrets_manager_secret_accessor" { + project = var.buildkite_secret_project + role = "roles/secretmanager.secretAccessor" + member = "serviceAccount:${var.project_id}.svc.id.goog[external-secrets/external-secrets]" +} + +resource "google_project_iam_member" "external_secrets_worker_secret_accessor" { + for_each = var.worker_clusters + + project = var.buildkite_secret_project + role = "roles/secretmanager.secretAccessor" + member = "serviceAccount:${each.value.project}.svc.id.goog[external-secrets/external-secrets]" +} + +# GKE Hub Service Agent IAM Binding +resource "google_project_iam_member" "gkehub_service_agent" { + for_each = var.worker_clusters + + project = each.value.project + role = "roles/gkehub.serviceAgent" + member = "serviceAccount:service-${data.google_project.manager.number}@gcp-sa-gkehub.iam.gserviceaccount.com" +} + +data "google_project" "manager" { + project_id = var.project_id +} + +# The manager node service account no longer holds gkehub.viewer or +# gkehub.gatewayEditor. +# +# Both existed for the impersonation model: Kueue, External Secrets and the +# launcher all ran as Kubernetes service accounts annotated onto this account, +# so its roles were theirs. Each of them now holds its own grant directly, and +# nothing impersonates this account, so its Fleet permissions are reachable by +# nobody. +# +# Kept: artifactregistry.reader, logging, monitoring and +# stackdriver.resourceMetadata.writer, which the nodes themselves use. Removing +# artifactregistry.reader on the worker projects broke a build and was restored +# in 64ca84f; that is the shape of mistake this comment exists to prevent +# repeating. + +# Connect Gateway for the Kueue controller: it creates, updates and deletes +# workloads on the worker clusters, so it needs write. gatewayEditor is that; +# gatewayAdmin additionally carries impersonation and policy verbs it never +# uses. +# +# gkehub.viewer for the same reason the launcher has it - resolving a Connect +# Gateway target means listing Fleet memberships first. Kueue worked without it +# only because the manager node SA happened to hold it; that grant is going +# away, so make the dependency explicit rather than inherited. +resource "google_project_iam_member" "connect_gateway_kueue_wi" { + for_each = toset(["roles/gkehub.gatewayEditor", "roles/gkehub.viewer"]) + + project = var.project_id + role = each.value + member = "serviceAccount:${var.project_id}.svc.id.goog[kueue-system/kueue-controller-manager]" +} + +resource "google_project_iam_member" "connect_gateway_kueue_worker_project" { + for_each = var.worker_clusters + + project = each.value.project + role = "roles/gkehub.gatewayEditor" + member = "serviceAccount:${var.project_id}.svc.id.goog[kueue-system/kueue-controller-manager]" +} diff --git a/terraform/gcp_old/tpu-inference/k8s/jobset.tf b/terraform/gcp_old/tpu-inference/k8s/jobset.tf new file mode 100644 index 00000000..149bf5c4 --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/jobset.tf @@ -0,0 +1,42 @@ +# JobSet operator. +# +# Required for any workload that spans more than one pod: multi-host TPU slices +# (one pod per host, gang admitted) and prefill/decode disaggregation (separate +# server pods plus a benchmark pod). A batch/v1 Job cannot express either. +# +# Installed on the manager and on every worker. MultiKueue mirrors the JobSet +# object onto the selected worker, so the CRD and the operator must exist on +# both sides at compatible versions. + +resource "helm_release" "jobset_manager" { + provider = helm.manager + name = "jobset" + repository = "oci://registry.k8s.io/jobset/charts" + chart = "jobset" + version = var.jobset_version + namespace = "jobset-system" + create_namespace = true + wait = true +} + +resource "null_resource" "jobset_worker" { + for_each = var.worker_clusters + + triggers = { + version = var.jobset_version + endpoint = google_container_cluster.worker[each.key].endpoint + } + + provisioner "local-exec" { + command = < item + } + + tpu_node_pools = { + for item in flatten([ + for worker_name, worker in var.worker_clusters : [ + for profile_name, pool in worker.tpu_pools : { + key = "${worker_name}/${profile_name}" + worker_name = worker_name + project = worker.project + location = worker.location + profile_name = profile_name + machine_type = pool.machine_type + topology = try(pool.topology, null) + chips_per_node = pool.chips_per_node + min_nodes = pool.min_nodes + max_nodes = pool.max_nodes + # Hosts in one slice. 1 is a single-host pool; more makes this a + # multi-host TPU slice pool, which GKE creates with a placement + # policy carrying the topology and scales atomically - every host + # or none - so its node counts are whole slices (see clusters.tf). + hosts = try(pool.hosts, 1) + is_multi_host = try(pool.hosts, 1) > 1 + reservation_name = try(pool.reservation_name, null) + node_service_account = try(worker.node_service_account, null) + } + ] + ]) : item.key => item + } +} diff --git a/terraform/gcp_old/tpu-inference/k8s/prod.auto.tfvars b/terraform/gcp_old/tpu-inference/k8s/prod.auto.tfvars new file mode 100644 index 00000000..73a5c7ab --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/prod.auto.tfvars @@ -0,0 +1,70 @@ +project_id = "cloud-ullm-inference-ci-cd" +name_prefix = "tpu-ci" +network = "projects/cloud-ullm-inference-ci-cd/global/networks/default" +buildkite_secret_project = "cloud-ullm-inference-ci-cd" +buildkite_queue = "kube" + +manager_region = "us-central1" +manager_zones = ["us-central1-a", "us-central1-b", "us-central1-c"] +manager_subnetwork = "projects/cloud-ullm-inference-ci-cd/regions/us-central1/subnetworks/default" +manager_master_ipv4_cidr_block = "172.16.0.0/28" + +worker_clusters = { + southamerica-west1-a = { + project = "cloud-tpu-inference-test" + location = "southamerica-west1-a" + node_service_account = "32478767326-compute@developer.gserviceaccount.com" + network = "projects/cloud-tpu-inference-test/global/networks/default" + subnetwork = "projects/cloud-tpu-inference-test/regions/southamerica-west1/subnetworks/default" + master_ipv4_cidr_block = "192.168.255.0/28" + system_machine_type = "e2-standard-4" + system_min_nodes = 1 + system_max_nodes = 3 + + tpu_pools = { + # Two shapes on one 18-chip reservation. Nominal quota is what a lane + # is guaranteed; max is how far it may grow by borrowing idle quota from + # the other, reclaimed (reclaimWithinCohort: Any) when the owner needs + # it back. Nominal sums to the reservation, so Kueue never admits more + # than can exist. A pool with hosts > 1 is a multi-host slice pool + # (clusters.tf); none is declared here yet. + v6e-1-1x1 = { + machine_type = "ct6e-standard-1t" + accelerator = "v6e" + accelerator_label = "tpu-v6e-slice" + topology = "1x1" + chips_per_node = 1 + # Ten chips of its own: with fourteen single-chip steps per suite this + # lane is always busy, so it owns most of the reservation and keeps two + # nodes warm. Borrowing across shapes here means tearing down one + # shape's nodes and building the other's, about five minutes each way, + # so nominal == max: nothing to gain from reaching into the other + # lanes and a cold start per step to lose. + min_nodes = 2 + nominal_nodes = 10 + max_nodes = 10 + reservation_name = "cloudtpu-20250327121505-861300654" + } + v6e-8-2x4 = { + machine_type = "ct6e-standard-8t" + accelerator = "v6e" + accelerator_label = "tpu-v6e-slice" + topology = "2x4" + chips_per_node = 8 + # One node. The multi-chip work is small - a handful of short steps + # per suite that serialise onto it and reuse it - so a second node + # would sit idle, and min_nodes 0 pays one cold start per build. + min_nodes = 0 + nominal_nodes = 1 + max_nodes = 1 + reservation_name = "cloudtpu-20250327121505-861300654" + } + } + } +} + +labels = { + environment = "production" + workload = "tpu-ci" + owner = "tpu-inference" +} diff --git a/terraform/gcp_old/tpu-inference/k8s/providers.tf b/terraform/gcp_old/tpu-inference/k8s/providers.tf new file mode 100644 index 00000000..8688460c --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/providers.tf @@ -0,0 +1,21 @@ +provider "google" { + project = var.project_id +} + +data "google_client_config" "current" {} + +provider "helm" { + alias = "manager" + kubernetes { + host = "https://${google_container_cluster.manager.endpoint}" + token = data.google_client_config.current.access_token + cluster_ca_certificate = base64decode(google_container_cluster.manager.master_auth[0].cluster_ca_certificate) + } +} + +provider "kubernetes" { + alias = "manager" + host = "https://${google_container_cluster.manager.endpoint}" + token = data.google_client_config.current.access_token + cluster_ca_certificate = base64decode(google_container_cluster.manager.master_auth[0].cluster_ca_certificate) +} diff --git a/terraform/gcp_old/tpu-inference/k8s/scripts/deploy_manifests.sh b/terraform/gcp_old/tpu-inference/k8s/scripts/deploy_manifests.sh new file mode 100755 index 00000000..740c93ea --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/scripts/deploy_manifests.sh @@ -0,0 +1,136 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Applies generated/ to the manager and every worker cluster. +# +# deploy_manifests.sh [--yes] +# +# What is applied is the committed generated/, never a fresh render: those +# manifests went through review, and a shared cluster should only ever see what +# someone read in a PR. The generator still runs - into a scratch directory, +# purely to prove the committed output is up to date with prod.auto.tfvars. +# +# Note this only ever creates and updates. A queue deleted from the tfvars +# disappears from generated/ but stays in the cluster until someone removes it +# by hand. + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +K8S_DIR="$(dirname "$SCRIPT_DIR")" +GENERATED_DIR="${K8S_DIR}/generated" + +# The generator imports hcl2 and yaml. Point PYTHON at the interpreter that has +# them - a virtualenv's - if your python3 does not. +PYTHON="${PYTHON:-python3}" + +MANAGER_CLUSTER="tpu-ci-manager" +MANAGER_LOCATION="us-central1" +MANAGER_PROJECT="cloud-ullm-inference-ci-cd" +WORKER_PROJECT="cloud-tpu-inference-test" + +ASSUME_YES=0 +for arg in "$@"; do + case "$arg" in + -y|--yes) ASSUME_YES=1 ;; + *) echo "usage: ${0##*/} [--yes]" >&2; exit 2 ;; + esac +done + +# Deploy gets a kubeconfig of its own. get-credentials writes to whatever +# KUBECONFIG names and repoints its current-context, so sharing yours would +# leave your kubectl aimed at the last worker this touched. +TMP_ROOT="$(mktemp -d)" +trap 'rm -rf "${TMP_ROOT}"' EXIT +export KUBECONFIG="${TMP_ROOT}/kubeconfig" + +# Set by use_cluster, read by apply_dir. Every kubectl call names it explicitly +# rather than trusting the current context, so a get-credentials that failed +# cannot send the manager's manifests to a worker. +KCTX="" + +use_cluster() { + local cluster="$1" location="$2" project="$3" + gcloud container clusters get-credentials "$cluster" \ + --location "$location" --project "$project" + KCTX="$(kubectl config current-context)" +} + +confirm() { + (( ASSUME_YES )) && return 0 + if [[ ! -t 0 ]]; then + echo "Error: nothing to read a confirmation from; pass --yes to apply unattended." >&2 + exit 1 + fi + local reply + read -r -p "$1 [y/N] " reply + case "$reply" in + [yY]|[yY][eE][sS]) return 0 ;; + *) echo "Aborted."; exit 1 ;; + esac +} + +apply_dir() { + local dir="$1" label="$2" + + echo "" + echo "--- ${label}: pending changes ---" + # kubectl diff exits 1 for "there are differences" and >1 for a real failure. + # A CRD the cluster does not have yet is the >1 case, which is what a + # first-ever deploy into an empty cluster looks like. + local rc=0 + kubectl --context "$KCTX" diff -f "$dir" || rc=$? + case "$rc" in + 0) echo "(no changes)"; return 0 ;; + 1) ;; + *) echo "Error: could not diff against ${label}." >&2; exit 1 ;; + esac + + confirm "Apply the above to ${label}?" + # kubectl reads a directory in lexical order, which is what the NN- prefixes + # encode: 01-base carries the Namespace that everything else is created into. + # The rest is soft ordering - a ClusterQueue naming a ResourceFlavor that does + # not exist yet goes inactive and recovers when it appears - so applying the + # directory in one call is equivalent to applying the files in sequence. + kubectl --context "$KCTX" apply -f "$dir" +} + +if [[ ! -d "${GENERATED_DIR}" ]]; then + echo "Error: ${GENERATED_DIR} does not exist. Run python3 scripts/generate_manifests.py first." >&2 + exit 1 +fi + +echo "Checking generated/ is up to date with prod.auto.tfvars" +if ! "${PYTHON}" "${SCRIPT_DIR}/generate_manifests.py" --out-dir "${TMP_ROOT}/fresh" >/dev/null; then + echo "Error: could not render the manifests with '${PYTHON}' (see above)." >&2 + echo "Set PYTHON to an interpreter with hcl2 and pyyaml installed." >&2 + exit 1 +fi +if ! diff -ruN --label committed "${GENERATED_DIR}" --label fresh "${TMP_ROOT}/fresh"; then + echo "" >&2 + echo "Error: generated/ does not match what prod.auto.tfvars renders (diff above)." >&2 + echo "Run 'python3 scripts/generate_manifests.py' and commit the result, so what" >&2 + echo "reaches the cluster is what was reviewed." >&2 + exit 1 +fi + +echo "=========================================================" +echo "Manager Cluster" +echo "=========================================================" +use_cluster "${MANAGER_CLUSTER}" "${MANAGER_LOCATION}" "${MANAGER_PROJECT}" +apply_dir "${GENERATED_DIR}/manager" "manager" + +for worker_dir in "${GENERATED_DIR}"/worker-*/; do + [[ -d "${worker_dir}" ]] || continue + worker_key=$(basename "${worker_dir}" | sed -e 's/^worker-//') + + echo "" + echo "=========================================================" + echo "Worker Cluster: ${worker_key}" + echo "=========================================================" + # The directory is named for the cluster's location, so it is both the + # --location and the suffix of the cluster name. + use_cluster "tpu-ci-${worker_key}" "${worker_key}" "${WORKER_PROJECT}" + apply_dir "${worker_dir}" "worker ${worker_key}" +done + +echo "" +echo "Deployment Complete!" diff --git a/terraform/gcp_old/tpu-inference/k8s/scripts/generate_manifests.py b/terraform/gcp_old/tpu-inference/k8s/scripts/generate_manifests.py new file mode 100755 index 00000000..0b2d91be --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/scripts/generate_manifests.py @@ -0,0 +1,324 @@ +#!/usr/bin/env python3 +import argparse +import hashlib +import os +import sys +import shutil +from pathlib import Path +from string import Template +import hcl2 + +def clean_val(val): + """Clean hcl2 string quotes.""" + if isinstance(val, str): + return val.strip('"\'') + return val + +def parse_tfvars(file_path): + """Parse HCL tfvars file using official hcl2 library.""" + with open(file_path, "r") as f: + data = hcl2.load(f) + + project_id = clean_val(data["project_id"]) + name_prefix = clean_val(data["name_prefix"]) + manager_region = clean_val(data["manager_region"]) + secret_project = clean_val(data.get("buildkite_secret_project", project_id)) + secret_id = clean_val(data.get("buildkite_secret_id", "buildkite-tpu-ci-agent-dev-token")) + + worker_clusters = {} + # hcl2 keeps the surrounding quotes on object keys that are written quoted + # in tfvars, and these keys become Kubernetes object names. Without this a + # quoted key yields a ClusterQueue literally named "v6e-1-1x1", which fails + # far from its cause. + for raw_key, wdata in data["worker_clusters"].items(): + key = clean_val(raw_key) + pools = {} + for raw_pk, pdata in wdata["tpu_pools"].items(): + pk = clean_val(raw_pk) + chips = int(pdata["chips_per_node"]) + nominal_nodes = int(pdata.get("nominal_nodes", pdata["max_nodes"])) + max_nodes = int(pdata["max_nodes"]) + accelerator = clean_val(pdata["accelerator"]) + + pools[pk] = { + "chips": chips, + "nominal_nodes": nominal_nodes, + "max_nodes": max_nodes, + "quota": chips * nominal_nodes, + "accelerator": accelerator, + } + + project = clean_val(wdata["project"]) + worker_clusters[key] = { + "project": project, + "location": clean_val(wdata["location"]), + "cluster_name": f"{name_prefix}-{key}", + "profile_name": f"{name_prefix}-{key}-global", + # Must match cache.tf exactly - same inputs, same hash, same name. + # If these two disagree the PersistentVolume points at a bucket that + # does not exist and every pod fails to mount. + "cache_bucket": bucket_name(name_prefix, project, key, "cache"), + "models_bucket": bucket_name(name_prefix, project, key, "models"), + "pools": pools + } + + return { + "project_id": project_id, + "name_prefix": name_prefix, + "manager_region": manager_region, + "manager_cluster": f"{name_prefix}-manager", + "secret_project": secret_project, + "secret_id": secret_id, + "namespace": "buildkite", + "worker_clusters": worker_clusters + } + +def bucket_name(name_prefix, project, cluster_key, purpose): + """A regional bucket for one worker cluster. + + Mirrors locals.cache_bucket in cache.tf. Bucket names are globally unique + and cannot be renamed, so the name carries a hash of project, cluster and + purpose - inputs that never change for a given bucket. Terraform creates + it; this only has to arrive at the same string. + """ + region = "-".join(cluster_key.split("-")[:2]) + digest = hashlib.sha256(f"{project}/{cluster_key}/{purpose}".encode()).hexdigest()[:6] + return f"{name_prefix}-{purpose}-{region}-{digest}" + + +def clean_doc(doc_str): + """Strip leading/trailing document separators and whitespace.""" + s = doc_str.strip() + while s.startswith("---"): + s = s[3:].strip() + while s.endswith("---"): + s = s[:-3].strip() + return s + +def format_admission_checks(accelerator): + """MultiKueue dispatch block appended to a ClusterQueue spec. + + Only the manager dispatches; worker ClusterQueues admit locally and render + this empty. This is the sole difference between the manager and worker + forms of queue_group.yaml.tpl. + """ + if not accelerator: + return "" + return ( + "\n admissionChecksStrategy:" + "\n admissionChecks:" + f"\n - name: {accelerator}-multikueue-dispatch" + ) + +def render_queue_group(template, pools, namespace, with_admission_checks): + """One ClusterQueue + LocalQueue per pool, sorted for stable output. + + pools maps queue name -> {"quota", "accelerator"}. + """ + tmpl = Template(template) + return [ + clean_doc(tmpl.substitute( + QUEUE_NAME=name, + ACCELERATOR=info["accelerator"], + NAMESPACE=namespace, + NOMINAL_QUOTA=info["quota"], + ADMISSION_CHECKS=format_admission_checks( + info["accelerator"] if with_admission_checks else None + ), + )) + for name, info in sorted(pools.items()) + ] + +def generate_manager_parts(config, templates): + parts = {} + + # 1. Base (Namespace, ClusterSecretStore, ExternalSecret) + base_tmpl = Template(templates["base"]) + parts["01-base.yaml"] = base_tmpl.substitute( + NAMESPACE=config["namespace"], + SECRET_PROJECT=config["secret_project"], + SECRET_ID=config["secret_id"], + CLUSTER_LOCATION=config["manager_region"], + CLUSTER_NAME=config["manager_cluster"] + ).strip() + "\n" + + # 2. MultiKueueCluster per worker + mk_cluster_tmpl = Template(templates["multikueue_cluster"]) + fleet_docs = [] + accel_workers = {} + + for wk, wconf in config["worker_clusters"].items(): + wname = wconf["cluster_name"] + pname = wconf["profile_name"] + fleet_docs.append(clean_doc(mk_cluster_tmpl.substitute( + WORKER_NAME=wname, + CLUSTER_PROFILE_NAME=pname + ))) + + for pk, pconf in wconf["pools"].items(): + accel = pconf["accelerator"] + if accel not in accel_workers: + accel_workers[accel] = set() + accel_workers[accel].add(wname) + parts["02-multikueue-fleet.yaml"] = "\n---\n".join(fleet_docs) + "\n" + + # 3. MultiKueueConfig & AdmissionCheck per accelerator cohort + mk_config_tmpl = Template(templates["multikueue_config"]) + adm_check_tmpl = Template(templates["admission_check"]) + cohort_docs = [] + + for accel, workers in sorted(accel_workers.items()): + w_list_str = "\n".join(f" - {w}" for w in sorted(workers)) + cohort_docs.append(clean_doc(mk_config_tmpl.substitute( + ACCELERATOR=accel, + WORKER_LIST=w_list_str + ))) + cohort_docs.append(clean_doc(adm_check_tmpl.substitute( + ACCELERATOR=accel + ))) + parts["03-cohorts.yaml"] = "\n---\n".join(cohort_docs) + "\n" + + # 4. ResourceFlavor per unique accelerator + rf_tmpl = Template(templates["resource_flavor"]) + rf_docs = [] + for accel in sorted(accel_workers.keys()): + rf_docs.append(clean_doc(rf_tmpl.substitute( + ACCELERATOR=accel + ))) + parts["04-resource-flavors.yaml"] = "\n---\n".join(rf_docs) + "\n" + + # 5. ClusterQueue, LocalQueue per pool across all workers + pool_data = {} + for wconf in config["worker_clusters"].values(): + for pk, pconf in wconf["pools"].items(): + if pk not in pool_data: + pool_data[pk] = {"quota": 0, "accelerator": pconf["accelerator"]} + pool_data[pk]["quota"] += pconf["quota"] + + queue_docs = render_queue_group( + templates["queue_group"], pool_data, config["namespace"], + with_admission_checks=True, + ) + parts["05-queues.yaml"] = "\n---\n".join(queue_docs) + "\n" + + return parts + +def generate_worker_parts(config, worker_key, templates): + wconf = config["worker_clusters"][worker_key] + parts = {} + + # 1. Base (Namespace, ClusterSecretStore, ExternalSecret) + base_tmpl = Template(templates["base"]) + parts["01-base.yaml"] = base_tmpl.substitute( + NAMESPACE=config["namespace"], + SECRET_PROJECT=config["secret_project"], + SECRET_ID=config["secret_id"], + CLUSTER_LOCATION=wconf["location"], + CLUSTER_NAME=wconf["cluster_name"] + ).strip() + "\n" + + # 2. ResourceFlavor per unique accelerator in worker + rf_tmpl = Template(templates["resource_flavor"]) + accelerators = sorted({pconf["accelerator"] for pconf in wconf["pools"].values()}) + rf_docs = [clean_doc(rf_tmpl.substitute(ACCELERATOR=accel)) for accel in accelerators] + parts["02-resource-flavors.yaml"] = "\n---\n".join(rf_docs) + "\n" + + # 3. ClusterQueue, LocalQueue per pool profile in worker + queue_docs = render_queue_group( + templates["queue_group"], wconf["pools"], config["namespace"], + with_admission_checks=False, + ) + parts["03-queues.yaml"] = "\n---\n".join(queue_docs) + "\n" + + # 4. What the manager's Kueue controller may do here. Scoped to mirroring + # workloads; it used to hold cluster-admin. + parts["04-multikueue-rbac.yaml"] = Template( + templates["multikueue_rbac_worker"] + ).substitute(PROJECT_ID=config["project_id"]).strip() + "\n" + + # 5. The identity workloads run as, which the cache buckets authorise. + parts["05-workload-sa.yaml"] = Template( + templates["workload_sa"] + ).substitute(NAMESPACE=config["namespace"], PROJECT=wconf["project"]).strip() + "\n" + + # 6. The caches, as claims. Named identically in every region so a workload + # manifest never carries a bucket name; the region binding lives here. + parts["06-cache-volumes.yaml"] = Template( + templates["cache_volumes"] + ).substitute( + NAMESPACE=config["namespace"], + CACHE_BUCKET=wconf["cache_bucket"], + MODELS_BUCKET=wconf["models_bucket"], + ).strip() + "\n" + + return parts + +def load_templates(templates_dir): + templates = {} + templates["base"] = (templates_dir / "base.yaml.tpl").read_text() + templates["multikueue_cluster"] = (templates_dir / "multikueue_cluster.yaml.tpl").read_text() + templates["multikueue_config"] = (templates_dir / "multikueue_config.yaml.tpl").read_text() + templates["admission_check"] = (templates_dir / "admission_check.yaml.tpl").read_text() + templates["resource_flavor"] = (templates_dir / "resource_flavor.yaml.tpl").read_text() + templates["queue_group"] = (templates_dir / "queue_group.yaml.tpl").read_text() + templates["workload_sa"] = (templates_dir / "workload_sa.yaml.tpl").read_text() + templates["multikueue_rbac_worker"] = (templates_dir / "multikueue_rbac_worker.yaml.tpl").read_text() + templates["cache_volumes"] = (templates_dir / "cache_volumes.yaml.tpl").read_text() + return templates + +def main(): + k8s_dir = Path(__file__).resolve().parent.parent + + parser = argparse.ArgumentParser( + description="Render the Kueue manifests for the manager and every worker " + "cluster from prod.auto.tfvars.") + parser.add_argument( + "--out-dir", type=Path, default=k8s_dir / "generated", + # deploy_manifests.sh renders into a scratch directory and diffs it + # against the committed one, which is how it tells stale manifests from + # fresh without writing over what is under review. + help="Where to write the manifests; emptied first. " + "Default: the committed generated/.") + args = parser.parse_args() + + tfvars_file = k8s_dir / "prod.auto.tfvars" + templates_dir = k8s_dir / "kueue" / "templates" + out_dir = args.out_dir + + if not tfvars_file.exists(): + print(f"Error: {tfvars_file} not found.") + sys.exit(1) + + print(f"Reading configuration from: {tfvars_file}") + config = parse_tfvars(tfvars_file) + + print(f"Loading template files from: {templates_dir}") + templates = load_templates(templates_dir) + + if out_dir.exists(): + shutil.rmtree(out_dir) + out_dir.mkdir(parents=True, exist_ok=True) + + # 1. Generate Manager Manifests (Modular directory + Consolidated file) + mgr_parts = generate_manager_parts(config, templates) + mgr_dir = out_dir / "manager" + mgr_dir.mkdir(parents=True, exist_ok=True) + + for filename in sorted(mgr_parts.keys()): + (mgr_dir / filename).write_text(mgr_parts[filename]) + print(f"Generated: manager/{filename}") + + # 2. Generate Worker Manifests (Modular directory + Consolidated file) + for worker_key in config["worker_clusters"].keys(): + wkr_parts = generate_worker_parts(config, worker_key, templates) + wkr_dir = out_dir / f"worker-{worker_key}" + wkr_dir.mkdir(parents=True, exist_ok=True) + + for filename in sorted(wkr_parts.keys()): + (wkr_dir / filename).write_text(wkr_parts[filename]) + print(f"Generated: worker-{worker_key}/{filename}") + + print("\nGeneration Complete! Apply a whole directory with kubectl apply -f, or\nfile by file - the numeric prefixes are the order kubectl uses either way.") + +if __name__ == "__main__": + main() diff --git a/terraform/gcp_old/tpu-inference/k8s/scripts/requirements.txt b/terraform/gcp_old/tpu-inference/k8s/scripts/requirements.txt new file mode 100644 index 00000000..0e68d546 --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/scripts/requirements.txt @@ -0,0 +1,4 @@ +# Dependencies for generate_manifests.py. +# +# python3 -m pip install -r scripts/requirements.txt +python-hcl2>=8,<9 diff --git a/terraform/gcp_old/tpu-inference/k8s/variables.tf b/terraform/gcp_old/tpu-inference/k8s/variables.tf new file mode 100644 index 00000000..07b0eb47 --- /dev/null +++ b/terraform/gcp_old/tpu-inference/k8s/variables.tf @@ -0,0 +1,185 @@ +variable "project_id" { + type = string + description = "The GCP project ID for the manager cluster" +} + +variable "name_prefix" { + type = string + description = "Prefix for created resources" + default = "tpu-ci" +} + +variable "network" { + type = string + description = "Network self link" +} + +variable "manager_region" { + type = string + description = "Region for the manager cluster" +} + +variable "manager_zones" { + type = list(string) + description = "Zones for the manager cluster nodes" +} + +variable "manager_subnetwork" { + type = string + description = "Subnetwork for the manager cluster" +} + +variable "manager_master_ipv4_cidr_block" { + type = string + description = "CIDR block for manager master" +} + +variable "manager_system_machine_type" { + type = string + default = "e2-standard-4" +} + +variable "manager_system_min_nodes" { + type = number + default = 3 +} + +variable "manager_system_max_nodes" { + type = number + default = 10 +} + +variable "worker_clusters" { + type = any + description = "Map of worker cluster definitions" + default = {} +} + +variable "labels" { + type = map(string) + description = "Labels to apply to resources" + default = {} +} + +variable "deletion_protection" { + type = bool + default = false +} + +variable "release_channel" { + type = string + default = "REGULAR" +} + +variable "enable_private_endpoint" { + type = bool + default = false +} + +variable "enable_private_nodes" { + type = bool + default = false +} + +variable "kueue_version" { + type = string + description = "Helm chart version for Kueue" + default = "0.19.0" +} + +variable "jobset_version" { + type = string + description = "Helm chart version for the JobSet operator. Must be identical on manager and workers; MultiKueue mirrors JobSet objects across them. The chart has no 'latest' tag, so this is required." + default = "0.12.0" +} + +variable "buildkite_secret_project" { + type = string + description = "GCP Project ID containing the Buildkite agent token secret" + default = "cloud-ullm-inference-ci-cd" +} + +variable "buildkite_secret_id" { + type = string + description = "GCP Secret Manager secret ID for Buildkite agent token" + default = "buildkite-tpu-ci-agent-dev-token" +} + +variable "buildkite_queue" { + type = string + description = "Buildkite queue name for Agent Stack" + default = "kube" +} + +variable "buildkite_agent_stack_chart_version" { + type = string + description = "Helm chart version for agent-stack-k8s. 0.47.0 is the first release containing the podless-Job cancellation fix (buildkite/agent-stack-k8s#933), so the stock controller image is sufficient from here on." + default = "0.47.0" +} + +variable "buildkite_agent_stack_debug" { + type = bool + description = "Verbose controller logging" + default = true +} + +variable "tpu_job_max_runtime_seconds" { + type = number + description = "Deadline for the agent pod agent-stack-k8s creates per Buildkite job (job-active-deadline-seconds). Must outlast the longest TPU step end to end." + default = 28800 # 8h +} + +variable "tpu_test_max_seconds" { + type = number + description = "How long a TPU test may run once it has chips, applied to the submitted workload as activeDeadlineSeconds. The only deadline that bounds a hung test, because waiting in the queue cannot consume it." + default = 10800 # 3h +} + +variable "tpu_total_max_seconds" { + type = number + description = "How long a step may take end to end: waiting for chips plus running. Every other deadline is derived from this and tpu_test_max_seconds, so these two are the only ones to set. Note that Buildkite contributes no queue time of its own - the agent pod starts almost immediately and then waits - so this whole budget is spent inside the step." + default = 28800 # 8h +} + +variable "create_service_accounts" { + type = bool + description = "Whether to create dedicated GKE node service accounts via Terraform" + default = true +} + +variable "manager_node_service_account" { + type = string + description = "Optional existing Service Account email for manager nodes (defaults to creating one or 'default')" + default = null +} + + + +variable "cache_lifecycle_age_days" { + type = number + default = 30 + description = <<-EOT + Days before a cache object is deleted. + + Longer than the four days the bare-metal bucket uses, because the two are + not the same kind of store. There, a persistent disk on each VM is the real + cache and GCS only distributes it, so an expired object is re-uploaded from + a host that still has it. Here pods are ephemeral and the bucket is the only + copy - and nothing refreshes an object's timestamp when it is read, because + a cache hit is a read. At four days an entry used every day would still be + deleted, and recompiling it costs far more than storing it. + EOT +} + +variable "models_lifecycle_age_days" { + type = number + default = 120 + description = <<-EOT + Days before a cached model object is deleted. + + Longer than the compilation cache, because the two are not alike. A + compilation entry is cheap to recreate and is invalidated by any code change + that alters its HLO; a model is expensive to fetch, comes from a third party + with a request quota, and does not change at all once its commit is pinned. + EOT +}