From 73b9095d83b6eceb5dc7fc54e3c43d776d46a147 Mon Sep 17 00:00:00 2001 From: Will Anderson Date: Tue, 5 May 2026 10:38:22 -0500 Subject: [PATCH] migrate Terraform backend to GCS, remove dead Vault GCE nodes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Switch gcp/ backend from Cloudflare R2 (S3-compatible) to GCS — uses Application Default Credentials automatically, no extra env vars needed. State has been migrated to gs://neuron-785695-terraform-state/gcp/. Remove GCE Vault node resources (vault-node-1/2/3) from vault-nodes.tf and Terraform state. GKE Vault (Workload Identity + vault-unseal SA) has been serving traffic since the NEG cutover; GCE nodes are confirmed dead. Removed from state and config: - google_compute_instance.vault_node (x3) - google_compute_disk.vault_data (x3) - google_compute_instance_group.vault (x3) - google_compute_firewall.vault_iap_ssh / vault_raft / vault_api (GCE-targeted) - google_storage_bucket_object.vault_startup - google_storage_bucket_iam_member.vault_node_bucket_read - google_project_iam_member.vault_node_{log_writer,metric_writer,compute_viewer} Kept: KMS, vault-unseal SA, WI binding, NEG data sources, GKE firewall, global LB (address, cert, backend service, url maps, proxies, fwd rules). --- servers/gcp/main.tf | 14 +-- servers/gcp/vault-nodes.tf | 240 ++----------------------------------- 2 files changed, 10 insertions(+), 244 deletions(-) diff --git a/servers/gcp/main.tf b/servers/gcp/main.tf index 2069a5e..b210773 100644 --- a/servers/gcp/main.tf +++ b/servers/gcp/main.tf @@ -20,17 +20,9 @@ terraform { } } - backend "s3" { - bucket = "legion-terraform-state" - key = "gcp/terraform.tfstate" - region = "auto" - endpoint = "https://651161e0a3d321561b4c90b5bcd5f15b.r2.cloudflarestorage.com" - - # R2 is S3-compatible but not AWS — skip AWS-specific checks - skip_credentials_validation = true - skip_metadata_api_check = true - skip_region_validation = true - force_path_style = true + backend "gcs" { + bucket = "neuron-785695-terraform-state" + prefix = "gcp" } } diff --git a/servers/gcp/vault-nodes.tf b/servers/gcp/vault-nodes.tf index 0cd3583..ca63856 100644 --- a/servers/gcp/vault-nodes.tf +++ b/servers/gcp/vault-nodes.tf @@ -1,7 +1,7 @@ # ── Vault LB — GCP Global HTTPS LB ─────────────────────────────────────────── # # LB frontend is unchanged: forwarding rule → target HTTPS proxy → url map → backend. -# Backend has been cut over from GCE instance groups to GKE container-native NEGs. +# Backend uses GKE container-native NEGs (cut over from GCE instance groups). # # Architecture (GKE): # - Vault Helm chart runs 3 pods in namespace: vault on GKE neuron-platform cluster @@ -10,23 +10,10 @@ # - NEG backends replace GCE instance group backends in the Global HTTPS LB # - DNS unchanged: vault.neuralplatform.ai → 34.54.164.21 (same GCP Global LB IP) # -# GCE nodes (vault-node-1/2/3) and their instance groups are retained in this file -# but no longer serve traffic. Remove them after validating GKE Vault is stable. -# -# NEG names: confirm with: -# gcloud compute network-endpoint-groups list \ -# --filter="name~k8s1-f08a4c22" --project neuron-785695 --format="table(name,zone)" -# Then fill in local.vault_neg_names below and run terraform apply. +# GCE nodes (vault-node-1/2/3) and their instance groups have been removed. +# GKE Vault (via Workload Identity + vault-unseal SA) is the sole Vault backend. locals { - vault_version = "1.19.2" - - vault_nodes = { - "vault-node-1" = { zone = "us-central1-a", node_id = "vault-node-1" } - "vault-node-2" = { zone = "us-central1-b", node_id = "vault-node-2" } - "vault-node-3" = { zone = "us-central1-c", node_id = "vault-node-3" } - } - # Container-native NEG names created by GKE from the Vault Service annotation: # cloud.google.com/neg: '{"exposed_ports":{"8200":{}}}' # GKE creates one NEG per zone where Vault pods are scheduled. @@ -40,7 +27,6 @@ locals { # ── GKE container-native NEG data sources ──────────────────────────────────── # GKE auto-creates these when the Vault Service annotation is applied. -# Populate local.vault_neg_names above with the actual NEG names, then apply. data "google_compute_network_endpoint_group" "vault_gke" { for_each = local.vault_neg_names @@ -49,193 +35,8 @@ data "google_compute_network_endpoint_group" "vault_gke" { project = var.project_id } -# ── Service Account for Vault nodes ────────────────────────────────────────── -# Reuses the vault-unseal SA from vault-kms.tf for KMS access. -# Additional roles: logging + metrics from ops agent. - -resource "google_project_iam_member" "vault_node_log_writer" { - project = var.project_id - role = "roles/logging.logWriter" - member = "serviceAccount:${google_service_account.vault_unseal.email}" -} - -resource "google_project_iam_member" "vault_node_metric_writer" { - project = var.project_id - role = "roles/monitoring.metricWriter" - member = "serviceAccount:${google_service_account.vault_unseal.email}" -} - -# Allow Vault nodes to read their own instance metadata (needed for GCP auth method later) -resource "google_project_iam_member" "vault_node_compute_viewer" { - project = var.project_id - role = "roles/compute.viewer" - member = "serviceAccount:${google_service_account.vault_unseal.email}" -} - -# ── Startup script staged in GCS ───────────────────────────────────────────── -# Stored in the existing runner-assets bucket (reuse infrastructure). -# The script installs Vault, writes the config, and starts the systemd unit. - -resource "google_storage_bucket_object" "vault_startup" { - name = "vault/startup.sh" - bucket = google_storage_bucket.runner_assets.name - source = "${path.module}/vault/startup.sh" - - metadata = { - sha256 = filesha256("${path.module}/vault/startup.sh") - } -} - -resource "google_storage_bucket_iam_member" "vault_node_bucket_read" { - bucket = google_storage_bucket.runner_assets.name - role = "roles/storage.objectViewer" - member = "serviceAccount:${google_service_account.vault_unseal.email}" -} - -# ── Vault node VMs ──────────────────────────────────────────────────────────── - -resource "google_compute_instance" "vault_node" { - for_each = local.vault_nodes - name = each.key - machine_type = "e2-small" - zone = each.value.zone - project = var.project_id - - tags = ["vault-node", "allow-iap-ssh"] - - boot_disk { - initialize_params { - image = "projects/debian-cloud/global/images/family/debian-12" - size = 20 - type = "pd-balanced" - } - } - - # Separate persistent disk for Raft data — survives VM recreation - attached_disk { - source = google_compute_disk.vault_data[each.key].self_link - device_name = "vault-data" - mode = "READ_WRITE" - } - - network_interface { - network = "default" - # No external IP — accessed via IAP SSH and the internal LB - # Vault API is published externally via the regional LB below - access_config {} - } - - service_account { - email = google_service_account.vault_unseal.email - scopes = ["cloud-platform"] - } - - metadata = { - # Pull and execute the real startup script from GCS - startup-script = <<-EOT - #!/usr/bin/env bash - set -euxo pipefail - apt-get update -y - apt-get install -y curl ca-certificates apt-transport-https gnupg google-cloud-cli - gsutil cat gs://${google_storage_bucket.runner_assets.name}/${google_storage_bucket_object.vault_startup.name} \ - > /tmp/vault-startup.sh - chmod +x /tmp/vault-startup.sh - VAULT_NODE_ID="${each.value.node_id}" \ - VAULT_VERSION="${local.vault_version}" \ - /tmp/vault-startup.sh - EOT - - enable-oslogin = "TRUE" - } - - allow_stopping_for_update = true - - depends_on = [ - google_compute_disk.vault_data, - google_storage_bucket_object.vault_startup, - google_storage_bucket_iam_member.vault_node_bucket_read, - ] -} - -# ── Persistent data disks ───────────────────────────────────────────────────── -# 10 GiB per node. Kept separate from the boot disk so Raft data -# survives a full VM deletion and recreation. - -resource "google_compute_disk" "vault_data" { - for_each = local.vault_nodes - name = "${each.key}-data" - zone = each.value.zone - project = var.project_id - type = "pd-ssd" - size = 10 - - labels = { - managed-by = "terraform" - service = "vault" - node = each.key - } - - lifecycle { - prevent_destroy = true - } -} - -# ── Firewall rules ──────────────────────────────────────────────────────────── - -# IAP SSH — ops access without a public SSH port -resource "google_compute_firewall" "vault_iap_ssh" { - name = "vault-iap-ssh" - network = "default" - project = var.project_id - direction = "INGRESS" - priority = 1000 - - allow { - protocol = "tcp" - ports = ["22"] - } - - source_ranges = ["35.235.240.0/20"] # GCP IAP CIDR - target_tags = ["vault-node"] -} - -# Raft cluster traffic — node-to-node port 8201 (internal only) -resource "google_compute_firewall" "vault_raft" { - name = "vault-raft-internal" - network = "default" - project = var.project_id - direction = "INGRESS" - priority = 1000 - - allow { - protocol = "tcp" - ports = ["8201"] - } - - source_tags = ["vault-node"] - target_tags = ["vault-node"] -} - -# Vault API — allow from the GCP health check ranges and the LB -resource "google_compute_firewall" "vault_api" { - name = "vault-api-from-lb" - network = "default" - project = var.project_id - direction = "INGRESS" - priority = 1000 - - allow { - protocol = "tcp" - ports = ["8200"] - } - - # GCP health check ranges (130.211.0.0/22, 35.191.0.0/16) - # and RFC1918 for any internal service access - source_ranges = ["130.211.0.0/22", "35.191.0.0/16", "10.0.0.0/8", "172.16.0.0/12", "192.168.0.0/16"] - target_tags = ["vault-node"] -} - -# GKE Vault pods — allow GCP health check probers to reach pod IPs on port 8200. +# ── Firewall — GKE health checks ───────────────────────────────────────────── +# Allow GCP health check probers to reach pod IPs on port 8200. # Container-native NEGs direct health checks to pod IPs (not node IPs). # GKE Autopilot pod CIDR: 10.45.128.0/22 (from cluster ip_allocation_policy). # Without this rule, GCP LB health checks fail → "no healthy upstream". @@ -257,29 +58,11 @@ resource "google_compute_firewall" "vault_api_gke" { source_ranges = ["130.211.0.0/22", "35.191.0.0/16"] } -# ── Instance Groups (one unmanaged group per zone for the LB) ───────────────── - -resource "google_compute_instance_group" "vault" { - for_each = local.vault_nodes - name = "vault-${each.key}" - zone = each.value.zone - project = var.project_id - instances = [google_compute_instance.vault_node[each.key].self_link] - - named_port { - name = "vault-api" - port = 8200 - } -} - # ── Regional HTTPS LB for vault.neuralplatform.ai ───────────────────────────── -# We use a global external HTTPS LB (same scheme as the prod marketing LB) -# so we can attach a Google-managed cert for vault.neuralplatform.ai. +# Global external HTTPS LB with a Google-managed cert for vault.neuralplatform.ai. # # TLS is terminated at the LB. Vault listens on plain 8200 internally. -# The Cloudflare DNS A record for vault.neuralplatform.ai (neuralplatform zone) -# must point to vault_lb_ip output below — add it in Cloudflare dashboard -# or in the legion Terraform if you bring that zone under TF management. +# Cloudflare DNS A record for vault.neuralplatform.ai → vault_lb_ip output below. resource "google_compute_global_address" "vault" { name = "vault-ip" @@ -318,8 +101,6 @@ resource "google_compute_backend_service" "vault" { protocol = "HTTP" # Vault serves plain HTTP; TLS terminates at the LB timeout_sec = 30 - # port_name not used with NEG backends (NEGs have explicit port in the service annotation) - health_checks = [google_compute_health_check.vault.self_link] # GKE container-native NEG backends — one per zone where Vault pods are scheduled. @@ -393,10 +174,3 @@ output "vault_lb_ip" { description = "Global IP for vault.neuralplatform.ai — set as DNS A record in Cloudflare (neuralplatform.ai zone)" value = google_compute_global_address.vault.address } - -output "vault_node_zones" { - description = "Zone placement of each Vault node" - value = { - for k, v in local.vault_nodes : k => v.zone - } -}