diff --git a/.gitignore b/.gitignore index 0059dc7..d1c794f 100644 --- a/.gitignore +++ b/.gitignore @@ -37,3 +37,4 @@ override.tf.json # Ignore CLI configuration files .terraformrc terraform.rc +*.local.md diff --git a/MIGRATION.md b/MIGRATION.md new file mode 100644 index 0000000..c1b0b2e --- /dev/null +++ b/MIGRATION.md @@ -0,0 +1,99 @@ +# Brownfield migration to v6.0.0 (DND-1573 / DND-1257) + +For clusters still on a v1.20.x or v2.1.x module version. Greenfield clusters and +anything already on v5.x/v6.x do not need this — bump straight to `v6.0.0`. + +## Why a temporary tag + +The v5 "Infra-Only + ArgoCD" refactor deleted the module's in-cluster and +eks-blueprints-addons resources; GitOps and native EKS add-ons own them now. +A cluster upgrading from v1/v2 still carries those objects in state, and because +the module also dropped its kubernetes/helm provider *config blocks*, `plan` fails +with `Provider configuration not present` for every one of them. + +`v6.0.1-migration-4` is `v6.0.0` plus: + +- `modules/comet_eks/removed.tf` — 13 `removed{}` blocks, all `destroy = false` +- `kubernetes` + `helm` back in both `versions.tf` files (requirements only, no + provider config blocks) + +It is consumed for exactly ONE apply per cluster, then the wrapper moves to the +permanent `v6.0.0`. + +`v6.0.1-migration` (the unsuffixed tag) is abandoned — it shipped `compute_config` +unconditionally, which EKS rejects on any cluster that never had Auto Mode. Do not +use it. + +Do not use `v5.6.2-migration` for this. It predates DND-875 and lacks +`rds_auto_minor_version_upgrade`, `rds_parameter_group_family`, +`rds_use_proxy_endpoint` and `rds_proxy_ack_no_iam_auth` — envs that pass any of +them fail on an unsupported argument before the removed blocks run. + +## Stage 1 — drop the orphans + +Wrapper: + +```hcl +module "comet" { + source = "github.com/comet-ml/terraform-aws-comet-stsaas?ref=v6.0.1-migration-4" + + providers = { + aws = aws + kubernetes = kubernetes + helm = helm + } + ... +} +``` + +The `providers` map is required: the orphans bind to +`module.comet.provider["...{aws,kubernetes,helm}"]`, and this populates that with +the wrapper's real assume_role/region. An empty `provider "aws" {}` inside the +module would hijack credentials instead (AccessDenied). + +Also in the same PR: + +- **Root-level orphans** — porsche and si carry `kubernetes_cluster_role.agentro_extras`, + `kubernetes_cluster_role_binding.{agentro_extras,agentro_view}` and + `kubernetes_role{,_binding}.agentro_portforward` at the STATE ROOT. They are not + module-addressed, so add `removed{}` blocks for them in the WRAPPER. +- **`mysql_vpn` import** — v6/DND-1522 creates the VPN→MySQL rule unconditionally, + but every env already has it out of state from the DND-752 era. Without an + `import{}` the apply fails `InvalidPermission.Duplicate`. See the rule-id table in + the rollout plan. +- **Drop deleted vars** — `enable_argocd_management_eks_access`, + `enable_vpn_eks_api_access`, `enable_ci_runners_eks_api_access`, + `enable_vpn_redis_access`, `enable_monitoring_setup`, `monitoring_namespace`, + `eks_create_comet_generic_storage_class`, `enable_redis_insights_ns`. + +Apply. The objects leave state; live infrastructure is untouched. + +## Stage 1.5 — GitOps adoption + +comet-infra (ArgoCD) / native EKS add-ons must own the ALB controller, cert-manager, +external-dns, the gp3 StorageClass and the monitoring namespace/secret. Coordinate +this with Stage 1 — it is the real risk in the sequence, not the terraform. + +## Stage 2 — land on the permanent tag + +```hcl +source = "github.com/comet-ml/terraform-aws-comet-stsaas?ref=v6.0.0" +``` + +Drop the `providers` map and `helm` from the wrapper's `required_providers`. **Keep +`kubernetes`** if the wrapper's own modules use it — `agentro_k8s_rbac` and +`automation_smoke_rbac` own kubernetes resources outside `module.comet`, and dropping +the provider strands them. + +Delete the wrapper's one-shot `imports.tf` and any Stage 1 `removed{}` blocks once +applied. + +Expected plan: no infrastructure change. Stage 1 already landed on the v6 surface. + +## Order + +Canary **waystar** — its orphan set is exactly the six blocks proven against bayer, +with no root-level extras and no `aws_cloudwatch_metrics`. Then zoox, si, porsche. +Group C (v1.20.x: circuit, circuit-dev, eonnext, fetch, mercedesamgf1, netflix) +after Group B is complete; those additionally hit the Karpenter-Helm → EKS Auto Mode +migration, which is its own piece of work. diff --git a/modules/comet_eks/main.tf b/modules/comet_eks/main.tf index 9a85785..5f63cc9 100644 --- a/modules/comet_eks/main.tf +++ b/modules/comet_eks/main.tf @@ -244,14 +244,21 @@ module "eks" { # EKS Auto Mode. When enabled, the control plane provisions nodes via the # built-in node pools and the upstream module auto-creates/wires the Auto Mode - # node IAM role (so node_role_arn is intentionally omitted). The block is - # always sent — enabled = false explicitly disables Auto Mode so a cluster that - # previously had it on can be turned back off (a bare null would omit the - # argument and leave the last-applied config in place). - compute_config = { - enabled = var.enable_auto_mode - node_pools = var.enable_auto_mode ? var.auto_mode_node_pools : [] - } + # node IAM role (so node_role_arn is intentionally omitted). + # + # null when disabled, NOT { enabled = false }. Sending an explicit disable to a + # cluster that never had Auto Mode makes EKS reject the whole UpdateClusterConfig + # with "Cannot modify EKS Auto Mode configuration. Auto Mode is not enabled on + # this cluster." That failed waystar's v6 apply (#2213) and would break every env + # where eks_enable_auto_mode is unset — 10 of 13. + # + # Turning Auto Mode back OFF on a cluster that has it therefore needs a + # deliberate one-off: set enabled = false here for that apply, or disable it out + # of band. That is the rarer operation, and unlike this failure it is not silent. + compute_config = var.enable_auto_mode ? { + enabled = true + node_pools = var.auto_mode_node_pools + } : null # Bake the Karpenter discovery tag directly into the node SG so it is never # dropped when Terraform modifies the security group during subsequent applies. @@ -312,6 +319,14 @@ module "eks" { } : {}, { configuration_values = local.coredns_config + # OVERWRITE so a brownfield cluster (upgrading from the eks_blueprints_addons + # Helm cert-manager) lets the native add-on ADOPT the existing Helm-owned + # objects (SAs/CRDs/Deployments/webhooks) instead of failing with + # "ConfigurationConflict … resolve conflicts mode" (default NONE). The add-on + # relabels them managed-by=EKS in place — cert-manager keeps running. Harmless + # on greenfield (nothing to conflict). DND-1573. + resolve_conflicts_on_create = "OVERWRITE" + resolve_conflicts_on_update = "OVERWRITE" } ) } : {}, @@ -328,6 +343,13 @@ module "eks" { role_arn = aws_iam_role.external_dns[0].arn service_account = "external-dns" }] + # OVERWRITE so a brownfield cluster adopts any external-dns k8s objects left by + # the old eks_blueprints_addons Helm release rather than conflict-failing. NOTE: + # a pre-existing Pod Identity association for external-dns:external-dns must be + # deleted first (ResourceInUseException otherwise) — resolve_conflicts does not + # cover Pod Identity associations. DND-1573. + resolve_conflicts_on_create = "OVERWRITE" + resolve_conflicts_on_update = "OVERWRITE" }, var.eks_external_dns_addon_version != null ? { addon_version = var.eks_external_dns_addon_version diff --git a/modules/comet_eks/removed.tf b/modules/comet_eks/removed.tf new file mode 100644 index 0000000..fc0d29a --- /dev/null +++ b/modules/comet_eks/removed.tf @@ -0,0 +1,149 @@ +# Brownfield state migration (v1.x/v2.x → v6.x) — DND-1573 / DND-1257. +# +# ⚠️ THIS FILE MAKES THIS A TEMPORARY "MIGRATION" MODULE VERSION. It is meant to be +# consumed for exactly ONE apply per brownfield cluster, then the wrapper bumps to +# the permanent v6.0.0 tag WITHOUT this file. See MIGRATION.md at the repo root. +# +# Supersedes v5.6.2-migration, which cannot be used by the remaining brownfield envs: +# it predates DND-875, so it lacks rds_auto_minor_version_upgrade (which porsche, si, +# waystar and zoox all pass) plus rds_parameter_group_family, rds_use_proxy_endpoint +# and rds_proxy_ack_no_iam_auth. Those envs would fail on an unsupported argument +# before the removed{} blocks below ever ran. +# +# Cut from v6.0.0 rather than v5.7.0 so Stage 1 lands directly on the v6 surface — +# Stage 2 is then only "drop the providers map", with no second behaviour change. +# +# The v5 "Infra-Only + ArgoCD" refactor DELETED these in-cluster / eks-blueprints-addons +# resources from the module (they are now owned by GitOps / native EKS add-ons). Clusters +# upgrading from a v1/v2 module version still carry them in state, and because the module +# root's kubernetes/helm/aws provider *config blocks* were also removed (child-module +# design), `terraform plan` fails with "Provider configuration not present" for all of +# them until the objects leave state. +# +# HOW THIS WORKS (proven against bayer, DND-1573 / #2205): +# - These `removed` blocks drop the objects from state WITHOUT destroying the live +# resources (`lifecycle { destroy = false }`) — the live workloads are adopted by +# comet-infra (ArgoCD) / native EKS add-ons as part of the coordinated cutover. +# - The orphans bind in state to `module.comet.provider["...{aws,kubernetes,helm}"]`. +# To satisfy that binding the module declares the kubernetes+helm *requirements* +# (versions.tf) — but declares NO provider config blocks. Instead the WRAPPER passes +# its fully-configured providers in: +# module "comet" { +# providers = { aws = aws, kubernetes = kubernetes, helm = helm } +# } +# This populates module.comet.provider[...] with the wrapper's real assume_role/region +# (an empty `provider "aws" {}` here would instead HIJACK credentials → AccessDenied). +# - Greenfield v6 clusters never had these resources, so the blocks are a harmless no-op. +# +# NOT COVERED HERE — root-level orphans. porsche and si additionally carry +# kubernetes_cluster_role.agentro_extras, kubernetes_cluster_role_binding.agentro_extras, +# kubernetes_cluster_role_binding.agentro_view, kubernetes_role.agentro_portforward and +# kubernetes_role_binding.agentro_portforward at the STATE ROOT, not under module.comet. +# They are not module-addressed, so they must be removed{} in the WRAPPER, not here. +# +# NOTE: do NOT `removed` the whole `module.eks_blueprints_addons` — its `aws_eks_addon.this[*]` +# children (coredns/kube-proxy/vpc-cni/metrics-server/ebs-csi) are relocated via the `moved` +# blocks in main.tf. Only the Helm-based sub-modules below were deleted. + +# --- eks-blueprints-addons Helm sub-modules --- +# Removing the sub-module call drops all of its resources (aws_iam_policy/role/attachment + +# helm_release). ALB controller → comet-infra (ArgoCD); cert-manager + external-dns → native +# EKS managed add-ons (see main.tf module.eks.addons). +removed { + from = module.eks_blueprints_addons.module.aws_load_balancer_controller + lifecycle { + destroy = false + } +} + +removed { + from = module.eks_blueprints_addons.module.cert_manager + lifecycle { + destroy = false + } +} + +removed { + from = module.eks_blueprints_addons.module.external_dns + lifecycle { + destroy = false + } +} + +# zoox only — enabled there via eks_aws_cloudwatch_metrics, which v6 removed. Harmless +# no-op on every other cluster. +removed { + from = module.eks_blueprints_addons.module.aws_cloudwatch_metrics + lifecycle { + destroy = false + } +} + +# --- In-cluster resources now owned by comet-infra (ArgoCD) --- +# gp3 StorageClass + monitoring namespace/secret moved to the comet-infra Helm chart. +removed { + from = kubernetes_storage_class.gp3 + lifecycle { + destroy = false + } +} + +removed { + from = kubernetes_namespace.monitoring + lifecycle { + destroy = false + } +} + +removed { + from = kubernetes_secret.monitoring + lifecycle { + destroy = false + } +} + +# --- v1.20.x-only in-cluster resources (Group C) --- +# Present on circuit, circuit-dev, eonnext, fetch, mercedesamgf1 and netflix; a no-op on +# the v2.1.x envs. Karpenter-via-Helm is superseded by EKS Auto Mode, external-secrets and +# the comet-generic StorageClass by comet-infra GitOps. +removed { + from = helm_release.karpenter_stsaas + lifecycle { + destroy = false + } +} + +removed { + from = helm_release.external_secrets + lifecycle { + destroy = false + } +} + +removed { + from = helm_release.external_secrets_crds + lifecycle { + destroy = false + } +} + +removed { + from = kubernetes_storage_class.comet_generic + lifecycle { + destroy = false + } +} + +removed { + from = kubernetes_annotations.admin_ns_node_selector + lifecycle { + destroy = false + } +} + +removed { + from = kubernetes_annotations.app_ns_node_selector + lifecycle { + destroy = false + } +} diff --git a/modules/comet_eks/versions.tf b/modules/comet_eks/versions.tf index 3060ce7..f80d535 100644 --- a/modules/comet_eks/versions.tf +++ b/modules/comet_eks/versions.tf @@ -11,5 +11,19 @@ terraform { source = "hashicorp/time" version = ">= 0.9" } + # kubernetes + helm are declared ONLY so brownfield clusters (upgrading from a + # v1/v2 module version) can associate their leftover in-cluster / helm_release + # resources with a provider long enough for the `removed` blocks in removed.tf to + # drop them from state (destroy = false). v6 no longer CREATES any kubernetes/helm + # resource — greenfield clusters never instantiate these providers. Removed again + # in the permanent v6.0.0 tag (DND-1573 / DND-1257 brownfield migration). + kubernetes = { + source = "hashicorp/kubernetes" + version = ">= 2.0" + } + helm = { + source = "hashicorp/helm" + version = ">= 2.0" + } } } diff --git a/versions.tf b/versions.tf index cd3ae62..60cf9df 100644 --- a/versions.tf +++ b/versions.tf @@ -13,5 +13,19 @@ terraform { source = "hashicorp/random" version = ">= 3.0" } + # kubernetes + helm are declared ONLY so brownfield clusters (upgrading from a + # v1/v2 module version) can associate their leftover in-cluster / helm_release + # resources with a provider long enough for the `removed` blocks in removed.tf to + # drop them from state (destroy = false). v6 no longer CREATES any kubernetes/helm + # resource — greenfield clusters never instantiate these providers. Removed again + # in the permanent v6.0.0 tag (DND-1573 / DND-1257 brownfield migration). + kubernetes = { + source = "hashicorp/kubernetes" + version = ">= 2.0" + } + helm = { + source = "hashicorp/helm" + version = ">= 2.0" + } } }