From cb7f99b5ac35dc87d9de731771ef8c1b8e93c6b0 Mon Sep 17 00:00:00 2001 From: Alex Bezpalko Date: Tue, 11 Aug 2026 17:56:15 +0200 Subject: [PATCH] feat: allow EKS Auto Mode nodes to reach ElastiCache/RDS/RDS-proxy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit EKS Auto Mode nodes attach the cluster PRIMARY security group (module.eks.cluster_primary_security_group_id), while managed node groups use the module's node shared SG (node_security_group_id). The data-layer SG ingress rules only referenced the managed node SG, so any pod on an Auto Mode node was blocked from Redis (6379) and MySQL (3306). The module already handles this SG split for node<->node coexistence (comet_eks/main.tf) but never extended it to the data layer. Symptom: on a cluster with an Auto Mode nodepool (e.g. stsaasuat's arm64 `multiarch` pool), a pod that lands there can't reach ElastiCache — e.g. the mysql-db-migration sync hook's waitForResources loops "redis timeout" forever and wedges the ArgoCD sync. - comet_eks: expose `cluster_primary_security_group_id` output (the SG Auto Mode nodes use). - comet_elasticache / comet_rds: add `*_auto_mode_allow_from_sg` var (string, default null) + a second ingress rule (Redis 6379 / MySQL 3306) created only when it's set. Mirrors the module's existing coexistence-rule pattern. - rds_proxy: allowed_sg_ids already list/for_each — root now passes both the managed node SG and (when Auto Mode is enabled) the cluster primary SG. - root main.tf: wire all three, guarded by `var.enable_eks && var.eks_enable_auto_mode`. Default null / Auto-Mode-off => ZERO diff for existing clusters. Suggested release: v5.5.0. Co-Authored-By: Claude Opus 4.8 --- main.tf | 25 ++++++++++++++++++------- modules/comet_eks/outputs.tf | 7 ++++++- modules/comet_elasticache/main.tf | 14 ++++++++++++++ modules/comet_elasticache/variables.tf | 6 ++++++ modules/comet_rds/main.tf | 14 ++++++++++++++ modules/comet_rds/variables.tf | 6 ++++++ 6 files changed, 64 insertions(+), 8 deletions(-) diff --git a/main.tf b/main.tf index 5c3ee20..64bddad 100644 --- a/main.tf +++ b/main.tf @@ -378,6 +378,9 @@ module "comet_elasticache" { elasticache_allow_from_sg = var.enable_ec2 ? module.comet_ec2[0].comet_ec2_sg_id : ( var.enable_eks ? module.comet_eks[0].nodegroup_sg_id : ( var.elasticache_allow_from_sg)) + # EKS Auto Mode nodes attach the cluster primary SG (distinct from the managed node SG + # above), so grant them Redis access too when Auto Mode is enabled. + elasticache_auto_mode_allow_from_sg = var.enable_eks && var.eks_enable_auto_mode ? module.comet_eks[0].cluster_primary_security_group_id : null elasticache_engine = var.elasticache_engine elasticache_engine_version = var.elasticache_engine_version elasticache_instance_type = var.elasticache_instance_type @@ -408,11 +411,14 @@ module "comet_rds" { rds_allow_from_sg = var.enable_ec2 ? module.comet_ec2[0].comet_ec2_sg_id : ( var.enable_eks ? module.comet_eks[0].nodegroup_sg_id : ( var.rds_allow_from_sg)) - rds_engine = var.rds_engine - rds_engine_version = var.rds_engine_version - rds_instance_type = var.rds_instance_type - rds_instance_count = var.rds_instance_count - rds_storage_encrypted = var.rds_storage_encrypted + # EKS Auto Mode nodes attach the cluster primary SG (distinct from the managed node SG + # above), so grant them MySQL access too when Auto Mode is enabled. + rds_auto_mode_allow_from_sg = var.enable_eks && var.eks_enable_auto_mode ? module.comet_eks[0].cluster_primary_security_group_id : null + rds_engine = var.rds_engine + rds_engine_version = var.rds_engine_version + rds_instance_type = var.rds_instance_type + rds_instance_count = var.rds_instance_count + rds_storage_encrypted = var.rds_storage_encrypted # Aurora Serverless v2 (optional) rds_serverless_v2_enabled = var.rds_serverless_v2_enabled @@ -460,8 +466,13 @@ module "comet_rds_proxy" { vpc_id = var.enable_vpc ? module.comet_vpc[0].vpc_id : var.comet_vpc_id subnet_ids = var.enable_vpc ? module.comet_vpc[0].private_subnets : var.comet_private_subnets - allowed_sg_ids = var.enable_eks ? [module.comet_eks[0].nodegroup_sg_id] : var.rds_proxy_allowed_sg_ids - allowed_cidrs = var.rds_proxy_allowed_cidrs + # Managed node SG + (when Auto Mode is enabled) the cluster primary SG that Auto Mode + # nodes attach, so pods on either node type can reach the RDS proxy. + allowed_sg_ids = var.enable_eks ? concat( + [module.comet_eks[0].nodegroup_sg_id], + var.eks_enable_auto_mode ? [module.comet_eks[0].cluster_primary_security_group_id] : [], + ) : var.rds_proxy_allowed_sg_ids + allowed_cidrs = var.rds_proxy_allowed_cidrs mysql_cluster_id = module.comet_rds[0].mysql_cluster_id mysql_sg_id = module.comet_rds[0].mysql_sg_id diff --git a/modules/comet_eks/outputs.tf b/modules/comet_eks/outputs.tf index 6adef09..631c7e3 100644 --- a/modules/comet_eks/outputs.tf +++ b/modules/comet_eks/outputs.tf @@ -14,10 +14,15 @@ output "cluster_certificate_authority_data" { } output "nodegroup_sg_id" { - description = "ID of the node shared security group" + description = "ID of the node shared security group (managed node groups attach this)" value = module.eks.node_security_group_id } +output "cluster_primary_security_group_id" { + description = "EKS-managed cluster primary security group. EKS Auto Mode nodes attach this SG (managed node groups use nodegroup_sg_id instead) — reference it where Auto Mode nodes need network access (e.g. data-layer SG ingress)." + value = module.eks.cluster_primary_security_group_id +} + output "cluster_autoscaler_irsa_role_arn" { description = "ARN of the Cluster Autoscaler IRSA role (serviceAccountName=kube-system/cluster-autoscaler). Wire this into the cluster-autoscaler Helm values as the service account annotation." value = var.eks_enable_cluster_autoscaler ? module.cluster_autoscaler_irsa_role[0].iam_role_arn : null diff --git a/modules/comet_elasticache/main.tf b/modules/comet_elasticache/main.tf index 85b8bef..351bfc5 100644 --- a/modules/comet_elasticache/main.tf +++ b/modules/comet_elasticache/main.tf @@ -65,6 +65,20 @@ resource "aws_vpc_security_group_ingress_rule" "redis_port_inbound_rule" { ip_protocol = "tcp" referenced_security_group_id = var.elasticache_allow_from_sg } + +# EKS Auto Mode nodes attach the cluster primary SG (not the managed node SG that +# elasticache_allow_from_sg references), so they need their own ingress rule to reach +# Redis. Only created when the Auto Mode SG is passed in. +resource "aws_vpc_security_group_ingress_rule" "redis_port_inbound_auto_mode" { + count = var.elasticache_auto_mode_allow_from_sg != null ? 1 : 0 + + security_group_id = aws_security_group.redis_inbound_sg.id + from_port = local.redis_port + to_port = local.redis_port + ip_protocol = "tcp" + referenced_security_group_id = var.elasticache_auto_mode_allow_from_sg + description = "Redis from EKS Auto Mode nodes (cluster primary SG)" +} # VPN ingress to Redis (DND-752) — gated by enable_vpn_redis_access. Allows # operators on the VPN to connect to Redis via kubectl port-forward through # the cluster's Redis SG. diff --git a/modules/comet_elasticache/variables.tf b/modules/comet_elasticache/variables.tf index d30e790..9aea48f 100644 --- a/modules/comet_elasticache/variables.tf +++ b/modules/comet_elasticache/variables.tf @@ -18,6 +18,12 @@ variable "elasticache_allow_from_sg" { type = string } +variable "elasticache_auto_mode_allow_from_sg" { + description = "Additional security group allowed to reach ElastiCache — the EKS Auto Mode cluster primary SG. Auto Mode nodes attach a different SG than managed node groups, so without this a pod on an Auto Mode node cannot reach Redis. Null (default) creates no extra rule." + type = string + default = null +} + variable "elasticache_engine" { description = "Engine type for Elasticache cluster" type = string diff --git a/modules/comet_rds/main.tf b/modules/comet_rds/main.tf index d10a3b0..9f08d32 100644 --- a/modules/comet_rds/main.tf +++ b/modules/comet_rds/main.tf @@ -254,3 +254,17 @@ resource "aws_vpc_security_group_ingress_rule" "mysql_port_inbound_ec2" { ip_protocol = "tcp" referenced_security_group_id = var.rds_allow_from_sg } + +# EKS Auto Mode nodes attach the cluster primary SG (not the managed node SG that +# rds_allow_from_sg references), so they need their own ingress rule to reach MySQL. +# Only created when the Auto Mode SG is passed in. +resource "aws_vpc_security_group_ingress_rule" "mysql_port_inbound_auto_mode" { + count = var.rds_auto_mode_allow_from_sg != null ? 1 : 0 + + security_group_id = aws_security_group.mysql_sg.id + from_port = local.mysql_port + to_port = local.mysql_port + ip_protocol = "tcp" + referenced_security_group_id = var.rds_auto_mode_allow_from_sg + description = "MySQL from EKS Auto Mode nodes (cluster primary SG)" +} diff --git a/modules/comet_rds/variables.tf b/modules/comet_rds/variables.tf index c1976d0..20e3d31 100644 --- a/modules/comet_rds/variables.tf +++ b/modules/comet_rds/variables.tf @@ -35,6 +35,12 @@ variable "rds_allow_from_sg" { type = string } +variable "rds_auto_mode_allow_from_sg" { + description = "Additional security group allowed to reach RDS — the EKS Auto Mode cluster primary SG. Auto Mode nodes attach a different SG than managed node groups, so without this a pod on an Auto Mode node cannot reach MySQL. Null (default) creates no extra rule." + type = string + default = null +} + variable "rds_engine" { description = "Engine type for RDS database" type = string