Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
19 commits
Select commit Hold shift + click to select a range
d34271a
Migrate from EKS Auto Mode to OSS Karpenter
theautoroboto Jul 30, 2026
60b1530
Cleanup: fix docs, revert timeouts, remove validation scripts
theautoroboto Jul 28, 2026
e099a41
Fix silent terraform import failure creating duplicate CodeStar conne…
theautoroboto Jul 29, 2026
e043bbe
Revert "Fix silent terraform import failure creating duplicate CodeSt…
theautoroboto Jul 29, 2026
fdb3f2b
Revert "Cleanup: fix docs, revert timeouts, remove validation scripts"
theautoroboto Jul 29, 2026
87e80de
Revert "Switch platform-api to official Tekton image with rate limiti…
theautoroboto Jul 30, 2026
5fb0e69
Remove DNS troubleshooting debug logging
theautoroboto Jul 30, 2026
5fa624a
Bootstrap: use helm status guard for Karpenter skip-if-deployed
theautoroboto Jul 30, 2026
27b9cff
Bootstrap: restore skip-if-exists guard for NodePool seeding
theautoroboto Jul 30, 2026
14a19d0
Bootstrap: remove Karpenter pre-warm step
theautoroboto Jul 30, 2026
a31660a
Bootstrap: restore skip-if-exists for ArgoCD, remove broken-release r…
theautoroboto Jul 30, 2026
0cafcb2
Bootstrap: remove duplicate CriticalAddonsOnly toleration --set flags
theautoroboto Jul 30, 2026
cf0261a
Bootstrap: clarify HyperShift wait is a CI accommodation
theautoroboto Jul 30, 2026
321118d
ci/e2e: add DNS diagnostics to HCP creation test failure path
theautoroboto Jul 30, 2026
2f595fd
hypershift-install Job: add verbose logging and softer patch error ha…
theautoroboto Jul 30, 2026
faffc4a
Remove k8s validation scripts — not viable from Prow
theautoroboto Jul 30, 2026
7092854
hypershift-install Job: fix CSI Secret chicken-and-egg deadlock
theautoroboto Jul 30, 2026
71dec45
change image
theautoroboto Jul 30, 2026
1dde297
Remove accidentally committed Claude worktree gitlink, ignore .claude…
theautoroboto Jul 30, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -45,3 +45,4 @@ ephemeral-logs*

# Dashboard (generated locally by ./dashboard/fetch-data.sh)
dashboard/data.json
.claude/worktrees/
2 changes: 1 addition & 1 deletion Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -121,7 +121,7 @@ terraform-validate: terraform-init ## Check formatting and validate all Terrafor

# Global values (aws_region, environment, cluster_type) are injected by the
# ApplicationSet at deploy time, so we supply stubs here for linting.
HELM_LINT_SET := --set global.aws_region=us-east-1 --set global.environment=lint --set global.cluster_type=lint
HELM_LINT_SET := --set global.aws_region=us-east-1 --set global.environment=lint --set global.cluster_type=lint --set global.cluster_name=lint
helm-lint: ## Lint all Helm charts
@echo "🔍 Linting Helm charts..."
@failed=false; \
Expand Down
Original file line number Diff line number Diff line change
@@ -1,15 +1,18 @@
apiVersion: eks.amazonaws.com/v1
kind: NodeClass
{{- $clusterName := required "global.cluster_name must be set via ApplicationSet valuesObject" .Values.global.cluster_name -}}
apiVersion: karpenter.k8s.aws/v1
kind: EC2NodeClass
metadata:
name: fips
spec:
role: "{{ .Values.global.cluster_name }}-auto-node-role"
amiSelectorTerms:
- alias: bottlerocket@latest
instanceProfile: {{ $clusterName }}-karpenter-node-role
subnetSelectorTerms:
- tags:
"kubernetes.io/cluster/{{ .Values.global.cluster_name }}": owned
kubernetes.io/cluster/{{ $clusterName }}: owned
securityGroupSelectorTerms:
- tags:
aws:eks:cluster-name: "{{ .Values.global.cluster_name }}"
advancedSecurity:
fips: true
kernelLockdown: Integrity
aws:eks:cluster-name: {{ $clusterName }}
metadataOptions:
httpTokens: required
httpPutResponseHopLimit: 2
Original file line number Diff line number Diff line change
Expand Up @@ -6,8 +6,8 @@ spec:
template:
spec:
nodeClassRef:
group: eks.amazonaws.com
kind: NodeClass
group: karpenter.k8s.aws
kind: EC2NodeClass
name: fips
requirements:
- key: karpenter.sh/capacity-type
Expand Down
3 changes: 3 additions & 0 deletions argocd/config/management-cluster/eks-nodepool/values.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -10,3 +10,6 @@ eksNodePool:
disruption:
consolidationPolicy: WhenEmpty
consolidateAfter: 60s

global:
cluster_name: ""
216 changes: 142 additions & 74 deletions argocd/config/management-cluster/hypershift/templates/05-job.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ metadata:
annotations:
argocd.argoproj.io/sync-options: Replace=true,Force=true
spec:
activeDeadlineSeconds: 1800
activeDeadlineSeconds: 3600
backoffLimit: 1
template:
spec:
Expand All @@ -18,83 +18,151 @@ spec:
volumeAttributes:
secretProviderClass: hypershift-config
containers:
- name: install
image: {{ .Values.hypershift.hypershift.image }}
volumeMounts:
- name: secrets-store
mountPath: /mnt/secrets-store
readOnly: true
env:
- name: OIDC_BUCKET_NAME
valueFrom:
secretKeyRef:
name: hypershift-config-env
key: OIDC_BUCKET_NAME
- name: OIDC_BUCKET_REGION
valueFrom:
secretKeyRef:
name: hypershift-config-env
key: OIDC_BUCKET_REGION
- name: OIDC_WRITER_ROLE_ARN
valueFrom:
secretKeyRef:
name: hypershift-config-env
key: OIDC_WRITER_ROLE_ARN
- name: DNS_ZONE_OPERATOR_ROLE_ARN
value: "{{ .Values.global.dns_zone_operator_role_arn }}"
command:
- /bin/sh
- -c
- |
mkdir -p /tmp/aws
- name: install
image: {{ .Values.hypershift.hypershift.image }}
volumeMounts:
- name: secrets-store
mountPath: /mnt/secrets-store
readOnly: true
env:
- name: DNS_ZONE_OPERATOR_ROLE_ARN
value: "{{ .Values.global.dns_zone_operator_role_arn }}"
command:
- /bin/bash
- -c
- |
set -euo pipefail
mkdir -p /tmp/aws

# Read OIDC configuration from CSI-mounted Secrets Manager secret
OIDC_BUCKET_NAME=$(cat /mnt/secrets-store/oidcBucketName)
OIDC_BUCKET_REGION=$(cat /mnt/secrets-store/oidcBucketRegion)
OIDC_WRITER_ROLE_ARN=$(cat /mnt/secrets-store/oidcWriterRoleArn)

# Private platform creds — Pod Identity provides MC-account credentials
echo -e "[default]\n# pod identity handles auth" > /tmp/aws/private-creds

# OIDC S3 creds — assume into the RC oidc-writer role for cross-account S3+KMS
if [ -n "${OIDC_WRITER_ROLE_ARN:-}" ]; then
cat > /tmp/aws/oidc-creds <<CREDS
[default]
role_arn = ${OIDC_WRITER_ROLE_ARN}
credential_source = EcsContainer
CREDS
else
cp /tmp/aws/private-creds /tmp/aws/oidc-creds
fi

# In-cluster API server credentials — used for CRD polling and
# the external-dns patch below. Defined here so both code paths
# share the same variables without repeating the setup.
_CA=/var/run/secrets/kubernetes.io/serviceaccount/ca.crt
_TOKEN=$(cat /var/run/secrets/kubernetes.io/serviceaccount/token)
_API=https://kubernetes.default.svc

# The hypershift-operator image does not include kubectl; use curl
# against the in-cluster API server to check CRD existence.
_crd_ready() {
curl -sf --cacert "$_CA" -H "Authorization: Bearer $_TOKEN" \
"$_API/apis/apiextensions.k8s.io/v1/customresourcedefinitions/$1" \
-o /dev/null 2>&1
}

# Private platform creds — Pod Identity provides MC-account credentials
echo -e "[default]\n# pod identity handles auth" > /tmp/aws/private-creds
# Prometheus Operator CRDs must exist before hypershift install applies
# ServiceMonitor/PrometheusRule. The monitoring chart may still be syncing.
_CRD_DEADLINE=$((SECONDS + 1800))
echo "Waiting for Prometheus Operator CRDs..."
until _crd_ready servicemonitors.monitoring.coreos.com && \
_crd_ready prometheusrules.monitoring.coreos.com; do
if [ $SECONDS -ge $_CRD_DEADLINE ]; then
echo "ERROR: Prometheus Operator CRDs not available after 30 minutes" >&2
echo "coreos.com CRDs present:" >&2
curl -sf --cacert "$_CA" -H "Authorization: Bearer $_TOKEN" \
"$_API/apis/apiextensions.k8s.io/v1/customresourcedefinitions" \
2>/dev/null | grep -o '"name":"[^"]*coreos[^"]*"' \
|| echo " (none or curl failed)" >&2
exit 1
fi
echo " Waiting for Prometheus Operator CRDs ($(( _CRD_DEADLINE - SECONDS ))s remaining)..."
sleep 15
done
echo "=== Prometheus Operator CRDs present — proceeding with hypershift install ==="

# OIDC S3 creds — assume into the RC oidc-writer role for cross-account S3+KMS
if [ -n "${OIDC_WRITER_ROLE_ARN:-}" ]; then
cat > /tmp/aws/oidc-creds <<CREDS
[default]
role_arn = ${OIDC_WRITER_ROLE_ARN}
credential_source = EcsContainer
CREDS
else
cp /tmp/aws/private-creds /tmp/aws/oidc-creds
fi
_hs_install_rc=0
echo "Running hypershift install..."
hypershift install \
--namespace hypershift \
--enable-conversion-webhook=false \
--hypershift-image {{ .Values.hypershift.hypershift.image }} \
--limit-crd-install AWS \
--private-platform AWS \
--aws-private-creds /tmp/aws/private-creds \
--aws-private-region {{ .Values.hypershift.region }} \
--oidc-storage-provider-s3-bucket-name "${OIDC_BUCKET_NAME}" \
--oidc-storage-provider-s3-region "${OIDC_BUCKET_REGION}" \
--oidc-storage-provider-s3-credentials /tmp/aws/oidc-creds \
{{- if .Values.hypershift.externalDns.domain }}
--external-dns-provider aws \
--external-dns-domain-filter {{ .Values.hypershift.externalDns.domain }} \
--external-dns-secret external-dns \
--external-dns-image {{ .Values.hypershift.externalDns.image }} \
{{- end }}
|| _hs_install_rc=$?
echo "hypershift install exit code: $_hs_install_rc"
if [ $_hs_install_rc -ne 0 ]; then
# The aws-iam-auth build of hypershift-operator has a known post-apply
# internal shell error that fires after all resources are applied. Check
# whether the namespace exists to distinguish this from a real failure.
_hs_ns_http=$(curl -s --cacert "$_CA" -H "Authorization: Bearer $_TOKEN" \
"$_API/api/v1/namespaces/hypershift" -o /dev/null -w "%{http_code}" 2>/dev/null || echo "000")
echo "hypershift namespace check: HTTP ${_hs_ns_http}"
if [ "$_hs_ns_http" != "200" ]; then
echo "ERROR: hypershift install failed (exit $_hs_install_rc) and HyperShift namespace not found (HTTP ${_hs_ns_http})" >&2
exit $_hs_install_rc
fi
echo "WARNING: hypershift install exited $_hs_install_rc but HyperShift namespace exists — resources applied (known post-apply issue in aws-iam-auth build)"
fi
{{- if .Values.hypershift.externalDns.domain }}
# TODO(hypershift): Upstream --aws-assume-role + Pod Identity support
# to hypershift install CLI, then replace this post-install patch
# with native flags.
if [ -n "${DNS_ZONE_OPERATOR_ROLE_ARN:-}" ]; then
echo "Patching external-dns with --aws-assume-role=${DNS_ZONE_OPERATOR_ROLE_ARN}"

hypershift install \
--namespace hypershift \
--enable-conversion-webhook=false \
--hypershift-image {{ .Values.hypershift.hypershift.image }} \
--limit-crd-install AWS \
--private-platform AWS \
--aws-private-creds /tmp/aws/private-creds \
--aws-private-region {{ .Values.hypershift.region }} \
--oidc-storage-provider-s3-bucket-name $(OIDC_BUCKET_NAME) \
--oidc-storage-provider-s3-region $(OIDC_BUCKET_REGION) \
--oidc-storage-provider-s3-credentials /tmp/aws/oidc-creds \
--external-dns-provider aws \
--external-dns-domain-filter {{ .Values.hypershift.externalDns.domain }} \
--external-dns-secret external-dns \
--external-dns-image {{ .Values.hypershift.externalDns.image }} \
_patch_http=$(curl -s --cacert "$_CA" -H "Authorization: Bearer $_TOKEN" \
-H "Content-Type: application/json-patch+json" -X PATCH \
"${_API}/apis/apps/v1/namespaces/hypershift/deployments/external-dns" \
-d '[
{"op":"add","path":"/spec/template/spec/containers/0/args/-","value":"--aws-assume-role='"${DNS_ZONE_OPERATOR_ROLE_ARN}"'"},
{"op":"add","path":"/spec/template/spec/tolerations","value":[{"key":"CriticalAddonsOnly","operator":"Exists","effect":"NoSchedule"}]}
]' \
-o /dev/null -w "%{http_code}" 2>/dev/null || echo "000")
echo " external-dns deployment patch: HTTP ${_patch_http}"
if [ "$_patch_http" != "200" ] && [ "$_patch_http" != "201" ]; then
echo "WARNING: external-dns deployment patch returned HTTP ${_patch_http} — skipping" >&2
fi

# TODO(hypershift): Upstream --aws-assume-role + Pod Identity support
# to hypershift install CLI, then replace this post-install patch
# with native flags.
if [ -n "${DNS_ZONE_OPERATOR_ROLE_ARN:-}" ]; then
echo "Patching external-dns with --aws-assume-role=${DNS_ZONE_OPERATOR_ROLE_ARN}"
_CA=/var/run/secrets/kubernetes.io/serviceaccount/ca.crt
_TOKEN=$(cat /var/run/secrets/kubernetes.io/serviceaccount/token)
_API="https://kubernetes.default.svc"
# Upstream v0.21.0 needs discovery.k8s.io and networking.k8s.io API groups
# that HyperShift's generated ClusterRole doesn't include.
_patch_http=$(curl -s --cacert "$_CA" -H "Authorization: Bearer $_TOKEN" \
-H "Content-Type: application/merge-patch+json" -X PATCH \
"${_API}/apis/rbac.authorization.k8s.io/v1/clusterroles/external-dns" \
-d '{"rules":[
{"apiGroups":["","discovery.k8s.io"],"resources":["services","endpoints","pods","nodes","endpointslices"],"verbs":["get","watch","list"]},
{"apiGroups":["extensions","networking.k8s.io"],"resources":["ingresses","ingressroutes","ingressroutetcps","ingressrouteudps"],"verbs":["get","list","watch"]},
{"apiGroups":["route.openshift.io"],"resources":["routes"],"verbs":["get","list","watch"]}
]}' \
-o /dev/null -w "%{http_code}" 2>/dev/null || echo "000")
echo " external-dns clusterrole patch: HTTP ${_patch_http}"
if [ "$_patch_http" != "200" ] && [ "$_patch_http" != "201" ]; then
echo "WARNING: external-dns clusterrole patch returned HTTP ${_patch_http} — skipping" >&2
fi
fi

# Patch the command and add --aws-assume-role.
curl -sf --cacert "$_CA" -H "Authorization: Bearer $_TOKEN" \
-H "Content-Type: application/json-patch+json" -X PATCH \
"${_API}/apis/apps/v1/namespaces/hypershift/deployments/external-dns" \
-d '[
{"op":"add","path":"/spec/template/spec/containers/0/args/-","value":"--aws-assume-role='"${DNS_ZONE_OPERATOR_ROLE_ARN}"'"}
]' \
-o /dev/null -w " HTTP %{http_code}\n"
fi
{{- end }}
restartPolicy: Never
serviceAccountName: hypershift-installer
tolerations:
- key: CriticalAddonsOnly
operator: Exists
effect: NoSchedule
25 changes: 14 additions & 11 deletions argocd/config/management-cluster/monitoring/values.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -16,18 +16,21 @@ kube-prometheus-stack:
enabled: false

prometheusOperator:
tolerations:
- key: CriticalAddonsOnly
operator: Exists
effect: NoSchedule
admissionWebhooks:
# During bootstrap, ArgoCD syncs itself and restarts its controllers
# while other app syncs are in flight. If a PreSync hook Job completes
# while no controller is watching, the Kubernetes TTL controller deletes
# the Job (upstream default: 60s) before the replacement controller can
# observe it, leaving the sync stuck indefinitely.
# Bump to 600s to give the new controller time to come back.
# Ref: https://github.com/argoproj/argo-cd/issues/21055
create:
ttlSecondsAfterFinished: 600
patch:
ttlSecondsAfterFinished: 600
# Disabled: the kube-webhook-certgen PreSync hook Jobs stall monitoring
# sync during bootstrap when ArgoCD controllers restart and miss the hook
# completion event (ArgoCD #21055).
enabled: false
tls:
# Disabled alongside admissionWebhooks: the PrometheusOperator Deployment
# mounts a tls-secret volume gated on tls.enabled (not admissionWebhooks.enabled).
# Without this flag the pod fails to start because the certgen Job that
# creates the secret was disabled above.
enabled: false
resources:
requests:
cpu: 50m
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
apiVersion: v2
name: aws-load-balancer-controller
description: AWS Load Balancer Controller — provides TargetGroupBinding CRDs for OSS Karpenter clusters
type: application
version: 0.1.0

dependencies:
- name: aws-load-balancer-controller
version: 1.17.1
repository: https://aws.github.io/eks-charts
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
aws-load-balancer-controller:
# clusterName injected via ApplicationSet valuesObject (aws-load-balancer-controller.clusterName)
serviceAccount:
create: true
name: aws-load-balancer-controller

podSecurityContext:
runAsNonRoot: true
runAsUser: 65534
fsGroup: 65534

securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop:
- ALL
Original file line number Diff line number Diff line change
@@ -1,15 +1,18 @@
apiVersion: eks.amazonaws.com/v1
kind: NodeClass
{{- $clusterName := required "global.cluster_name must be set via ApplicationSet valuesObject" .Values.global.cluster_name -}}
apiVersion: karpenter.k8s.aws/v1
kind: EC2NodeClass
metadata:
name: fips
spec:
role: "{{ .Values.global.cluster_name }}-auto-node-role"
amiSelectorTerms:
- alias: bottlerocket@latest
instanceProfile: {{ $clusterName }}-karpenter-node-role
subnetSelectorTerms:
- tags:
"kubernetes.io/cluster/{{ .Values.global.cluster_name }}": owned
kubernetes.io/cluster/{{ $clusterName }}: owned
securityGroupSelectorTerms:
- tags:
aws:eks:cluster-name: "{{ .Values.global.cluster_name }}"
advancedSecurity:
fips: true
kernelLockdown: Integrity
aws:eks:cluster-name: {{ $clusterName }}
metadataOptions:
httpTokens: required
httpPutResponseHopLimit: 2
Original file line number Diff line number Diff line change
Expand Up @@ -7,8 +7,8 @@ spec:
template:
spec:
nodeClassRef:
group: eks.amazonaws.com
kind: NodeClass
group: karpenter.k8s.aws
kind: EC2NodeClass
name: fips
requirements:
- key: karpenter.sh/capacity-type
Expand Down
3 changes: 3 additions & 0 deletions argocd/config/regional-cluster/eks-nodepool/values.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -10,3 +10,6 @@ eksNodePool:
disruption:
consolidationPolicy: WhenEmpty
consolidateAfter: 60s

global:
cluster_name: ""
Loading