From 6d38982935b957f41397ab894b7fe0ccb7e3526a Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Tue, 11 Aug 2026 14:37:43 -0500 Subject: [PATCH 01/13] ci: SASL E2E lane, rolling-restart wiring, and collection version pin The harness had zero coverage of SASL, user_management, or the broker rolling restart -- exactly the areas where the collection's worst bugs lived. The new ci:aws:rp:sasl lane fresh-installs the candidate collection with kafka_enable_authorization, reconciles a test user and topic ACL via the user_management role, and asserts real enforcement: superuser access works, anonymous access is denied, and the ACL-scoped user can use exactly its topic and nothing else. The lane then performs a SASL-authenticated rolling restart via operation-rolling-restart.yml (previously exercised by no lane) and re-verifies enforcement and data access afterwards. The restart playbook gains conditional RPK_USER/RPK_PASS environment for SASL clusters -- populated only when authorization is on, since rpk rejects empty credential env vars. requirements.yml pins redpanda.cluster to the released 0.12.0: floating on Galaxy latest meant any bad collection release broke every lane immediately; bumps are now deliberate. Co-Authored-By: Claude Fable 5 --- .buildkite/pipeline.yml | 18 +++++++++ .tasks/ci.yml | 56 +++++++++++++++++++++++++++ .tasks/cluster.yml | 18 +++++++++ .tasks/operations.yml | 27 ++++++++++++- .tasks/testing.yml | 35 +++++++++++++++++ ansible/operation-rolling-restart.yml | 4 ++ ansible/operation-user-management.yml | 19 +++++++++ ansible/provision-cluster-sasl.yml | 25 ++++++++++++ requirements.yml | 3 ++ 9 files changed, 203 insertions(+), 2 deletions(-) create mode 100644 ansible/operation-user-management.yml create mode 100644 ansible/provision-cluster-sasl.yml diff --git a/.buildkite/pipeline.yml b/.buildkite/pipeline.yml index 64722fe9..86bd129a 100755 --- a/.buildkite/pipeline.yml +++ b/.buildkite/pipeline.yml @@ -22,6 +22,24 @@ steps: - BASELINE_COLLECTION_REF - CANDIDATE_COLLECTION_REF - REDPANDA_VERSION + - label: aws ubuntu sasl + key: aws-up-ubuntu-sasl + concurrency_group: aws-ub + concurrency: 1 + timeout_in_minutes: 120 + command: DEPLOYMENT_ID=ci-sa-ub-`tr -dc a-z0-9 - + CI workflow - AWS Redpanda with SASL authorization. Single-phase: fresh-installs the + CANDIDATE collection with kafka_enable_authorization, reconciles a test user + topic ACL + via user_management, verifies enforcement (superuser ok, anonymous denied, ACL-scoped + user constrained to its topic), performs a SASL-authenticated rolling restart via + operation-rolling-restart.yml, and re-verifies enforcement and data access afterwards. + vars: + DISTRO: '{{.DISTRO | default "ubuntu-focal"}}' + SASL_USER: admin + SASL_PASSWORD: '{{.DEPLOYMENT_ID}}-sasl-pw' + SASL_TEST_USER: apptest + SASL_TEST_PASSWORD: '{{.DEPLOYMENT_ID}}-app-pw' + SASL_TEST_TOPIC: apptopic + cmds: + - defer: + task: :infra:aws:destroy + - task: :tools:keygen + - task: :tools:rpk:install + - task: :infra:aws:build + vars: {DISTRO: "{{.DISTRO}}"} + - task: :ansible:prereqs + - task: :ansible:collection:candidate + vars: {CANDIDATE_COLLECTION_REF: "{{.CANDIDATE_COLLECTION_REF}}"} + - task: :cluster:sasl + vars: + SASL_USER: "{{.SASL_USER}}" + SASL_PASSWORD: "{{.SASL_PASSWORD}}" + - task: :ops:users:apply + vars: + SASL_USER: "{{.SASL_USER}}" + SASL_PASSWORD: "{{.SASL_PASSWORD}}" + SASL_USERS_JSON: '{"sasl_users":[{"username":"{{.SASL_TEST_USER}}","password":"{{.SASL_TEST_PASSWORD}}","state":"present"}]}' + SASL_ACLS_JSON: '{"sasl_acls":[{"principal":"{{.SASL_TEST_USER}}","operation":["read","write","describe"],"resource_type":"topic","resource_name":"{{.SASL_TEST_TOPIC}}","permission":"allow"}]}' + - task: :test:cluster:sasl + vars: + SASL_USER: "{{.SASL_USER}}" + SASL_PASSWORD: "{{.SASL_PASSWORD}}" + SASL_TEST_USER: "{{.SASL_TEST_USER}}" + SASL_TEST_PASSWORD: "{{.SASL_TEST_PASSWORD}}" + SASL_TEST_TOPIC: "{{.SASL_TEST_TOPIC}}" + # rolling restart with SASL credentials; enforcement and the app user's + # data must survive it + - task: :ops:broker:restart + vars: + SASL_USER: "{{.SASL_USER}}" + SASL_PASSWORD: "{{.SASL_PASSWORD}}" + - task: :test:cluster:sasl + vars: + SASL_USER: "{{.SASL_USER}}" + SASL_PASSWORD: "{{.SASL_PASSWORD}}" + SASL_TEST_USER: "{{.SASL_TEST_USER}}" + SASL_TEST_PASSWORD: "{{.SASL_TEST_PASSWORD}}" + SASL_TEST_TOPIC: "{{.SASL_TEST_TOPIC}}" + - task: :infra:aws:destroy + aws:rp:tiered:unstable: desc: >- CI workflow - AWS tiered storage (TLS) against the UNSTABLE Redpanda repo. Single-phase, diff --git a/.tasks/cluster.yml b/.tasks/cluster.yml index d998309d..d1fdf968 100644 --- a/.tasks/cluster.yml +++ b/.tasks/cluster.yml @@ -21,6 +21,24 @@ tasks: {{if .RP_VERSION}}--extra-vars redpanda_version={{.RP_VERSION}}{{end}} {{if .RP_INSTALL_STATUS}}--extra-vars redpanda_install_status={{.RP_INSTALL_STATUS}}{{end}} + sasl: + desc: >- + Provision/converge Redpanda with SASL authorization enabled. SASL_USER/SASL_PASSWORD + set the bootstrap superuser; RP_VERSION/RP_INSTALL_STATUS as in :cluster:provision. + deps: + - :ansible:prereqs + - :ensure-logs-dir + cmds: + - >- + ansible-playbook ansible/provision-cluster-sasl.yml + --private-key {{.PRIVATE_KEY}} + --inventory {{.ANSIBLE_INVENTORY}} + --extra-vars is_using_unstable={{.IS_USING_UNSTABLE}} + --extra-vars sasl_user={{.SASL_USER}} + --extra-vars sasl_password={{.SASL_PASSWORD}} + {{if .RP_VERSION}}--extra-vars redpanda_version={{.RP_VERSION}}{{end}} + {{if .RP_INSTALL_STATUS}}--extra-vars redpanda_install_status={{.RP_INSTALL_STATUS}}{{end}} + tiered: desc: >- Provision/converge Redpanda with tiered storage. RP_VERSION/RP_INSTALL_STATUS as in diff --git a/.tasks/operations.yml b/.tasks/operations.yml index c63aec43..d6af2d0e 100644 --- a/.tasks/operations.yml +++ b/.tasks/operations.yml @@ -18,12 +18,35 @@ tasks: - ansible-playbook ansible/operation-rolling-restart-connect.yml --private-key {{.PRIVATE_KEY}} --inventory {{.ANSIBLE_INVENTORY}} broker:restart: - desc: Perform rolling restart of Redpanda broker cluster + desc: >- + Perform rolling restart of Redpanda broker cluster. On SASL clusters pass + SASL_USER/SASL_PASSWORD so the playbook's rpk calls authenticate. deps: - :ansible:prereqs - :ensure-logs-dir cmds: - - ansible-playbook ansible/operation-rolling-restart.yml --private-key {{.PRIVATE_KEY}} --inventory {{.ANSIBLE_INVENTORY}} + - >- + ansible-playbook ansible/operation-rolling-restart.yml + --private-key {{.PRIVATE_KEY}} + --inventory {{.ANSIBLE_INVENTORY}} + {{if .SASL_PASSWORD}}--extra-vars kafka_enable_authorization=true --extra-vars sasl_user={{.SASL_USER}} --extra-vars sasl_password={{.SASL_PASSWORD}}{{end}} + + users:apply: + desc: >- + Reconcile SASL users and ACLs via the user_management role. Desired state is passed + as JSON in SASL_USERS_JSON / SASL_ACLS_JSON; admin creds via SASL_USER/SASL_PASSWORD. + deps: + - :ansible:prereqs + - :ensure-logs-dir + cmds: + - >- + ansible-playbook ansible/operation-user-management.yml + --private-key {{.PRIVATE_KEY}} + --inventory {{.ANSIBLE_INVENTORY}} + --extra-vars sasl_user={{.SASL_USER}} + --extra-vars sasl_password={{.SASL_PASSWORD}} + --extra-vars '{{.SASL_USERS_JSON}}' + --extra-vars '{{.SASL_ACLS_JSON}}' license:apply: desc: Apply Redpanda license to the cluster diff --git a/.tasks/testing.yml b/.tasks/testing.yml index 45bebc4a..1ea55ee8 100644 --- a/.tasks/testing.yml +++ b/.tasks/testing.yml @@ -92,6 +92,41 @@ tasks: echo "Consuming from topic" {{.RPK_PATH}} topic consume {{.TEST_TOPIC_NAME}} -X brokers=$REDPANDA_BROKERS -v -o :end | grep squirrel || exit 1 + cluster:sasl: + desc: >- + Verify SASL enforcement: superuser rpk works, anonymous rpk is denied, and a + user_management-created user can use exactly the topic its ACL allows. + cmds: + - chmod 775 {{.RPK_PATH}} + - | + REDPANDA_BROKERS=$(sed -n '/^\[redpanda\]/,/^\[/p' "{{.HOSTS_FILE}}" | grep '^[0-9]' | cut -d' ' -f1 | sed 's/$/:9092/' | paste -sd ',' -) + echo "Brokers: $REDPANDA_BROKERS" + SUPER="-X user={{.SASL_USER}} -X pass={{.SASL_PASSWORD}} -X sasl.mechanism=SCRAM-SHA-256" + APP="-X user={{.SASL_TEST_USER}} -X pass={{.SASL_TEST_PASSWORD}} -X sasl.mechanism=SCRAM-SHA-256" + + echo "Superuser can read cluster status" + {{.RPK_PATH}} cluster status -X brokers=$REDPANDA_BROKERS $SUPER -v || exit 1 + + echo "Anonymous topic creation must be denied" + if {{.RPK_PATH}} topic create anon-denied -X brokers=$REDPANDA_BROKERS 2>/dev/null; then + echo "ERROR: anonymous client was allowed to create a topic on a SASL cluster" + exit 1 + fi + + echo "Superuser creates the app topic" + {{.RPK_PATH}} topic create {{.SASL_TEST_TOPIC}} -p 3 -X brokers=$REDPANDA_BROKERS $SUPER -v || exit 1 + + echo "ACL-scoped user produces and consumes its topic" + echo sasl-squirrel | {{.RPK_PATH}} topic produce {{.SASL_TEST_TOPIC}} -X brokers=$REDPANDA_BROKERS $APP || exit 1 + {{.RPK_PATH}} topic consume {{.SASL_TEST_TOPIC}} -X brokers=$REDPANDA_BROKERS $APP -o :end | grep sasl-squirrel || exit 1 + + echo "ACL-scoped user must not create unrelated topics" + if {{.RPK_PATH}} topic create off-limits -X brokers=$REDPANDA_BROKERS $APP 2>/dev/null; then + echo "ERROR: ACL-scoped user was allowed to create a topic outside its ACL" + exit 1 + fi + echo "SASL enforcement verified" + cluster:tls: desc: Test Redpanda cluster with TLS cmds: diff --git a/ansible/operation-rolling-restart.yml b/ansible/operation-rolling-restart.yml index 63224d69..317deeaa 100644 --- a/ansible/operation-rolling-restart.yml +++ b/ansible/operation-rolling-restart.yml @@ -5,6 +5,10 @@ serial: 1 vars: rpk_bin: rpk + # On SASL-enabled clusters every rpk invocation below needs credentials; + # rpk rejects empty RPK_USER/RPK_PASS ("Malformed Authorization header"), + # so the environment is only populated when authorization is on. + environment: "{{ {'RPK_USER': sasl_user | default('admin'), 'RPK_PASS': sasl_password | default('')} if (kafka_enable_authorization | default(false) | bool) else {} }}" tasks: - name: Check cluster health diff --git a/ansible/operation-user-management.yml b/ansible/operation-user-management.yml new file mode 100644 index 00000000..4fde0f86 --- /dev/null +++ b/ansible/operation-user-management.yml @@ -0,0 +1,19 @@ +# Reconciles SASL users and ACLs via the user_management role. +# The desired users/ACLs are passed by the caller via --extra-vars, e.g.: +# -e sasl_password=... \ +# -e '{"sasl_users":[{"username":"app","password":"s3cret","state":"present"}]}' \ +# -e '{"sasl_acls":[{"principal":"app","operation":["read","write","describe","create"],"resource_type":"topic","resource_name":"apptopic","permission":"allow"}]}' +# Admin credentials flow from sasl_superuser_* (the role follows the broker's +# variables), so the same extra-vars used for provisioning work here. +--- +- name: Manage SASL users and ACLs + hosts: redpanda + run_once: true + vars: + kafka_enable_authorization: true + sasl_superuser_username: "{{ sasl_user | default('admin') }}" + sasl_superuser_password: "{{ sasl_password | default('ci-only-password') }}" + tasks: + - name: Reconcile users, roles and ACLs + ansible.builtin.include_role: + name: redpanda.cluster.user_management diff --git a/ansible/provision-cluster-sasl.yml b/ansible/provision-cluster-sasl.yml new file mode 100644 index 00000000..f06b72fa --- /dev/null +++ b/ansible/provision-cluster-sasl.yml @@ -0,0 +1,25 @@ +# Provisions a Redpanda cluster with SASL authorization enabled. +# variables +# advertise_public_ips : advertise public ips so the CI runner can reach the cluster +# sasl_superuser_username / sasl_superuser_password : bootstrap superuser; supply +# real values via --extra-vars (the defaults here are CI-only placeholders) +--- +- name: Provision SASL-enabled cluster + hosts: redpanda + vars: + advertise_public_ips: true + redpanda_version: latest + kafka_enable_authorization: true + sasl_superuser_username: "{{ sasl_user | default('admin') }}" + sasl_superuser_password: "{{ sasl_password | default('ci-only-password') }}" + tasks: + - name: Install system prereqs + ansible.builtin.include_role: + name: redpanda.cluster.system_setup + - name: Handle sysctl changes + ansible.builtin.include_role: + name: redpanda.cluster.sysctl_setup + - name: Install and start redpanda + ansible.builtin.include_role: + name: redpanda.cluster.redpanda_broker + when: not skip_node | default(false) | bool diff --git a/requirements.yml b/requirements.yml index ed394291..4489e14c 100644 --- a/requirements.yml +++ b/requirements.yml @@ -2,6 +2,9 @@ collections: - name: community.general - name: redpanda.cluster type: galaxy + # Pinned: an unpinned 'latest' means any bad collection release breaks + # every CI lane immediately. Bump deliberately when a release ships. + version: 0.12.0 - name: ansible.posix - name: grafana.grafana version: 5.6.0 From 1e8d6b912d3d7cffc38801ba57c829de0c991dd1 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Tue, 11 Aug 2026 15:28:36 -0500 Subject: [PATCH 02/13] chore: gitignore .env The Taskfile auto-loads .env (dotenv) for local credentials; it must never be committable. Co-Authored-By: Claude Fable 5 --- .gitignore | 1 + 1 file changed, 1 insertion(+) diff --git a/.gitignore b/.gitignore index e9c39148..61cb0f91 100644 --- a/.gitignore +++ b/.gitignore @@ -22,3 +22,4 @@ ansible/tls/clients/client.crt /aws-extra/ ansible/proxy/tls/** .ansible +.env From cf113817ce389151cbbdad3f30d1f87edfc03331 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Tue, 11 Aug 2026 15:31:12 -0500 Subject: [PATCH 03/13] fix: forward AWS_SESSION_TOKEN so SSO/STS temporary credentials work The env block re-exported only the static access key pair, silently stripping the session token -- any developer running lanes locally with aws sso credentials got InvalidClientTokenId from terraform. Static CI keys leave the token empty, which AWS ignores. Co-Authored-By: Claude Fable 5 --- Taskfile.yml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/Taskfile.yml b/Taskfile.yml index e7a48347..656d2b6f 100644 --- a/Taskfile.yml +++ b/Taskfile.yml @@ -161,6 +161,9 @@ env: # AWS Credentials AWS_ACCESS_KEY_ID: '{{.AWS_ACCESS_KEY_ID | default .DA_AWS_ACCESS_KEY_ID}}' AWS_SECRET_ACCESS_KEY: '{{.AWS_SECRET_ACCESS_KEY | default .DA_AWS_SECRET_ACCESS_KEY}}' + # Session token must pass through or SSO/STS temporary credentials fail + # with InvalidClientTokenId (static CI keys leave it empty, which is fine) + AWS_SESSION_TOKEN: '{{.AWS_SESSION_TOKEN}}' AWS_DEFAULT_REGION: '{{.AWS_DEFAULT_REGION}}' # Ansible Environment From ba72886472b424cdeb171a3586bcb839be988465 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Tue, 11 Aug 2026 15:34:22 -0500 Subject: [PATCH 04/13] fix: stop clobbering .env-supplied AWS credentials in the global env block Global env: templates evaluate before dotenv merges, so the aliasing entries ('{{.AWS_ACCESS_KEY_ID | default .DA_AWS_ACCESS_KEY_ID}}' etc.) resolved empty for anyone supplying credentials via .env and overrode them -- terraform then fell back to stale ~/.aws keys and failed with InvalidClientTokenId. Verified against go-task 3.39 and 3.52; the ordering is by design. CI passes real process env, which flows through untouched without these lines; the legacy DA_AWS_* aliases are dropped. This also supersedes the earlier session-token passthrough attempt, which could never work from dotenv for the same ordering reason. Co-Authored-By: Claude Fable 5 --- Taskfile.yml | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/Taskfile.yml b/Taskfile.yml index 656d2b6f..1e471460 100644 --- a/Taskfile.yml +++ b/Taskfile.yml @@ -158,12 +158,12 @@ env: # Terraform TF_IN_AUTOMATION: "{{.CI}}" - # AWS Credentials - AWS_ACCESS_KEY_ID: '{{.AWS_ACCESS_KEY_ID | default .DA_AWS_ACCESS_KEY_ID}}' - AWS_SECRET_ACCESS_KEY: '{{.AWS_SECRET_ACCESS_KEY | default .DA_AWS_SECRET_ACCESS_KEY}}' - # Session token must pass through or SSO/STS temporary credentials fail - # with InvalidClientTokenId (static CI keys leave it empty, which is fine) - AWS_SESSION_TOKEN: '{{.AWS_SESSION_TOKEN}}' + # AWS credentials are deliberately NOT re-exported here: global env + # templates evaluate before dotenv merges, so aliasing lines like + # '{{.AWS_ACCESS_KEY_ID}}' resolve empty for .env users and clobber their + # credentials (leaving terraform to fall back to stale ~/.aws keys). + # Process env (CI) and dotenv (.env) both flow through untouched without + # them. The legacy DA_AWS_* fallback aliases are dropped with this. AWS_DEFAULT_REGION: '{{.AWS_DEFAULT_REGION}}' # Ansible Environment From 0701e51eb036f49be55018aadd5fbc23b8eb1ca8 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Tue, 11 Aug 2026 15:37:48 -0500 Subject: [PATCH 05/13] fix: cleanup taskfile env block clobbered credentials for every task go-task merges included-taskfile env into the whole run, so cleanup.yml's env block applied far beyond the reaper tasks: its credential templates resolve empty before dotenv merges (blanking .env-supplied keys everywhere) and it exported the comma-joined reaper region list ('us-east-2,us-west-2') as the global AWS_DEFAULT_REGION, which the aws CLI rejects outright. The reaper already receives its regions via --region arguments and credentials flow from process env or .env untouched, so the block is deleted. Co-Authored-By: Claude Fable 5 --- .tasks/cleanup.yml | 15 ++++++--------- 1 file changed, 6 insertions(+), 9 deletions(-) diff --git a/.tasks/cleanup.yml b/.tasks/cleanup.yml index 968307d8..f137daae 100644 --- a/.tasks/cleanup.yml +++ b/.tasks/cleanup.yml @@ -7,15 +7,12 @@ vars: GCP_PROJECT_ID: '{{.GCP_PROJECT_ID | default "hallowed-ray-376320"}}' MIN_AGE: '{{.MIN_AGE | default ""}}' -env: - # AWS Credentials (inherited from parent or environment) - AWS_ACCESS_KEY_ID: '{{.AWS_ACCESS_KEY_ID}}' - AWS_SECRET_ACCESS_KEY: '{{.AWS_SECRET_ACCESS_KEY}}' - AWS_DEFAULT_REGION: '{{.AWS_CLEANUP_REGION}}' - - # GCP Credentials (supports both variable names) - GCP_CREDS: '{{.GCP_CREDS}}' - GOOGLE_CREDENTIALS_BASE64: '{{.GOOGLE_CREDENTIALS_BASE64}}' +# No env block here: included-taskfile env merges into the WHOLE run, and +# these entries clobbered dotenv-supplied credentials globally (the cred +# templates resolve empty before dotenv merges) while exporting the +# comma-joined reaper region list as everyone's AWS_DEFAULT_REGION. The +# reaper gets its regions via --region; credentials flow from process env +# or .env untouched. tasks: aws: From eb04e1fd7648fb3751ee09c7f0ee7de93abe2228 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Tue, 11 Aug 2026 16:09:46 -0500 Subject: [PATCH 06/13] fix: SASL verification must be re-runnable after the rolling restart The post-restart re-verification re-created the app topic and failed on TOPIC_ALREADY_EXISTS; creation now falls back to describe, the same idiom the upgrade seed task uses. Co-Authored-By: Claude Fable 5 --- .tasks/testing.yml | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/.tasks/testing.yml b/.tasks/testing.yml index 1ea55ee8..86eb91ef 100644 --- a/.tasks/testing.yml +++ b/.tasks/testing.yml @@ -113,8 +113,10 @@ tasks: exit 1 fi - echo "Superuser creates the app topic" - {{.RPK_PATH}} topic create {{.SASL_TEST_TOPIC}} -p 3 -X brokers=$REDPANDA_BROKERS $SUPER -v || exit 1 + echo "Superuser creates the app topic (or confirms it exists on re-verification)" + {{.RPK_PATH}} topic create {{.SASL_TEST_TOPIC}} -p 3 -X brokers=$REDPANDA_BROKERS $SUPER \ + || {{.RPK_PATH}} topic describe {{.SASL_TEST_TOPIC}} -X brokers=$REDPANDA_BROKERS $SUPER >/dev/null \ + || exit 1 echo "ACL-scoped user produces and consumes its topic" echo sasl-squirrel | {{.RPK_PATH}} topic produce {{.SASL_TEST_TOPIC}} -X brokers=$REDPANDA_BROKERS $APP || exit 1 From b82131c51dda51655d3bc636943cbeb504cdbe14 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Tue, 11 Aug 2026 23:05:29 -0500 Subject: [PATCH 07/13] TEMP: point redpanda.cluster at the collection-hardening branch DO NOT MERGE this commit -- it exists so CI validates this harness against the 0.13.0 candidate collection. Revert to the pinned Galaxy release before merging. Co-Authored-By: Claude Fable 5 --- requirements.yml | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/requirements.yml b/requirements.yml index 4489e14c..80b681f8 100644 --- a/requirements.yml +++ b/requirements.yml @@ -1,10 +1,11 @@ collections: - name: community.general - - name: redpanda.cluster - type: galaxy - # Pinned: an unpinned 'latest' means any bad collection release breaks - # every CI lane immediately. Bump deliberately when a release ships. - version: 0.12.0 + # TEMP (DO NOT MERGE): points at the collection-hardening branch so CI can + # validate this harness against the 0.13.0 candidate. Revert to the pinned + # Galaxy release before merging. + - name: https://github.com/redpanda-data/redpanda-ansible-collection.git + type: git + version: collection-hardening - name: ansible.posix - name: grafana.grafana version: 5.6.0 From 8442d9ba215365365b602749e1451a034dd912a7 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Wed, 12 Aug 2026 16:43:32 -0500 Subject: [PATCH 08/13] TEMP: pin CANDIDATE_COLLECTION_REF to the collection-hardening branch DO NOT MERGE. The agent environment supplies CANDIDATE_COLLECTION_REF pointing at collection main via the pipeline secret, which force-installs main over the requirements.yml branch pin -- build 445 validated main instead of the candidate (its sasl-lane auth failure is main's broken SASL bootstrap, fixed in the candidate). Pipeline-level env pins the candidate for every lane, same mechanism the devprod-4109 branch used. Drop together with the requirements TEMP commit before merging. Co-Authored-By: Claude Fable 5 --- .buildkite/pipeline.yml | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/.buildkite/pipeline.yml b/.buildkite/pipeline.yml index 86bd129a..cefe038d 100755 --- a/.buildkite/pipeline.yml +++ b/.buildkite/pipeline.yml @@ -1,6 +1,15 @@ agents: queue: "k8s-m6ixlarge" +# TEMP (DO NOT MERGE): the agent environment supplies +# CANDIDATE_COLLECTION_REF=git+...,main from the pipeline secret, which +# force-installs collection main over the requirements.yml branch pin (seen +# in build 445 -- it validated main, not the candidate). Pin the candidate +# here so every lane tests PR #152's branch; drop together with the +# requirements TEMP commit before merging. +env: + CANDIDATE_COLLECTION_REF: "git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening" + steps: - label: aws ubuntu key: aws-up-ubuntu From e51348e965d562272253d898f4395b564dec9b44 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Wed, 12 Aug 2026 17:04:42 -0500 Subject: [PATCH 09/13] fix: retry the monitoring deploy once on transient download failures The node_exporter tarball download from GitHub fails intermittently under concurrent-lane rate limiting (multiple lanes each fetching to several hosts from one egress IP -- builds 445 and 446 both lost lanes to 'Remote end closed connection without response'). The deploy plays are idempotent, so a single bounded retry with a 30s backoff absorbs the flake without masking real failures. Co-Authored-By: Claude Fable 5 --- .tasks/monitoring.yml | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/.tasks/monitoring.yml b/.tasks/monitoring.yml index cd2e2dbe..6fa0eb39 100644 --- a/.tasks/monitoring.yml +++ b/.tasks/monitoring.yml @@ -8,7 +8,14 @@ tasks: - :ensure-logs-dir cmds: - export OBJC_DISABLE_INITIALIZE_FORK_SAFETY=YES - - ansible-playbook ansible/deploy-monitor.yml --private-key {{.PRIVATE_KEY}} --inventory {{.ANSIBLE_INVENTORY}} --extra-vars is_using_unstable={{.IS_USING_UNSTABLE}} + - | + # node_exporter downloads from GitHub flake under concurrent-lane rate + # limiting (builds 445/446); the play is idempotent, so retry once + for attempt in 1 2; do + ansible-playbook ansible/deploy-monitor.yml --private-key {{.PRIVATE_KEY}} --inventory {{.ANSIBLE_INVENTORY}} --extra-vars is_using_unstable={{.IS_USING_UNSTABLE}} && break + [ "$attempt" = 2 ] && exit 1 + echo "monitor deploy failed (attempt $attempt); retrying in 30s"; sleep 30 + done deploy:tls: desc: Deploy monitoring stack with TLS @@ -16,7 +23,14 @@ tasks: - :ansible:prereqs - :ensure-logs-dir cmds: - - ansible-playbook ansible/deploy-monitor-tls.yml --private-key {{.PRIVATE_KEY}} --inventory {{.ANSIBLE_INVENTORY}} --extra-vars is_using_unstable={{.IS_USING_UNSTABLE}} + - | + # node_exporter downloads from GitHub flake under concurrent-lane rate + # limiting (builds 445/446); the play is idempotent, so retry once + for attempt in 1 2; do + ansible-playbook ansible/deploy-monitor-tls.yml --private-key {{.PRIVATE_KEY}} --inventory {{.ANSIBLE_INVENTORY}} --extra-vars is_using_unstable={{.IS_USING_UNSTABLE}} && break + [ "$attempt" = 2 ] && exit 1 + echo "monitor deploy failed (attempt $attempt); retrying in 30s"; sleep 30 + done extra:deploy: desc: Deploy monitoring on secondary cluster From fc5de7a17ef562d907cad7c705c586cee6b14eb1 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Wed, 12 Aug 2026 17:06:50 -0500 Subject: [PATCH 10/13] TEMP: pin the candidate collection inside each container spec DO NOT MERGE. Build 446 proved the pipeline-level env pin is dead on arrival: the aws-sm secret hook exports CANDIDATE_COLLECTION_REF (git+...,main) into the job environment after yaml env applies, so every lane still validated collection main. The docker plugin's KEY=VALUE environment entries are set inside the container and cannot be clobbered by the job env; each step now pins the collection-hardening branch directly. Drop with the other TEMP commits before merging. Co-Authored-By: Claude Fable 5 --- .buildkite/pipeline.yml | 76 +++++++++++++++++++++++++++-------------- 1 file changed, 51 insertions(+), 25 deletions(-) diff --git a/.buildkite/pipeline.yml b/.buildkite/pipeline.yml index cefe038d..874c5467 100755 --- a/.buildkite/pipeline.yml +++ b/.buildkite/pipeline.yml @@ -1,14 +1,6 @@ agents: queue: "k8s-m6ixlarge" -# TEMP (DO NOT MERGE): the agent environment supplies -# CANDIDATE_COLLECTION_REF=git+...,main from the pipeline secret, which -# force-installs collection main over the requirements.yml branch pin (seen -# in build 445 -- it validated main, not the candidate). Pin the candidate -# here so every lane tests PR #152's branch; drop together with the -# requirements TEMP commit before merging. -env: - CANDIDATE_COLLECTION_REF: "git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening" steps: - label: aws ubuntu @@ -29,7 +21,9 @@ steps: - AWS_SECRET_ACCESS_KEY - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: aws ubuntu sasl key: aws-up-ubuntu-sasl @@ -48,7 +42,9 @@ steps: - AWS_ACCESS_KEY_ID - AWS_SECRET_ACCESS_KEY - AWS_DEFAULT_REGION - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - label: aws ubuntu tiered key: aws-up-ubuntu-tiered concurrency_group: aws-ub @@ -68,7 +64,9 @@ steps: - REDPANDA_LICENSE - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: aws ubuntu tiered large key: aws-up-ubuntu-ts-large @@ -89,7 +87,9 @@ steps: - REDPANDA_LICENSE - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: aws fedora key: aws-up-fedora @@ -109,7 +109,9 @@ steps: - AWS_SECRET_ACCESS_KEY - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: aws fedora connect key: aws-up-fed-con @@ -130,7 +132,9 @@ steps: - AWS_DEFAULT_REGION - CONNECT_RPM_TOKEN - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: aws fedora tiered key: aws-up-fedora-tiered @@ -151,7 +155,9 @@ steps: - REDPANDA_LICENSE - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: aws fedora tiered connect key: aws-up-fed-cts @@ -173,7 +179,9 @@ steps: - REDPANDA_LICENSE - CONNECT_RPM_TOKEN - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: aws fedora tiered large key: aws-up-fedora-ts-large @@ -194,7 +202,9 @@ steps: - REDPANDA_LICENSE - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: gcp ubuntu basic key: gcp-up-ubuntu @@ -212,7 +222,9 @@ steps: environment: - GCP_CREDS - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: gcp ubuntu tiered key: gcp-up-ubuntu-tiered @@ -232,7 +244,9 @@ steps: - GCP_CREDS - REDPANDA_LICENSE - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: gcp fedora basic key: gcp-up-fedora @@ -250,7 +264,9 @@ steps: environment: - GCP_CREDS - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: gcp fedora tiered key: gcp-up-fedora-tiered @@ -269,7 +285,9 @@ steps: - GCP_CREDS - REDPANDA_LICENSE - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: unstable aws fedora tiered key: aws-us-fedora-tiered @@ -290,7 +308,9 @@ steps: - REDPANDA_LICENSE - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: unstable aws fedora tiered large key: aws-us-fedora-ts-large @@ -311,7 +331,9 @@ steps: - REDPANDA_LICENSE - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: unstable aws ubuntu tiered key: aws-us-ubuntu-tiered @@ -332,7 +354,9 @@ steps: - REDPANDA_LICENSE - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: unstable aws ubuntu tiered large key: aws-us-ubuntu-ts-large @@ -353,7 +377,9 @@ steps: - REDPANDA_LICENSE - AWS_DEFAULT_REGION - BASELINE_COLLECTION_REF - - CANDIDATE_COLLECTION_REF + # TEMP (DO NOT MERGE): the aws-sm secret exports git+...,main into the + # job env AFTER yaml env applies; only an in-container pin wins + - CANDIDATE_COLLECTION_REF=git+https://github.com/redpanda-data/redpanda-ansible-collection.git,collection-hardening - REDPANDA_VERSION - label: cleanup aws resources From 062de53566ca5c6abfb74c017cdefb47bec90417 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Wed, 12 Aug 2026 17:19:41 -0500 Subject: [PATCH 11/13] ci: drop the ansible-lint GitHub Action Lint for the redpanda.cluster collection now runs in the collection repo's own CI (production profile, pinned tooling); linting it again from the consuming harness double-reports the same findings against whichever collection version happens to be pinned here. Co-Authored-By: Claude Fable 5 --- .github/workflows/ansible-lint.yaml | 16 ---------------- 1 file changed, 16 deletions(-) delete mode 100644 .github/workflows/ansible-lint.yaml diff --git a/.github/workflows/ansible-lint.yaml b/.github/workflows/ansible-lint.yaml deleted file mode 100644 index 81f5709d..00000000 --- a/.github/workflows/ansible-lint.yaml +++ /dev/null @@ -1,16 +0,0 @@ -name: ansible-lint -on: [ pull_request ] - -jobs: - build: - name: Ansible Lint - runs-on: ubuntu-latest - - steps: - - name: Checkout code - uses: actions/checkout@v4 - with: - fetch-depth: 0 - - - name: Ansible Lint - uses: ansible/ansible-lint@main From a4806a7abd2e5a1d8f4ee5550616652f099ddef4 Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Wed, 12 Aug 2026 17:44:38 -0500 Subject: [PATCH 12/13] fix: wait for cluster health before draining the next node in the rolling restart With serial: 1 the per-host entry health check runs immediately after the previous node rejoined, while partitions are still healing -- the one-shot probe read 'Healthy: false' and failed the play (build 447's sasl lane). The entry gate now uses the same --watch --exit-when-healthy + retry treatment as the post-restart checks. Co-Authored-By: Claude Fable 5 --- ansible/operation-rolling-restart.yml | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/ansible/operation-rolling-restart.yml b/ansible/operation-rolling-restart.yml index 317deeaa..c8bdd902 100644 --- a/ansible/operation-rolling-restart.yml +++ b/ansible/operation-rolling-restart.yml @@ -11,11 +11,17 @@ environment: "{{ {'RPK_USER': sasl_user | default('admin'), 'RPK_PASS': sasl_password | default('')} if (kafka_enable_authorization | default(false) | bool) else {} }}" tasks: + # With serial: 1 this gate runs right after the previous node rejoined; + # the cluster is often still healing, so wait for health like the + # post-restart checks do instead of failing on a one-shot probe. - name: Check cluster health ansible.builtin.shell: | - {{ rpk_bin }} cluster health | grep -i 'healthy:' | tr -d '[:space:]' | awk -F ':' '{print tolower($2)}' + {{ rpk_bin }} cluster health --watch --exit-when-healthy | grep -i 'healthy:' | tr -d '[:space:]' | awk -F ':' '{print tolower($2)}' register: health_check failed_when: "health_check.stdout != 'true'" + retries: 10 + delay: 30 + until: "health_check.stdout == 'true'" changed_when: false - name: Get node ID From 618dab850240303426a5e7ecd4b0ae72a418e22f Mon Sep 17 00:00:00 2001 From: gene-redpanda <123959009+gene-redpanda@users.noreply.github.com> Date: Thu, 13 Aug 2026 09:45:30 -0500 Subject: [PATCH 13/13] fix: prefetch the node_exporter tarball with retries and pin its version The geerlingguy.node_exporter role streams the tarball from GitHub on every host with no retry, and resolves 'latest' via a GitHub API call per host -- under concurrent-lane rate limiting GitHub resets those connections, which has now killed five separate CI lanes across builds 445-448 (including one that outlived the lane-level retry). Both monitor playbooks now prefetch the tarball via get_url with 10 retries and point the role at the local file, pinning 1.12.1 (which also skips the per-host latest lookup). Co-Authored-By: Claude Fable 5 --- ansible/deploy-monitor-tls.yml | 17 +++++++++++++++++ ansible/deploy-monitor.yml | 17 +++++++++++++++++ 2 files changed, 34 insertions(+) diff --git a/ansible/deploy-monitor-tls.yml b/ansible/deploy-monitor-tls.yml index c99a21af..ee6f8dc0 100644 --- a/ansible/deploy-monitor-tls.yml +++ b/ansible/deploy-monitor-tls.yml @@ -7,11 +7,28 @@ - name: Set node_exporter_arch based on host architecture ansible.builtin.set_fact: rp_arch: "{{ 'amd64' if ansible_architecture == 'x86_64' else 'arm64' }}" + # Prefetched with retries because GitHub resets connections under + # concurrent-lane rate limiting (five separate CI lane failures), and the + # role's own download/latest-lookup hit GitHub per host with no retry. + # Pinning the version also skips the role's 'latest' GitHub API lookup. + - name: Prefetch node_exporter tarball with retries + ansible.builtin.get_url: + url: "https://github.com/prometheus/node_exporter/releases/download/v{{ node_exporter_pinned_version }}/node_exporter-{{ node_exporter_pinned_version }}.linux-{{ rp_arch }}.tar.gz" + dest: "/tmp/node_exporter-{{ node_exporter_pinned_version }}.linux-{{ rp_arch }}.tar.gz" + mode: "0644" + register: node_exporter_prefetch + until: node_exporter_prefetch is succeeded + retries: 10 + delay: 15 + vars: + node_exporter_pinned_version: "1.12.1" - name: Include geerlingguy nodeexporter ansible.builtin.include_role: name: geerlingguy.node_exporter vars: node_exporter_arch: "{{ rp_arch }}" + node_exporter_version: "1.12.1" + node_exporter_download_url: "/tmp/node_exporter-1.12.1.linux-{{ rp_arch }}.tar.gz" - name: Install certs on monitor hosts: monitor diff --git a/ansible/deploy-monitor.yml b/ansible/deploy-monitor.yml index 52eb635b..be139ed3 100644 --- a/ansible/deploy-monitor.yml +++ b/ansible/deploy-monitor.yml @@ -5,11 +5,28 @@ - name: Set node_exporter_arch based on host architecture ansible.builtin.set_fact: rp_arch: "{{ 'amd64' if ansible_architecture == 'x86_64' else 'arm64' }}" + # Prefetched with retries because GitHub resets connections under + # concurrent-lane rate limiting (five separate CI lane failures), and the + # role's own download/latest-lookup hit GitHub per host with no retry. + # Pinning the version also skips the role's 'latest' GitHub API lookup. + - name: Prefetch node_exporter tarball with retries + ansible.builtin.get_url: + url: "https://github.com/prometheus/node_exporter/releases/download/v{{ node_exporter_pinned_version }}/node_exporter-{{ node_exporter_pinned_version }}.linux-{{ rp_arch }}.tar.gz" + dest: "/tmp/node_exporter-{{ node_exporter_pinned_version }}.linux-{{ rp_arch }}.tar.gz" + mode: "0644" + register: node_exporter_prefetch + until: node_exporter_prefetch is succeeded + retries: 10 + delay: 15 + vars: + node_exporter_pinned_version: "1.12.1" - name: Include geerlingguy nodeexporter ansible.builtin.include_role: name: geerlingguy.node_exporter vars: node_exporter_arch: "{{ rp_arch }}" + node_exporter_version: "1.12.1" + node_exporter_download_url: "/tmp/node_exporter-1.12.1.linux-{{ rp_arch }}.tar.gz" - name: Install prometheus hosts: monitor