diff --git a/postgresql/cutover/test/pgha-dryrun.yaml b/postgresql/cutover/test/pgha-dryrun.yaml index 7045b11..e775f52 100644 --- a/postgresql/cutover/test/pgha-dryrun.yaml +++ b/postgresql/cutover/test/pgha-dryrun.yaml @@ -4,15 +4,83 @@ # Validates the ADR-0001 cutover mechanics WITHOUT touching production: # - own overlay network (pgha-test_db-backend), own stack name # - throwaway "legacy" postgres:17 seeded with marker data, carrying the -# in-network alias "postgresql" (mirrors prod CLONE_HOST) -# - same etcd/patroni/haproxy topology, images, and env wiring as the -# real staging file — including clone-from-live via pg_basebackup +# in-network alias "postgresql" (mirrors prod CLONE_HOST/standby_cluster +# host) +# - same etcd/patroni/haproxy topology and images as the real staging file # - reuses the real Docker secrets (read-only mounts; harmless) # - disposable data dirs under /volume1/docker/PostgreSQL/dryrun/ # # Deploy: docker stack deploy -c pgha-test # Teardown: docker stack rm pgha-test && rm -rf /volume1/docker/PostgreSQL/dryrun # +# ─── REVISION (2026-08-01): switched from CLONE_WITH_BASEBACKUP to a +# Patroni "standby cluster" (continuous streaming) — see rationale below ─── +# +# WHY THIS CHANGED: the original CLONE_WITH_BASEBACKUP design (still used by +# postgresql/postgresql.yaml and postgresql/cutover/postgresql-ha-staging.yaml +# as of this revision — NOT YET ported here) is a ONE-SHOT snapshot: once +# pg_basebackup completes, the new cluster has ZERO further connection to +# the old database. Any write landing on the old DB between the snapshot +# and the actual traffic cutover is silently lost — a real gap given ~13 +# active consumers and a 42GB production database. Patroni's "standby +# cluster" mode (https://patroni.readthedocs.io/en/latest/standby_cluster.html) +# instead makes the new cluster's leader ("standby leader") continuously +# stream from the remote primary indefinitely, right up until an explicit +# promotion — closing that gap to near-zero. This file is the dry-run +# validation of that mode, BEFORE porting it to the real staging/production +# files. Do not port until this dry run passes end-to-end. +# +# KEY DIFFERENCES FROM THE CLONE-BASED DESIGN: +# - No CLONE_METHOD/CLONE_SCOPE/CLONE_HOST/CLONE_PORT/CLONE_USER/ +# CLONE_PASSWORD anywhere. Replaced by a `bootstrap.dcs.standby_cluster` +# block inside SPILO_CONFIGURATION on BOTH patroni-0 and patroni-1 — +# both get it, not just one, because standby_cluster config is written +# ONCE into shared DCS state by whichever node wins the initial +# bootstrap race (same non-determinism previously handled the same way +# for CLONE_*). Per Patroni docs: "these options will be applied only +# once during cluster bootstrap, and the only way to change them +# afterwards is through DCS." +# - `legacy` now needs a REAL replication role named "standby" (matching +# PGUSER_STANDBY), not just the PGadmin superuser. CLONE_WITH_BASEBACKUP +# ran pg_basebackup as CLONE_USER=PGadmin (a superuser, so no dedicated +# role was needed); standby_cluster streaming instead authenticates +# using the cluster's own replication identity (PGUSER_STANDBY/ +# PGPASSWORD_STANDBY), which never existed on "legacy" before — added +# via a second initdb.d hook, reading the password from the SAME +# postgresql_replication_password secret already used between +# patroni-0/patroni-1 for their own internal replication. +# - patroni-1 no longer references CLONE_* either (it never talked to +# legacy directly in the old design either — it bootstrapped from +# whichever node held the leader lock, exactly as a Patroni "cascade +# replica" does in standby-cluster mode too — this is actually simpler +# now, not different). +# - create_replica_methods references "basebackup_fast_xlog" — this is +# Spilo's OWN generated method key (confirmed from configure_spilo.py's +# TEMPLATE constant, fetched directly from zalando/spilo source this +# project), not the generic "basebackup" name shown in Patroni's own +# docs example (that name assumes a plain Patroni method map without +# Spilo's wrapper naming). +# +# OPEN QUESTIONS THIS DRY RUN IS DESIGNED TO ANSWER (do not assume the +# answer — observe actual logs): +# - Does Patroni require a pre-created replication slot on "legacy" for +# the standby_cluster link (primary_slot_name), or does it work +# without one since we haven't set that key? Docs say slots are only +# required "if you use replication slots" for this link — ambiguous +# whether that's opt-in or on-by-default given use_slots:true is set +# cluster-wide by Spilo's own template for OTHER purposes. +# - Does "legacy" (vanilla, unmanaged postgres:17) actually need +# postgresql.conf discoverable in PGDATA for Patroni's remote checks? +# (Docs: "Patroni expects to find postgresql.conf or +# postgresql.conf.backup in PGDATA of the remote primary" — vanilla +# images keep it there by default, expected to be fine, but confirm.) +# - Does promotion off the standby cluster (detaching from "legacy" and +# becoming a normal read-write cluster) work cleanly and pick up the +# very latest streamed WAL, closing the gap as intended? This is the +# actual feature under test — Step 7c below. +# +# ─── Fixes still carried forward from the CLONE_WITH_BASEBACKUP dry run ─── +# # NOTE: command: blocks use $$(...) not $(...) — Compose's own variable # interpolation parses $( as an attempted ${VAR} reference and fails with # "invalid interpolation format" / "you may need to escape any $ with @@ -31,11 +99,12 @@ # v3-API client. # # NOTE: "legacy" needs a pg_hba.conf rule permitting REPLICATION-type -# connections (distinct from normal client connections), fixed via a -# /docker-entrypoint-initdb.d/ hook script (official extension point). -# "trust" is acceptable ONLY because this container is fully disposable. -# See note "ADR-0001 Addendum — pg_hba.conf replication prerequisite -# discovered in dry run" for the real production prerequisite this exposes. +# connections (distinct from normal client connections) — this is STILL +# required under standby_cluster mode, since it's still a genuine +# replication connection under the hood, just a continuous one instead of +# one-shot. Fixed via a /docker-entrypoint-initdb.d/ hook script (official +# extension point). "trust" is acceptable ONLY because this container is +# fully disposable. # # NOTE: patroni-0/patroni-1 override bootstrap.post_init via # SPILO_CONFIGURATION (Spilo's documented, supported mechanism for @@ -43,34 +112,22 @@ # user-supplied SPILO_CONFIGURATION on top of its own generated config). # This is needed because Spilo's own /scripts/post_init.sh hardcodes # "ALTER VIEW ... OWNER TO postgres" with no way to parameterize that role -# name via env vars. Since our superuser is PGUSER_SUPERUSER=PGadmin (to -# match production), there is no role literally named "postgres" in the -# cloned data, so Spilo's unmodified script fails with -# 'ERROR: role "postgres" does not exist'. The override points -# bootstrap.post_init at our own wrapper script instead, which creates a -# harmless, idempotent, NOLOGIN "postgres" role first, then execs Spilo's -# real, UNMODIFIED post_init.sh with all original arguments passed through -# — Zalando's script itself is never patched or forked. This affects real -# production too (superuser there is also PGadmin, not postgres) and must -# be carried into the real staging/production compose files as well. -# -# NOTE: the placeholder "postgres" role created by the wrapper above must -# be a genuine SUPERUSER (not just exist) — confirmed by tracing the -# failure past the ALTER VIEW step: Spilo's post_init.sh unconditionally -# cats and pipes in Zalando's own _zmon_schema.dump, whose first lines are -# "RESET ROLE; SET ROLE TO postgres;" followed by -# "CREATE EXTENSION IF NOT EXISTS plpython3u", which requires superuser -# privileges. A plain NOLOGIN role satisfies SET ROLE (membership-based) -# but not the subsequent superuser-only extension creation. Since the role -# is NOLOGIN, granting SUPERUSER carries no real exposure — it can never be -# used to establish a connection. This must also be carried into the real -# staging/production compose files alongside the post_init override above. +# name via env vars, AND its embedded _zmon_schema.dump does +# "SET ROLE TO postgres; CREATE EXTENSION plpython3u" which requires +# genuine superuser. Our wrapper creates a harmless, idempotent +# NOLOGIN+SUPERUSER "postgres" role first (SUPERUSER carries no real +# exposure since NOLOGIN means it can never authenticate a connection), +# then execs Spilo's real, UNMODIFIED post_init.sh with all original +# arguments passed through — Zalando's script itself is never patched or +# forked. This is UNRELATED to the standby_cluster change and still +# required for the same reasons as before. # ───────────────────────────────────────────────────────────────────────── version: "3.6" services: - # Stand-in for the production single-instance postgres (clone source) + # Stand-in for the production single-instance postgres (remote primary + # for standby_cluster streaming) legacy: image: public.ecr.aws/docker/library/postgres:17 hostname: db @@ -83,12 +140,16 @@ services: echo "host replication all all trust" >> "$$PGDATA/pg_hba.conf" EOF chmod +x /docker-entrypoint-initdb.d/zz-enable-replication.sh + cat > /docker-entrypoint-initdb.d/zz-create-standby-role.sql < /scripts/post_init_wrapper.sh <<'WRAP' #!/bin/bash @@ -198,14 +258,22 @@ services: PGUSER_STANDBY: standby PATRONI_RESTAPI_USERNAME: patroni PGROOT: /home/postgres/pgdata/pgroot - CLONE_METHOD: CLONE_WITH_BASEBACKUP - CLONE_SCOPE: legacy-single - CLONE_HOST: postgresql - CLONE_PORT: "5432" - CLONE_USER: PGadmin + # Standby-cluster mode: continuous streaming from "legacy" instead of + # a one-shot CLONE_WITH_BASEBACKUP snapshot. Written once into shared + # DCS state by whichever of patroni-0/patroni-1 wins the initial + # bootstrap race — both nodes carry the identical block for that + # reason. "postgresql" is legacy's network alias (mirrors prod + # CLONE_HOST naming). basebackup_fast_xlog is Spilo's own generated + # replica-method key (confirmed from configure_spilo.py source). SPILO_CONFIGURATION: | bootstrap: post_init: /scripts/post_init_wrapper.sh "zalandos" + dcs: + standby_cluster: + host: postgresql + port: 5432 + create_replica_methods: + - basebackup_fast_xlog secrets: - postgresql_password - postgresql_replication_password @@ -229,7 +297,6 @@ services: export PGPASSWORD_SUPERUSER="$$(cat /run/secrets/postgresql_password)" export PGPASSWORD_STANDBY="$$(cat /run/secrets/postgresql_replication_password)" export PATRONI_RESTAPI_PASSWORD="$$(cat /run/secrets/postgresql_patroni_password)" - export CLONE_PASSWORD="$$(cat /run/secrets/postgresql_password)" mkdir -p /scripts cat > /scripts/post_init_wrapper.sh <<'WRAP' #!/bin/bash @@ -247,14 +314,17 @@ services: PGUSER_STANDBY: standby PATRONI_RESTAPI_USERNAME: patroni PGROOT: /home/postgres/pgdata/pgroot - CLONE_METHOD: CLONE_WITH_BASEBACKUP - CLONE_SCOPE: legacy-single - CLONE_HOST: postgresql - CLONE_PORT: "5432" - CLONE_USER: PGadmin + # Same standby_cluster block as patroni-0 — see comment there for why + # both nodes carry it identically. SPILO_CONFIGURATION: | bootstrap: post_init: /scripts/post_init_wrapper.sh "zalandos" + dcs: + standby_cluster: + host: postgresql + port: 5432 + create_replica_methods: + - basebackup_fast_xlog secrets: - postgresql_password - postgresql_replication_password