Compare commits
24
Commits
d72a8ebd04
..
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c3c12ef0f9 | ||
|
|
a2308b76f8 | ||
|
|
0a334fa781 | ||
|
|
8c2ee593d4 | ||
|
|
a2df135693 | ||
|
|
9895249e88 | ||
|
|
9ec1b1811b | ||
|
|
8fa085ebde | ||
|
|
8ef1cea5b0 | ||
|
|
409971f741 | ||
|
|
6ddb31c594 | ||
|
|
4f4c5a467b | ||
|
|
e27d2f6ac2 | ||
|
|
5ab1ae1e8b | ||
|
|
afc96a6d37 | ||
|
|
cd42e35265 | ||
|
|
2a3f2bdd1e | ||
|
|
d9f151a1c3 | ||
|
|
201d8418ee | ||
|
|
5edebd977d | ||
|
|
ffec3ea5b7 | ||
|
|
5dda5e0031 | ||
|
|
d5db4c1a4e | ||
|
|
333bb82c16 |
+27
-1
@@ -70,6 +70,16 @@ when:
|
||||
# tolerating a failing stack-ps in verify (|| true) so a genuinely missing
|
||||
# stack produces the designed WARNING instead of killing the step. This
|
||||
# hazard was first flagged in July (PR #3, closed unmerged).
|
||||
#
|
||||
# 2026-09-12 MIGRATION: traefik moved from Pattern B (secrets hand-typed
|
||||
# into the host-only traefik.env, no CI involvement at all) to the same
|
||||
# manifest-driven provisioning as ai) below. Root cause of the migration:
|
||||
# traefik.env had been accidentally committed to git with a literal
|
||||
# "***REDACTED***" placeholder as KEEPALIVED_PASSWORD; every git-guard
|
||||
# resync/checkout silently restored that broken value, which diverged from
|
||||
# keepalived-backup's stale-but-correct in-memory value and produced a
|
||||
# continuous VRRP auth failure ("received an invalid passwd!") and VIP
|
||||
# instability. The manifest-driven path never commits the rendered file.
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
steps:
|
||||
@@ -216,6 +226,11 @@ steps:
|
||||
from_secret: flowagent_azure_tenant_id
|
||||
FLOWAGENT_AZURE_CLIENT_SECRET:
|
||||
from_secret: flowagent_azure_client_secret
|
||||
# ── traefik stack (manifest-driven — secrets/secrets-map.yaml +
|
||||
# traefik/traefik.env.template + deploy/provision-stack.py).
|
||||
# Migrated 2026-09-12; see header note above for root cause. ──
|
||||
TRAEFIK_KEEPALIVED_PASSWORD:
|
||||
from_secret: traefik_keepalived_password
|
||||
commands:
|
||||
- apk add --no-cache openssh-client python3 py3-yaml
|
||||
- mkdir -p ~/.ssh
|
||||
@@ -243,8 +258,19 @@ steps:
|
||||
for STACK in $ALL_STACKS; do
|
||||
echo "Provisioning: $STACK"
|
||||
case "$STACK" in
|
||||
maintenance|media|unifi|guacamole|security|auth|traefik|meshcentral|ddm)
|
||||
maintenance|media|unifi|guacamole|security|auth|meshcentral|ddm)
|
||||
echo " No Docker secrets for $STACK — secrets in host .env";;
|
||||
traefik)
|
||||
# MIGRATED (2026-09-12) to manifest-driven provisioning after
|
||||
# discovering traefik.env was committed to git with a literal
|
||||
# "***REDACTED***" placeholder as KEEPALIVED_PASSWORD — every
|
||||
# git-guard resync/checkout restored the broken value, causing
|
||||
# VRRP auth mismatch between keepalived-master/-backup and VIP
|
||||
# instability. All logic lives in deploy/provision-stack.py;
|
||||
# the authoritative key list lives in
|
||||
# traefik/traefik.env.template; the mapping lives in
|
||||
# secrets/secrets-map.yaml. This case is intentionally one line.
|
||||
python3 deploy/provision-stack.py traefik;;
|
||||
immich)
|
||||
ssh -o StrictHostKeyChecking=no root@$${SWARM_MANAGER_IP} "source /tmp/cs.sh
|
||||
create_or_update_secret 'immich_db_password' '$${IMMICH_DB_PASSWORD}'
|
||||
|
||||
+115
-2
@@ -23,18 +23,131 @@
|
||||
# remote both moved) backup/stash/reset recovery steps and exits.
|
||||
#
|
||||
# Exit codes: 0 = safe to deploy, 1 = blocked, needs human intervention
|
||||
#
|
||||
# ── Flags (added 2026-09-12, circular-dependency bootstrap incident) ────────
|
||||
#
|
||||
# -e, --emergency
|
||||
# Before the normal fetch, tries each candidate Gitea endpoint in order
|
||||
# (git.bryanmail.net -> 192.168.4.30 VIP -> .31 -> .32 -> .33 node IPs,
|
||||
# each over plain http on port 3000, which is Gitea's direct ingress
|
||||
# port — bypasses Traefik/VIP entirely) and switches `origin` to the
|
||||
# first one that responds to `git ls-remote` within a short timeout.
|
||||
# This solves the bootstrap circular dependency where Traefik is down,
|
||||
# so HTTPS access to git.bryanmail.net is unreachable, so git-guard
|
||||
# can't fetch, so stack-deploy.sh can't redeploy traefik to fix itself.
|
||||
# Once a reachable endpoint is found, NORMAL sync logic still runs
|
||||
# (behind/ahead/diverged handling is unchanged) — this flag only changes
|
||||
# *which host* is used, never skips the safety checks themselves.
|
||||
# Prints a loud reminder to restore the real origin URL afterward; never
|
||||
# commits the swapped URL anywhere.
|
||||
#
|
||||
# -f, --force
|
||||
# Skips the sync check ENTIRELY — no fetch, no comparison, no commit/push
|
||||
# offer. Deploys whatever is on disk right now, as-is. This is the last
|
||||
# resort for a genuine emergency where NONE of the candidate hosts in
|
||||
# --emergency are reachable (e.g. Gitea itself is down, not just
|
||||
# routing). Prints a loud warning banner. Never use this for routine
|
||||
# work — it defeats the entire purpose of this script and is the exact
|
||||
# failure mode (deploying a stale/unreviewed tree) git-guard exists to
|
||||
# prevent.
|
||||
#
|
||||
# Both flags are passed through from stack-deploy.sh's own -e/-f flags;
|
||||
# see that script's header for the calling convention.
|
||||
# ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
DIR="/volume1/docker/compose-files"
|
||||
cd "$DIR"
|
||||
|
||||
# ---- Flag parsing ----
|
||||
EMERGENCY=0
|
||||
FORCE=0
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
-e|--emergency) EMERGENCY=1 ;;
|
||||
-f|--force) FORCE=1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
# ---- Force mode: skip everything ----
|
||||
if [ "$FORCE" -eq 1 ]; then
|
||||
echo "!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!"
|
||||
echo "!! FORCE MODE (-f/--force): git-guard sync check SKIPPED ENTIRELY."
|
||||
echo "!! Deploying whatever is on disk right now, as-is. No fetch, no"
|
||||
echo "!! comparison with origin/main was performed. This is a LAST RESORT"
|
||||
echo "!! for emergencies where origin is completely unreachable — verify"
|
||||
echo "!! independently that the local tree is what you intend to deploy."
|
||||
echo "!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Non-interactive detection (Woodpecker/cron have no TTY on stdin)
|
||||
INTERACTIVE=0
|
||||
[ -t 0 ] && INTERACTIVE=1
|
||||
|
||||
echo "==> git-guard: checking repo sync state"
|
||||
|
||||
# ---- Emergency mode: find a reachable Gitea endpoint before fetching ----
|
||||
# Candidate order: public hostname (normal path) -> VIP -> each node's direct
|
||||
# IP. Each is tried over plain http on port 3000 (Gitea's direct ingress
|
||||
# port, published outside Traefik — see traefik/traefik.yaml git service
|
||||
# port mapping), since the whole point is to bypass Traefik/VIP when THOSE
|
||||
# are what's broken. A short `git ls-remote` timeout keeps an unreachable
|
||||
# candidate from stalling the whole check for long.
|
||||
if [ "$EMERGENCY" -eq 1 ]; then
|
||||
echo "!! EMERGENCY MODE (-e/--emergency): probing candidate Gitea endpoints"
|
||||
echo "!! (bypassing the normal https://git.bryanmail.net path if needed)..."
|
||||
|
||||
ORIGINAL_URL="$(git remote get-url origin)"
|
||||
CANDIDATES=(
|
||||
"https://git.bryanmail.net/homelab/compose-files.git"
|
||||
"http://192.168.4.30:3000/homelab/compose-files.git"
|
||||
"http://192.168.4.31:3000/homelab/compose-files.git"
|
||||
"http://192.168.4.32:3000/homelab/compose-files.git"
|
||||
"http://192.168.4.33:3000/homelab/compose-files.git"
|
||||
)
|
||||
|
||||
FOUND=""
|
||||
for candidate in "${CANDIDATES[@]}"; do
|
||||
echo -n " trying $candidate ... "
|
||||
if timeout 5 git ls-remote "$candidate" HEAD >/dev/null 2>&1; then
|
||||
echo "OK"
|
||||
FOUND="$candidate"
|
||||
break
|
||||
else
|
||||
echo "unreachable"
|
||||
fi
|
||||
done
|
||||
|
||||
if [ -z "$FOUND" ]; then
|
||||
echo "!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!"
|
||||
echo "!! EMERGENCY MODE: none of the candidate endpoints responded."
|
||||
echo "!! Gitea itself may be down (not just routing) — this is beyond what"
|
||||
echo "!! an alternate host path can fix. Options:"
|
||||
echo "!! A) Diagnose Gitea directly: check the git_gitea-server service"
|
||||
echo "!! and container on docker-1."
|
||||
echo "!! B) If you are certain the on-disk tree is correct and Gitea is"
|
||||
echo "!! genuinely unreachable, re-run with -f/--force instead — but"
|
||||
echo "!! read that flag's warning carefully first."
|
||||
echo "!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ "$FOUND" != "$ORIGINAL_URL" ]; then
|
||||
git remote set-url origin "$FOUND"
|
||||
echo "==> origin temporarily switched to: $FOUND"
|
||||
echo "!! REMINDER: once the normal path (git.bryanmail.net / Traefik) is"
|
||||
echo "!! confirmed healthy again, restore the real origin URL:"
|
||||
echo "!! cd $DIR && git remote set-url origin \"$ORIGINAL_URL\""
|
||||
echo "!! This swap is never committed anywhere and only affects this"
|
||||
echo "!! local checkout's git config."
|
||||
else
|
||||
echo "==> Normal origin URL ($FOUND) is reachable — no swap needed."
|
||||
fi
|
||||
echo
|
||||
fi
|
||||
|
||||
git fetch origin --quiet
|
||||
|
||||
LOCAL="$(git rev-parse main)"
|
||||
@@ -218,7 +331,7 @@ if [ "$DIRTY" -eq 1 ]; then
|
||||
echo "==> Resync succeeded. Reapplying stashed changes..."
|
||||
if git stash pop; then
|
||||
echo "==> Stash reapplied cleanly. Re-checking sync state..."
|
||||
exec "$0" "$@"
|
||||
exec bash "$0" "$@"
|
||||
else
|
||||
echo "ERROR: 'git stash pop' did not complete successfully."
|
||||
echo
|
||||
@@ -308,7 +421,7 @@ if [ "$DIRTY" -eq 1 ]; then
|
||||
|
||||
if git push origin main; then
|
||||
echo "==> Pushed. Re-checking sync state..."
|
||||
exec "$0" "$@"
|
||||
exec bash "$0" "$@"
|
||||
else
|
||||
echo "ERROR: push failed (likely diverged from origin). Aborting deploy."
|
||||
echo "Run:"
|
||||
|
||||
+28
-2
@@ -38,10 +38,36 @@
|
||||
# pre-escaping) or values get doubled twice. Getting this wrong
|
||||
# silently corrupts any secret/hash containing '$' (confirmed impact:
|
||||
# LITELLM keys truncated, IMMICH_KIOSK_BASICAUTH bcrypt hash mismatched).
|
||||
#
|
||||
# Usage: stack-deploy.sh [-e|--emergency] [-f|--force] <stack-name>
|
||||
# Flags may appear before or after the stack name, e.g. both
|
||||
# `stack-deploy.sh traefik -e` and `stack-deploy.sh -e traefik` work.
|
||||
#
|
||||
# -e/--emergency and -f/--force (added 2026-09-12, circular-dependency
|
||||
# bootstrap incident — see deploy/git-guard.sh header for full detail):
|
||||
# Both are passed straight through to git-guard.sh unchanged; this script
|
||||
# does not interpret them itself beyond stripping them from the stack-name
|
||||
# argument list. -e tries alternate Gitea endpoints (VIP, then each node's
|
||||
# direct IP) before falling back to normal sync logic against whichever
|
||||
# one responds. -f skips the sync check entirely — last resort only, read
|
||||
# the warning banner it prints. Neither flag changes anything about the
|
||||
# render/mount-guard/deploy steps below; they only affect whether and how
|
||||
# git-guard.sh's pre-flight check runs.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
STACK="${1:?Usage: stack-deploy.sh <stack-name>}"
|
||||
# ── Flag parsing (stack name is whatever's left after flags are stripped) ──
|
||||
GUARD_FLAGS=()
|
||||
STACK=""
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
-e|--emergency) GUARD_FLAGS+=(-e) ;;
|
||||
-f|--force) GUARD_FLAGS+=(-f) ;;
|
||||
*) STACK="$arg" ;;
|
||||
esac
|
||||
done
|
||||
: "${STACK:?Usage: stack-deploy.sh [-e|--emergency] [-f|--force] <stack-name>}"
|
||||
|
||||
DIR="/volume1/docker/compose-files"
|
||||
PY="$DIR/deploy/envparse.py"
|
||||
MOUNT_GUARD="$DIR/deploy/mount-guard.py"
|
||||
@@ -50,7 +76,7 @@ GLOBAL_ENV="$DIR/deploy/global.env"
|
||||
# ── Pre-flight: ensure local checkout is in sync with Gitea ─────────────────
|
||||
# Prevents deploying from a stale/diverged local tree (see incident 2026-08-26).
|
||||
# Invoked via `bash` explicitly so the tracked file's exec bit doesn't matter.
|
||||
bash "$DIR/deploy/git-guard.sh" || { echo "ERROR: git-guard check failed. Deploy aborted."; exit 1; }
|
||||
bash "$DIR/deploy/git-guard.sh" "${GUARD_FLAGS[@]:-}" || { echo "ERROR: git-guard check failed. Deploy aborted."; exit 1; }
|
||||
|
||||
# ── Locate compose file(s) ──────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -63,6 +63,15 @@ services:
|
||||
FRIGATE_RTSP_PASSWORD: "${FRIGATE_RTSP_PASSWORD}"
|
||||
networks:
|
||||
- traefik_backend
|
||||
ports:
|
||||
- target: 8555
|
||||
published: 8555
|
||||
protocol: udp
|
||||
mode: ingress
|
||||
- target: 8555
|
||||
published: 8555
|
||||
protocol: tcp
|
||||
mode: ingress
|
||||
deploy:
|
||||
labels:
|
||||
- traefik.enable=true
|
||||
|
||||
+18
-5
@@ -6,9 +6,9 @@ This directory documents the secrets required for each Docker Swarm stack.
|
||||
|
||||
```
|
||||
Secret values live in Woodpecker (encrypted)
|
||||
↓ pipeline reads them at deploy time
|
||||
→ pipeline reads them at deploy time
|
||||
Docker Swarm secret store (encrypted Raft DB, replicated across all nodes)
|
||||
↓ mounted into containers at runtime
|
||||
→ mounted into containers at runtime
|
||||
/run/secrets/<secret_name>
|
||||
```
|
||||
|
||||
@@ -38,7 +38,10 @@ Use the stack's `.secrets.example` file as your checklist.
|
||||
|
||||
### Step 2 — Add the stack's case to `.woodpecker.yml`
|
||||
In the `provision-secrets` step, add a case for the stack that calls
|
||||
`create_or_update_secret` for each secret.
|
||||
`create_or_update_secret` for each secret. (Or, preferred for new
|
||||
migrations: add an entry to `secrets/secrets-map.yaml` + a
|
||||
`<stack>.env.template` and call `deploy/provision-stack.py <stack>`
|
||||
instead — see the `ai` and `traefik` entries for the current pattern.)
|
||||
|
||||
### Step 3 — Test by pushing a trivial change to the stack's yaml file
|
||||
Watch the pipeline run: provision-secrets → validate → deploy → verify → notify.
|
||||
@@ -71,10 +74,20 @@ Examples:
|
||||
- **PostgreSQL is highest risk.** Its master password is used by nearly every
|
||||
other stack. Migrate it last.
|
||||
|
||||
- **Never commit a rendered env file, even by accident.** `traefik/traefik.env`
|
||||
was committed to git for a period (discovered/fixed 2026-09-12) with a
|
||||
literal "***REDACTED***" placeholder as KEEPALIVED_PASSWORD, which was
|
||||
silently restored every time the file was deleted or the checkout resynced
|
||||
from git — causing a real VRRP auth outage. `.gitignore` blanket-excludes
|
||||
`*.env`, but that rule does NOT retroactively untrack a file already
|
||||
committed before the rule existed. If you ever see a stack's `.env` file
|
||||
show up in `git status` as tracked, stop and untrack it (`git rm --cached`)
|
||||
before doing anything else.
|
||||
|
||||
## Migration Status
|
||||
|
||||
| Stack | Secrets in Woodpecker | Pipeline Step Added | .env Removed |
|
||||
|-------|----------------------|---------------------|--------------|
|
||||
|-------|------------------------|----------------------|---------------|
|
||||
| 3dprint | ⏳ | ⏳ | ⏳ |
|
||||
| ai | ⏳ | ⏳ | ⏳ |
|
||||
| auth | ⏳ | ⏳ | ⏳ |
|
||||
@@ -91,7 +104,7 @@ Examples:
|
||||
| postgresql | ⏳ | ⏳ | ⏳ |
|
||||
| productivity | ⏳ | ⏳ | ⏳ |
|
||||
| security | ⏳ | ⏳ | ⏳ |
|
||||
| traefik | ⏳ | ⏳ | ⏳ |
|
||||
| traefik | ✅ (manifest-driven, 2026-09-12) | ✅ (manifest-driven, 2026-09-12) | ⏳ |
|
||||
| unifi | ✅ N/A (no secrets) | ✅ N/A | ⏳ |
|
||||
| vaultwarden | ⏳ | ⏳ | ⏳ |
|
||||
| woodpecker | ⏳ Manual only | ⏳ N/A | ⏳ |
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# ─────────────────────────────────────────────────────────────────────────
|
||||
# secrets-map.yaml — DATA-ONLY manifest for deploy/provision-stack.py
|
||||
#
|
||||
# RULES:
|
||||
@@ -27,7 +27,7 @@
|
||||
#
|
||||
# Stacks not listed here fall through to deploy.yml's legacy case-entries
|
||||
# untouched. Migration is deliberately one stack per PR.
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# ─────────────────────────────────────────────────────────────────────────
|
||||
stacks:
|
||||
ai:
|
||||
env_template: ai/ai.env.template
|
||||
@@ -36,3 +36,6 @@ stacks:
|
||||
flowagent_azure_client_id: FLOWAGENT_AZURE_CLIENT_ID
|
||||
flowagent_azure_tenant_id: FLOWAGENT_AZURE_TENANT_ID
|
||||
flowagent_azure_client_secret: FLOWAGENT_AZURE_CLIENT_SECRET
|
||||
traefik:
|
||||
env_template: traefik/traefik.env.template
|
||||
env_dest: traefik/traefik.env
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
# traefik Stack — Secrets Reference
|
||||
# Source: traefik.env
|
||||
# Source: traefik.env.template (rendered by deploy/provision-stack.py per
|
||||
# secrets/secrets-map.yaml — see secrets-map.yaml header for how
|
||||
# this works). traefik/traefik.env is rendered fresh on every
|
||||
# provisioning run and is NEVER committed to git.
|
||||
#
|
||||
# Add SECRET values to Woodpecker at:
|
||||
# https://woodpecker.bryanmail.net
|
||||
@@ -8,27 +11,44 @@
|
||||
# ⚠️ HIGH RISK: Traefik is the entry point for all homelab services.
|
||||
# If this stack fails, nothing is reachable from outside.
|
||||
# Migrate carefully. The Keepalived VIP (192.168.4.30) depends on this stack.
|
||||
#
|
||||
# ⚠️ 2026-09-12 INCIDENT: traefik.env was previously committed to git with a
|
||||
# literal "***REDACTED***" placeholder as KEEPALIVED_PASSWORD. Every
|
||||
# git-guard resync/checkout restored that broken value onto disk,
|
||||
# diverging from keepalived-backup's stale-but-correct in-memory value
|
||||
# and causing continuous VRRP auth failures + VIP instability. This was
|
||||
# the trigger for migrating this stack to Pattern C. If you ever see
|
||||
# KEEPALIVED_PASSWORD as a literal placeholder-looking string on disk
|
||||
# again, do NOT hand-edit it — check `git log traefik/` for a stray
|
||||
# commit and fix the template in Gitea instead.
|
||||
|
||||
# ── SECRETS (add to Woodpecker) ───────────────────────────────────────────────
|
||||
# ── SECRETS (add to Woodpecker) ─────────────────────────────────────────
|
||||
|
||||
# Woodpecker secret name: traefik_keepalived_password
|
||||
# Used for: Keepalived VRRP authentication password
|
||||
# Must match across all 3 nodes (docker-1, docker-2, docker-3)
|
||||
# Env var in .env: KEEPALIVED_PASSWORD
|
||||
# NOTE: classic VRRP simple-auth is silently
|
||||
# truncated to 8 chars by keepalived itself — keep
|
||||
# the value <= 8 characters, or be aware only the
|
||||
# first 8 are actually significant on the wire.
|
||||
# Env var in .env.template: KEEPALIVED_PASSWORD (via TRAEFIK_KEEPALIVED_PASSWORD)
|
||||
traefik_keepalived_password=
|
||||
|
||||
# ── NON-SECRETS (safe in compose file or .env) ────────────────────────────────
|
||||
# ── NON-SECRETS (safe in compose file or .env.template) ─────────────────
|
||||
|
||||
# KEEPALIVED_UNICAST_PEERS Python2BASH list of peer IPs
|
||||
# KEEPALIVED_VIRTUAL_IPS Python2BASH list of VIP addresses (192.168.4.30)
|
||||
# KEEPALIVED_VIRTUAL_IPS Python2BASH list of VIP addresses (192.168.4.30) — literal in template
|
||||
# ACME_EMAIL Let's Encrypt certificate email
|
||||
# TRUSTED_IPS Trusted proxy CIDR ranges
|
||||
# TRAEFIK_HOST Traefik dashboard hostname
|
||||
# SPEEDTEST_HOST Speedtest Traefik hostname
|
||||
# WHOAMI_HOST Whoami Traefik hostname
|
||||
|
||||
# ── Woodpecker provision-secrets case entry ───────────────────────────────────
|
||||
# ── Provisioning (manifest-driven, deploy/provision-stack.py) ───────────
|
||||
#
|
||||
# traefik)
|
||||
# create_or_update_secret "traefik_keepalived_password" "$TRAEFIK_KEEPALIVED_PASSWORD"
|
||||
# ;;
|
||||
# This stack is migrated — provisioning happens automatically via:
|
||||
# secrets/secrets-map.yaml (traefik: entry)
|
||||
# traefik/traefik.env.template (authoritative key list)
|
||||
# deploy/provision-stack.py (renders + ships traefik/traefik.env)
|
||||
#
|
||||
# The .woodpecker/deploy.yml provision-secrets step calls this with a
|
||||
# single line: `python3 deploy/provision-stack.py traefik`
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
# Traefik Log Rotation
|
||||
|
||||
## Why this exists
|
||||
|
||||
`traefik/traefik.yaml` runs Traefik with:
|
||||
- `--accesslog.filePath=/traefik/logs/access.log`
|
||||
- `--log.filePath=/traefik/logs/traefik.log`
|
||||
- `--log.level=DEBUG`
|
||||
|
||||
Neither file has any built-in rotation -- Traefik has no native rotate-on-size
|
||||
nor a SIGUSR1/reopen handler. Docker's `json-file` log-driver rotation
|
||||
(`max-size`/`max-file`) only applies to stdout, not to files Traefik writes
|
||||
directly via `--accesslog.filePath`/`--log.filePath`. Result: `access.log` grew
|
||||
to **~15.5GB unrotated** before this was caught, on a CephFS volume
|
||||
(`/volume1/docker-root`) already at 82-84% used. Both the disk pressure and
|
||||
the ongoing write latency of appending to a 15GB file on a network filesystem
|
||||
on every request through Traefik were flagged as a real risk factor during
|
||||
troubleshooting (Uptime Kuma WebSocket flapping investigation, Sep 2026).
|
||||
|
||||
**Note:** during that investigation, the actual root cause of the WebSocket
|
||||
flapping turned out to be Uptime Kuma monitor misconfigurations (a Postgres
|
||||
monitor throwing a null-reference error, and several monitors failing TLS
|
||||
validation against self-signed/internal-IP certs) -- not this log file. This
|
||||
rotation fix is still worth doing as general disk/IO hygiene, just not
|
||||
causally tied to that incident.
|
||||
|
||||
## Architecture constraint this design accounts for
|
||||
|
||||
`traefik_reverse-proxy` runs Swarm *`mode: global`* -- one instance on **each**
|
||||
of docker-1, docker-2, docker-3. All three write to the **same physical file**
|
||||
via the shared CephFS bind mount `/volume1/docker/traefik -> /traefik`
|
||||
(identical mount, visible identically from any node). That rules out:
|
||||
|
||||
- **Signal-based rotation** (classic `create` + `postrotate` sending SIGUSR1):
|
||||
Traefik doesn't implement a reopen signal, and even if it did, you'd need to
|
||||
signal 3 separate per-node containers in lockstep.
|
||||
- **Running logrotate on just one node**: works until that node is down, then
|
||||
rotation silently stops with no alert.
|
||||
|
||||
So this setup uses:
|
||||
1. **`copytruncate`** (see `traefik-logs.conf`) -- all three Traefik processes
|
||||
keep writing to the same inode, no signaling needed. Tradeoff: a few log
|
||||
lines written in the exact copy/truncate instant can be lost -- fine for
|
||||
diagnostic logs.
|
||||
2. **Cron on all three nodes**, coordinated via a shared `flock` + shared
|
||||
logrotate state file, both also on the CephFS mount (see
|
||||
`traefik-logrotate.sh`). Whichever node's cron fires first grabs the lock,
|
||||
rotates if due, and updates the shared state so the other two nodes' cron
|
||||
runs see it's already done. No single node is a rotation SPOF.
|
||||
3. **Size-triggered** (`size 250M`) rather than calendar-triggered (`daily`) --
|
||||
this file can grow fast under bursts (see incident background above); a
|
||||
purely daily interval would still let it balloon between runs. Cron checks
|
||||
every 15 minutes, so it cannot grow much past the 250M threshold in practice.
|
||||
|
||||
## Install (one-time, per node)
|
||||
|
||||
Must run on **all three** docker LXCs -- this is host-level cron/logrotate
|
||||
config, not something `stack-deploy.sh` can reach (it only touches Swarm
|
||||
services, not host cron jobs).
|
||||
|
||||
```bash
|
||||
# On each of docker-1 (192.168.4.31), docker-2 (192.168.4.32), docker-3 (192.168.4.33):
|
||||
ssh root@<192.168.4.31|.32|.33>
|
||||
cd /volume1/docker/compose-files
|
||||
bash deploy/git-guard.sh # confirm sync first, as always
|
||||
sudo bash traefik/logrotate/install.sh
|
||||
```
|
||||
|
||||
Verify on each node:
|
||||
```bash
|
||||
cat /etc/cron.d/traefik-logrotate
|
||||
logrotate -d /etc/logrotate.d/traefik-logs # dry-run, confirms syntax
|
||||
```
|
||||
|
||||
## Bootstrapping -- shrinking the *existing* oversized log file
|
||||
|
||||
**Not done automatically by `install.sh`.** The new size-triggered config only
|
||||
prevents *future* unbounded growth -- it won't touch the current 15.5GB file
|
||||
until the next time it crosses 250M (i.e. never, since it's already well past
|
||||
that and logrotate only acts on crossing the threshold going forward from its
|
||||
recorded size at last check).
|
||||
|
||||
`copytruncate` always copies the full current file before truncating it --
|
||||
that's inherent to how it works, not a bug. Forcing a rotation of the current
|
||||
15.5GB file would momentarily need roughly another 15-GB-sized chunk of free
|
||||
space -- and `/volume1/docker-root` only had **~14G free** at last check.
|
||||
Doing this blindly could tip the volume to 100% mid-operation.
|
||||
|
||||
**Recommended manual step (operator-run, on any one node -- it's the same
|
||||
CephFS file from all three)**: these are diagnostic access/debug logs, not
|
||||
something worth preserving in full, so just truncate directly rather than
|
||||
compress-then-truncate:
|
||||
|
||||
```bash
|
||||
# Optional: keep a small tail sample for reference before truncating
|
||||
tail -c 50000000 /volume1/docker/traefik/logs/access.log > \
|
||||
/volume1/docker/traefik-access-log-archive-$(date +%Y%m%d).log
|
||||
|
||||
# Then truncate in place (safe -- Traefik's existing file handles on all 3
|
||||
# nodes stay valid, same as copytruncate's own mechanism):
|
||||
: > /volume1/docker/traefik/logs/access.log
|
||||
: > /volume1/docker/traefik/logs/traefik.log
|
||||
|
||||
# Confirm:
|
||||
df -h /volume1/docker-root
|
||||
ls -la /volume1/docker/traefik/logs/
|
||||
```
|
||||
|
||||
After this one-time bootstrap, the cron+logrotate setup keeps it bounded
|
||||
(rotates at 250M, keeps 48 compressed generations, prunes anything over 14
|
||||
days old) going forward without needing any further manual intervention.
|
||||
|
||||
## Related, NOT included in this change (follow-up to consider separately)
|
||||
|
||||
`traefik.yaml` currently runs `--log.level=DEBUG` -- verbose debug logging in
|
||||
production, which is a meaningful contributor to how fast these files grow.
|
||||
Lowering to `INFO` would reduce volume significantly but requires a real
|
||||
modification + redeploy of the high-blast-radius `traefik` stack (all
|
||||
HTTP/HTTPS routing depends on it), so it's intentionally left out of this PR
|
||||
and should be its own reviewed change if wanted.
|
||||
@@ -0,0 +1,44 @@
|
||||
#!/usr/bin/env bash
|
||||
# One-time installer for Traefik log rotation.
|
||||
#
|
||||
# MUST be run manually, once, on EACH of docker-1, docker-2, docker-3.
|
||||
# This is intentionally NOT part of stack-deploy.sh / the Woodpecker
|
||||
# pipeline: it installs a host-level cron.d entry and /etc/logrotate.d
|
||||
# config, and `docker stack deploy` has no mechanism to reach outside the
|
||||
# Swarm/container boundary onto host cron. See README.md for why this needs
|
||||
# to run on all three nodes.
|
||||
#
|
||||
# Usage (from a checkout of this repo, on each node):
|
||||
# sudo bash traefik/logrotate/install.sh
|
||||
set -euo pipefail
|
||||
|
||||
if [[ $EUID -ne 0 ]]; then
|
||||
echo "Run as root (sudo)." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
REPO_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
|
||||
echo "Installing logrotate config..."
|
||||
install -m 0644 "$REPO_DIR/traefik-logs.conf" /etc/logrotate.d/traefik-logs
|
||||
|
||||
echo "Installing rotation wrapper..."
|
||||
install -m 0755 "$REPO_DIR/traefik-logrotate.sh" /usr/local/sbin/traefik-logrotate.sh
|
||||
|
||||
echo "Installing cron.d schedule (every 15 minutes)..."
|
||||
cat > /etc/cron.d/traefik-logrotate << 'EOF'
|
||||
# Managed by homelab/compose-files traefik/logrotate/install.sh -- do not
|
||||
# hand-edit; update traefik/logrotate/*.{conf,sh} in Gitea and re-run
|
||||
# install.sh instead.
|
||||
*/15 * * * * root /usr/local/sbin/traefik-logrotate.sh
|
||||
EOF
|
||||
chmod 0644 /etc/cron.d/traefik-logrotate
|
||||
|
||||
echo "Verifying logrotate config syntax..."
|
||||
logrotate -d /etc/logrotate.d/traefik-logs
|
||||
|
||||
echo "Done. First rotation check runs on the next cron tick (up to 15 min)."
|
||||
echo "This only rotates going forward once the file crosses the size threshold."
|
||||
echo "To shrink the EXISTING already-large log file, see README.md -- that is"
|
||||
echo "a separate, deliberate manual step (not done by this script) because it"
|
||||
echo "needs disk headroom awareness first."
|
||||
@@ -0,0 +1,29 @@
|
||||
#!/usr/bin/env bash
|
||||
# Wrapper invoked by cron on docker-1/docker-2/docker-3 to rotate the shared
|
||||
# Traefik access/error logs (see traefik-logs.conf for the "full why").
|
||||
#
|
||||
# Because /volume1/docker/traefik/logs is the SAME physical CephFS path on
|
||||
# all three nodes, and cron on all three nodes runs this independently, we
|
||||
# use a shared flock (also on the CephFS mount, so it's visible cluster-wide)
|
||||
# to guarantee only one node actually executes logrotate at a time, and a
|
||||
# SHARED state file so whichever node runs it knows the true last-rotated
|
||||
# time regardless of which node rotated it last. If a node is down, the
|
||||
# other two still cover the schedule -- none of this relies on a specific
|
||||
# node being up.
|
||||
set -euo pipefail
|
||||
|
||||
LOCK_DIR="/volume1/docker/traefik/logrotate-state"
|
||||
LOCK_FILE="$LOCK_DIR/rotate.lock"
|
||||
STATE_FILE="$LOCK_DIR/status"
|
||||
CONF_FILE="/etc/logrotate.d/traefik-logs"
|
||||
|
||||
mkdir -p "$LOCK_DIR"
|
||||
touch "$STATE_FILE"
|
||||
|
||||
exec 200>"$LOCK_FILE"
|
||||
if ! flock -n 200; then
|
||||
# Another node already holds the lock this cycle -- normal, not an error.
|
||||
exit 0
|
||||
fi
|
||||
|
||||
/usr/sbin/logrotate -s "$STATE_FILE" "$CONF_FILE"
|
||||
@@ -0,0 +1,43 @@
|
||||
# Traefik access/error log rotation
|
||||
#
|
||||
# CONTEXT: traefik_reverse-proxy runs in Swarm `mode: global` (traefik.yaml),
|
||||
# meaning one instance runs on EACH of docker-1/docker-2/docker-3. All three
|
||||
# write to the SAME physical file via the shared CephFS bind mount
|
||||
# /volume1/docker/traefik/logs -> /traefik (identical path from any node --
|
||||
# see infra context: /volume1/docker is a shared CephFS mount).
|
||||
#
|
||||
# `copytruncate` is REQUIRED here (not the default create+signal approach):
|
||||
# Traefik has no SIGUSR1/SIGHUP "reopen log file" handling, and even if it
|
||||
# did, coordinating a reopen signal across 3 independent per-node containers
|
||||
# writing to one shared inode is unnecessary complexity. copytruncate keeps
|
||||
# every writer's existing file descriptor valid (truncates in place) so all
|
||||
# three Traefik processes keep appending to the same inode with zero
|
||||
# signaling. Tradeoff: a handful of log lines written in the exact
|
||||
# copy/truncate instant can be lost -- acceptable for diagnostic access/error
|
||||
# logs, not used for anything transactional.
|
||||
#
|
||||
# Size-triggered (not calendar-triggered) on purpose: this file can grow fast
|
||||
# under bursts (see incident that prompted this -- 15.5GB accumulated with
|
||||
# --log.level=DEBUG set). `size` is checked every time the wrapper script runs
|
||||
# (cron, every 15 minutes -- see install.sh), so it cannot balloon unbounded
|
||||
# between checks the way a plain `daily` interval would.
|
||||
#
|
||||
# Installed via install.sh on ALL THREE docker LXCs (docker-1, docker-2,
|
||||
# docker-3) -- see README.md. This is a HOST-level cron/logrotate config,
|
||||
# outside the Woodpecker/stack-deploy.sh pipeline (docker stack deploy has no
|
||||
# mechanism to touch host cron), so it must be applied manually once per node,
|
||||
# not via a stack redeploy.
|
||||
|
||||
/volume1/docker/traefik/logs/*.log {
|
||||
size 250M
|
||||
rotate 48
|
||||
maxage 14
|
||||
compress
|
||||
delaycompress
|
||||
missingok
|
||||
notifempty
|
||||
copytruncate
|
||||
dateext
|
||||
dateformat -%Y%m%d-%H%M%S
|
||||
su root root
|
||||
}
|
||||
@@ -1,2 +0,0 @@
|
||||
KEEPALIVED_PASSWORD=***REDACTED***
|
||||
KEEPALIVED_VIRTUAL_IPS="#PYTHON2BASH:['192.168.4.30']"
|
||||
@@ -0,0 +1,52 @@
|
||||
# ─────────────────────────────────────────────────────────────────────────
|
||||
# traefik.env.template — AUTHORITATIVE template for traefik/traefik.env
|
||||
# (rendered by deploy/provision-stack.py per secrets/secrets-map.yaml)
|
||||
#
|
||||
# - This file IS the complete key list for traefik.env. The whole file is
|
||||
# rendered on every provisioning run — no line surgery, so a key can
|
||||
# never silently go missing.
|
||||
# - Non-secret config lives here as LITERAL values (visible, reviewable).
|
||||
# - Secret values are dollar-brace placeholders resolved from the CI env
|
||||
# (Woodpecker from_secret vars) at provisioning time. provision-stack.py
|
||||
# FAILS HARD if any placeholder is missing/empty.
|
||||
# - The rendered traefik/traefik.env exists only on the host (gitignored).
|
||||
# - Rendered by provision-stack.py, NOT Woodpecker's yaml preprocessor —
|
||||
# single-dollar placeholders are safe here (deploy.yml's double-dollar
|
||||
# rule does NOT apply to this file).
|
||||
#
|
||||
# Consumed by traefik/traefik.yaml (KEEPALIVED_PASSWORD, KEEPALIVED_VIRTUAL_IPS).
|
||||
# DOMAIN comes from deploy/global.env, not here.
|
||||
#
|
||||
# ⚠️ HIGH RISK: Traefik is the entry point for all homelab services, and the
|
||||
# Keepalived VIP (192.168.4.30) depends on this stack. KEEPALIVED_PASSWORD
|
||||
# MUST be identical across keepalived-master and keepalived-backup — both
|
||||
# consume this same rendered value — or VRRP auth fails and the VIP
|
||||
# becomes unstable (this is exactly what the 2026-09-12 incident was:
|
||||
# a committed literal "***REDACTED***" placeholder in git, restored every
|
||||
# time the file was deleted/re-synced, diverging from -backup's stale but
|
||||
# correct in-memory value).
|
||||
#
|
||||
# Classic VRRP simple-auth is silently truncated to 8 characters by
|
||||
# keepalived itself. Set TRAEFIK_KEEPALIVED_PASSWORD in Woodpecker to a
|
||||
# value 8 characters or fewer (or accept that only the first 8 chars are
|
||||
# actually significant) so the effective negotiated value is unambiguous.
|
||||
#
|
||||
# 2026-09-12 RE-TRIGGER: traefik.env was found MISSING on disk after the
|
||||
# Pattern C migration merged — the manifest-driven provisioning render only
|
||||
# happens on a push that changes traefik/, and traefik is bootstrap-tier
|
||||
# (stack-deploy.sh must be run manually; the pipeline never auto-deploys
|
||||
# this stack). A manual `stack-deploy.sh traefik` had apparently been run
|
||||
# while traefik.env was absent, causing Swarm's own interpolation to
|
||||
# silently render KEEPALIVED_PASSWORD/KEEPALIVED_VIRTUAL_IPS as EMPTY
|
||||
# STRINGS on keepalived-master (not missing — empty, which is worse,
|
||||
# since it's a valid-looking VRRP auth string that mismatches everything).
|
||||
# This comment-only change re-triggers provision-secrets to re-render and
|
||||
# re-ship traefik.env; a manual `stack-deploy.sh traefik` must still be run
|
||||
# afterward to actually apply it to the running services.
|
||||
# ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
# ── Keepalived (non-secret config) ────────────────────────────────────────
|
||||
KEEPALIVED_VIRTUAL_IPS="#PYTHON2BASH:['192.168.4.30']"
|
||||
|
||||
# ── Keepalived (secret) ────────────────────────────────────────────────────
|
||||
KEEPALIVED_PASSWORD=${TRAEFIK_KEEPALIVED_PASSWORD}
|
||||
Reference in New Issue
Block a user