After #14 the deploy-host snapshot still stopped right after 'config -> config/' with no further output — neither the LDAP/Redis 'snapshotting...' lines nor the post-snapshot 'Building + starting sso-manager' line, so the stall was somewhere in the no-container path (no containers are up yet when ./setup.sh is first run) but invisible because every skip was silent and there was no marker between the config copy and the function return. Instrument + harden backup_before_rebuild() so the next run localizes it: - ERR trap (scoped to the function): if a command trips set -e and aborts the snapshot, print 'snapshot aborted by command: <cmd>' instead of dying mute after 'Snapshotting state to ...'. - Explicit 'skipped' branches: 'LDAP: sso-manager not running — skipped' and 'Redis (<svc>): not running — skipped' so a no-container run shows which path was taken instead of going quiet. - 'pruning old backups (keep=N)...' before the retention loop and 'snapshot complete.' at the end, so a stall is pinned to the retention loop (or ruled out of the snapshot entirely). - Retention: clamp with [[ ]] not (( )) — (( keep < 1 )) returns exit 1 when false, a classic set -e landmine; guard rm -rf with '|| true'; skip symlinks and non-dir entries so a stray symlink in ./backups can't point rm at an arbitrary tree. Co-authored-by: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -314,6 +314,11 @@ ensure_config
|
|||||||
# Snapshot ./config + LDAP (slapcat) + both Redis (BGSAVE + dump.rdb) before the
|
# Snapshot ./config + LDAP (slapcat) + both Redis (BGSAVE + dump.rdb) before the
|
||||||
# rebuild. No-op on the very first run (nothing running, no config to lose yet).
|
# rebuild. No-op on the very first run (nothing running, no config to lose yet).
|
||||||
backup_before_rebuild() {
|
backup_before_rebuild() {
|
||||||
|
# If a command trips `set -e` and aborts the snapshot, name the offending
|
||||||
|
# command instead of dying silently after "Snapshotting state to ..." (the
|
||||||
|
# ERR trap fires for the same failures set -e would exit on, with the same
|
||||||
|
# if/&&/|| exemptions, and is scoped to this function).
|
||||||
|
trap 'warn " snapshot aborted by command: $BASH_COMMAND"' ERR
|
||||||
local any_running=0
|
local any_running=0
|
||||||
running sso-manager && any_running=1
|
running sso-manager && any_running=1
|
||||||
running proxy && any_running=1
|
running proxy && any_running=1
|
||||||
@@ -370,6 +375,8 @@ backup_before_rebuild() {
|
|||||||
else
|
else
|
||||||
warn " could not read ldapBaseDn from sso-secrets.js — LDAP not snapshotted"
|
warn " could not read ldapBaseDn from sso-secrets.js — LDAP not snapshotted"
|
||||||
fi
|
fi
|
||||||
|
else
|
||||||
|
info " LDAP: sso-manager not running — skipped"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Redis — snapshot each running service. Capture LASTSAVE *before* issuing
|
# Redis — snapshot each running service. Capture LASTSAVE *before* issuing
|
||||||
@@ -384,7 +391,10 @@ backup_before_rebuild() {
|
|||||||
# it works regardless of which compose project owns the container.
|
# it works regardless of which compose project owns the container.
|
||||||
local svc before ok rdir rfile rpath
|
local svc before ok rdir rfile rpath
|
||||||
for svc in sso-manager proxy; do
|
for svc in sso-manager proxy; do
|
||||||
running "$svc" || continue
|
if ! running "$svc"; then
|
||||||
|
info " Redis ($svc): not running — skipped"
|
||||||
|
continue
|
||||||
|
fi
|
||||||
info " Redis ($svc): snapshotting..."
|
info " Redis ($svc): snapshotting..."
|
||||||
before="$(docker exec "$svc" redis-cli LASTSAVE 2>/dev/null | tr -dc '0-9' || echo 0)"
|
before="$(docker exec "$svc" redis-cli LASTSAVE 2>/dev/null | tr -dc '0-9' || echo 0)"
|
||||||
docker exec "$svc" redis-cli BGSAVE >/dev/null 2>&1 || true
|
docker exec "$svc" redis-cli BGSAVE >/dev/null 2>&1 || true
|
||||||
@@ -412,15 +422,23 @@ backup_before_rebuild() {
|
|||||||
fi
|
fi
|
||||||
done
|
done
|
||||||
|
|
||||||
# Retention: keep the newest BACKUP_KEEP (min 1).
|
# Retention: keep the newest BACKUP_KEEP (min 1). Use `[[ ]]` (not `(( ))`)
|
||||||
local keep="$BACKUP_KEEP"; (( keep < 1 )) && keep=1
|
# for the min-1 clamp: `(( keep < 1 ))` returns exit 1 when false, which is a
|
||||||
|
# classic set -e landmine — `[[ ]]` is exempt as the left operand of `&&`.
|
||||||
|
local keep="${BACKUP_KEEP:-5}"
|
||||||
|
[[ "$keep" -lt 1 ]] && keep=1
|
||||||
|
info " pruning old backups (keep=$keep)..."
|
||||||
local removed=0
|
local removed=0
|
||||||
while read -r old; do
|
while read -r old; do
|
||||||
[[ -n "$old" ]] || continue
|
[[ -n "$old" ]] || continue
|
||||||
rm -rf "$BACKUP_DIR/$old"
|
# Only prune real backup dirs — skip symlinks (a stray symlink could
|
||||||
|
# point rm at an arbitrary tree) and non-dir entries.
|
||||||
|
[[ -d "$BACKUP_DIR/$old" && ! -L "$BACKUP_DIR/$old" ]] || continue
|
||||||
|
rm -rf "$BACKUP_DIR/$old" || true
|
||||||
removed=$((removed + 1))
|
removed=$((removed + 1))
|
||||||
done < <(ls -1 "$BACKUP_DIR" 2>/dev/null | sort -r | tail -n +$((keep + 1)))
|
done < <(ls -1 "$BACKUP_DIR" 2>/dev/null | sort -r | tail -n +$((keep + 1)))
|
||||||
[[ "$removed" -gt 0 ]] && info " pruned $removed old backup(s) (keeping $keep)."
|
[[ "$removed" -gt 0 ]] && info " pruned $removed old backup(s) (keeping $keep)."
|
||||||
|
info " snapshot complete."
|
||||||
}
|
}
|
||||||
backup_before_rebuild
|
backup_before_rebuild
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user