feat(cleanup): guarantee step-2b dev deploys get reaped
- /recipe-upgrade step 2b: teardown is now MANDATORY on every exit path (finally), with a verify-no-leak check; tear down even on failure before reporting. - reap-dev-deploys.sh: safe, age-gated backstop that removes only idle dev-* stacks (never CI per-run stacks, warm-*, infra; an active dev loop stays fresh). - orchestrator: hourly cc-ci-reap-dev-deploys systemd timer runs it against cc-ci, bounding any leaked dev deploy from a crashed/abandoned loop. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
77ba7ee075
commit
23bba98be4
+49
@@ -0,0 +1,49 @@
|
||||
#!/usr/bin/env bash
|
||||
# Reap LEAKED step-2b dev deploys on the cc-ci server.
|
||||
#
|
||||
# /recipe-upgrade step 2b deploys a recipe under a `dev-<recipe>` domain to debug an upgrade with live
|
||||
# logs, and REQUIRES the agent to tear it down when done. This is the automated backstop for when that
|
||||
# teardown is missed (agent crashed / killed / abandoned mid-loop): it removes `dev-*` Swarm stacks
|
||||
# (+ their now-dangling volumes) whose newest service has not been updated in THRESHOLD seconds.
|
||||
#
|
||||
# SAFE to run anytime — even while CI is mid-run — because it is scoped + age-gated:
|
||||
# - it touches ONLY the `dev-` naming convention used by step 2b. CI per-run stacks
|
||||
# (`<recipe[:4]>-<hash>`), `warm-*` canonicals, and infra are never `dev-*`, so never matched.
|
||||
# - an ACTIVE dev loop redeploys (refreshing the service UpdatedAt), so it stays "fresh" and is NOT
|
||||
# reaped mid-use; only an idle/abandoned `dev-*` ages past THRESHOLD and is removed.
|
||||
# - volume cleanup uses `dangling=true`, so an active deploy's attached volumes are never removed.
|
||||
#
|
||||
# Run ON the cc-ci host: ssh cc-ci 'THRESHOLD=14400 bash -s' < reap-dev-deploys.sh
|
||||
set -uo pipefail
|
||||
export PATH=/run/current-system/sw/bin:$PATH
|
||||
|
||||
THRESHOLD="${THRESHOLD:-14400}" # 4h — generous, so a long but ACTIVE dev loop is never reaped
|
||||
now=$(date +%s)
|
||||
reaped=0
|
||||
|
||||
mapfile -t STACKS < <(docker stack ls --format '{{.Name}}' 2>/dev/null | grep -E '^dev-' || true)
|
||||
for s in "${STACKS[@]}"; do
|
||||
[ -z "$s" ] && continue
|
||||
newest=0
|
||||
for sid in $(docker service ls --filter "label=com.docker.stack.namespace=$s" -q 2>/dev/null); do
|
||||
ua=$(docker service inspect "$sid" --format '{{.UpdatedAt}}' 2>/dev/null)
|
||||
e=$(date -d "$ua" +%s 2>/dev/null || echo 0)
|
||||
[ "$e" -gt "$newest" ] && newest="$e"
|
||||
done
|
||||
age=$(( now - newest ))
|
||||
if [ "$newest" -gt 0 ] && [ "$age" -gt "$THRESHOLD" ]; then
|
||||
echo "reap: dev stack '$s' idle ${age}s (> ${THRESHOLD}s) — removing"
|
||||
docker stack rm "$s" >/dev/null 2>&1 || true
|
||||
reaped=$((reaped + 1))
|
||||
else
|
||||
echo "keep: dev stack '$s' active (last update ${age}s ago)"
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$reaped" -gt 0 ]; then
|
||||
sleep 8 # let removed stacks' services drain so their volumes become dangling
|
||||
for v in $(docker volume ls -qf dangling=true 2>/dev/null | grep -E '^dev-' || true); do
|
||||
docker volume rm "$v" >/dev/null 2>&1 && echo "reap: removed leaked volume $v"
|
||||
done
|
||||
fi
|
||||
echo "reap-dev-deploys: ${reaped} stale dev deploy(s) removed"
|
||||
Reference in New Issue
Block a user