Compare commits

..
Author SHA1 Message Date
notplants bd5f181737 fix(db): bump DB_ENTRYPOINT_VERSION to v3 so the entrypoint config reloads
cc-ci/testme cc-ci: success
The install-user fix changed the entrypoint content; swarm configs are
immutable, so the config name (which embeds DB_ENTRYPOINT_VERSION) must change
for a redeploy to pick up the new script.
2026-06-16 18:04:05 +00:00
notplants 57f5ee2531 fix(db): run pg_upgrade as the old cluster's real install user
pg_upgrade must run as the old cluster's bootstrap superuser (oid 10), and the
new cluster must be initialised with that same user, otherwise it fails the
"database user is the install user" consistency check. The install user is not
necessarily $POSTGRES_USER: clusters created with the default "postgres"
superuser plus a separate app role (e.g. discourse) are common.

Detect it from the old cluster by briefly starting it and reading pg_roles
(oid = 10) as the known app role, then use it for both the new cluster's initdb
and the pg_upgrade -U argument.
2026-06-16 17:59:26 +00:00
notplants 101ffe1964 fix(db): make pg_upgrade migration idempotent & crash-safe
The postgres major-version migration in the db entrypoint was not safe to
re-run. If the container was killed mid-migration it could crash-loop forever
("mkdir: cannot create directory .../old_data: File exists") or silently initdb
a fresh empty cluster over the live data once PG_VERSION had been moved out of
$PGDATA but before the in-progress marker was written.

Replace the marker file with a state-driven guard keyed on the scratch dirs:
empty old_data/new_data means the run was interrupted before any data moved, so
discard and retry (idempotent); non-empty means data may only live there, so
stop for manual recovery. Bump DB_ENTRYPOINT_VERSION v1->v2 so swarm picks up
the new (immutable) config.
2026-06-16 17:00:16 +00:00
notplants 433ce12dbc Merge pull request 'chore: upgrade to 0.10.0+3.5.0' (#2) from upgrade-0.8.0+3.5.0 into main
Reviewed-on: #2
2026-06-15 17:37:14 +00:00
autonomic-bot b7d8a244d7 chore: upgrade to 0.10.0+3.5.0 (redis 8.0->8.8-alpine)
cc-ci/testme cc-ci: success
2026-06-11 22:52:37 +00:00
autonomic-bot 7ae7b0f76e chore: upgrade to 0.9.0+3.5.0
cc-ci/testme cc-ci: success
2026-06-05 02:03:34 +00:00
notplants b0f9ae743a fix(db): switch postgres image to pgvector/pgvector:pg17 + bump PG_BACKUP_VERSION
cc-ci/testme cc-ci: success
2026-06-02 20:07:06 +00:00
notplants 5091fd999e improved comments
cc-ci/testme cc-ci: failure
2026-06-02 19:10:27 +00:00
notplants ec7bbdf786 fix(backup): add pg_backup.sh + proper backup/restore hooks, 20m start_period 2026-06-02 19:10:27 +00:00
notplants 0f873433ba chore: upgrade to 0.8.0+3.5.0 2026-06-02 19:10:27 +00:00
5 changed files with 43 additions and 34 deletions
-1
View File
@@ -19,4 +19,3 @@ LETS_ENCRYPT_ENV=production
#SECRET_SMTP_PASSWORD_VERSION=v1 #SECRET_SMTP_PASSWORD_VERSION=v1
SECRET_DB_PASSWORD_VERSION=v1 SECRET_DB_PASSWORD_VERSION=v1
+2 -2
View File
@@ -1,2 +1,2 @@
export DB_ENTRYPOINT_VERSION=v1 export DB_ENTRYPOINT_VERSION=v3
export PG_BACKUP_VERSION=v1 export PG_BACKUP_VERSION=v2
+5 -5
View File
@@ -3,7 +3,7 @@ version: "3.8"
services: services:
app: app:
image: bitnamilegacy/discourse:3.3.1 image: bitnamilegacy/discourse:3.5.0
networks: networks:
- proxy - proxy
- internal - internal
@@ -43,7 +43,7 @@ services:
#- "traefik.http.routers.${STACK_NAME}.middlewares=${STACK_NAME}-redirect" #- "traefik.http.routers.${STACK_NAME}.middlewares=${STACK_NAME}-redirect"
#- "traefik.http.middlewares.${STACK_NAME}-redirect.headers.SSLForceHost=true" #- "traefik.http.middlewares.${STACK_NAME}-redirect.headers.SSLForceHost=true"
#- "traefik.http.middlewares.${STACK_NAME}-redirect.headers.SSLHost=${DOMAIN}" #- "traefik.http.middlewares.${STACK_NAME}-redirect.headers.SSLHost=${DOMAIN}"
- "coop-cloud.${STACK_NAME}.version=0.8.0+3.3.1" - "coop-cloud.${STACK_NAME}.version=0.10.0+3.5.0"
healthcheck: healthcheck:
test: "ruby -e \"require 'uri'; require 'net/http'; uri = URI('http://localhost:3000/srv/status'); res = Net::HTTP.get_response(uri); if res.is_a?(Net::HTTPSuccess) then exit (0) else exit (1) end\"" test: "ruby -e \"require 'uri'; require 'net/http'; uri = URI('http://localhost:3000/srv/status'); res = Net::HTTP.get_response(uri); if res.is_a?(Net::HTTPSuccess) then exit (0) else exit (1) end\""
interval: 30s interval: 30s
@@ -52,7 +52,7 @@ services:
start_period: 20m start_period: 20m
db: db:
image: postgres:13 image: pgvector/pgvector:pg17
networks: networks:
- internal - internal
secrets: secrets:
@@ -80,14 +80,14 @@ services:
backupbot.restore.post-hook: "/pg_backup.sh restore" backupbot.restore.post-hook: "/pg_backup.sh restore"
redis: redis:
image: redis:7.4-alpine image: redis:8.8-alpine
networks: networks:
- internal - internal
volumes: volumes:
- 'redis_data:/data' - 'redis_data:/data'
sidekiq: sidekiq:
image: bitnamilegacy/discourse:3.3.1 image: bitnamilegacy/discourse:3.5.0
networks: networks:
- proxy - proxy
- internal - internal
+28 -10
View File
@@ -2,16 +2,23 @@
set -e set -e
MIGRATION_MARKER=$PGDATA/migration_in_progress
OLDDATA=$PGDATA/old_data OLDDATA=$PGDATA/old_data
NEWDATA=$PGDATA/new_data NEWDATA=$PGDATA/new_data
echo "Running as $(id)" echo "Running as $(id)"
if [ -e $MIGRATION_MARKER ]; then # The migration uses $OLDDATA/$NEWDATA as scratch and removes them when it
echo "FATAL: migration was started but did not complete in a previous run. manual recovery necessary" # finishes; a leftover *empty* one means a run was interrupted before any data
exit 1 # moved (data still intact at $PGDATA) so we clear it and retry, while a
fi # *non-empty* one means data may live only there, so we stop for manual recovery.
for scratch in $OLDDATA $NEWDATA; do
if [ -d "$scratch" ] && [ -n "$(ls -A "$scratch")" ]; then
echo "FATAL: $scratch exists and is not empty - a previous migration did not"
echo "complete and the data may only exist there. manual recovery necessary."
exit 1
fi
done
rm -rf $OLDDATA $NEWDATA
if [ -f $PGDATA/PG_VERSION ]; then if [ -f $PGDATA/PG_VERSION ]; then
DATA_VERSION=$(cat $PGDATA/PG_VERSION) DATA_VERSION=$(cat $PGDATA/PG_VERSION)
@@ -23,22 +30,33 @@ if [ -f $PGDATA/PG_VERSION ]; then
apt-get update && apt-get install -y --no-install-recommends \ apt-get update && apt-get install -y --no-install-recommends \
postgresql-$DATA_VERSION \ postgresql-$DATA_VERSION \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
# pg_upgrade must run as the old cluster's bootstrap superuser (the "install
# user", oid 10), and the new cluster must be initialised with that same
# user. It is not necessarily $POSTGRES_USER (e.g. clusters created with the
# default "postgres" superuser and a separate app role), so read it from the
# old cluster: briefly start it and ask, connecting as the app role we know.
PGBIN=/usr/lib/postgresql/$DATA_VERSION/bin
gosu postgres $PGBIN/pg_ctl -D $PGDATA -w \
-o "-c listen_addresses= -c unix_socket_directories=/tmp" start
INSTALL_USER=$(gosu postgres psql -h /tmp -U "$POSTGRES_USER" -d postgres -tAc \
"select rolname from pg_roles where oid = 10")
gosu postgres $PGBIN/pg_ctl -D $PGDATA -w stop
echo "old cluster install user: $INSTALL_USER"
echo "shuffling around" echo "shuffling around"
gosu postgres mkdir $OLDDATA $NEWDATA gosu postgres mkdir $OLDDATA $NEWDATA
chmod 700 $OLDDATA $NEWDATA chmod 700 $OLDDATA $NEWDATA
mv $PGDATA/* $OLDDATA/ || true mv $PGDATA/* $OLDDATA/ || true
touch $MIGRATION_MARKER
echo "running initdb" echo "running initdb"
# abuse entrypoint script for initdb by making server error out # abuse entrypoint script for initdb by making server error out; initialise
gosu postgres bash -c "export PGDATA=$NEWDATA ; /usr/local/bin/docker-entrypoint.sh --invalid-arg || true" # the new cluster with the same superuser as the old one so pg_upgrade matches
gosu postgres bash -c "export PGDATA=$NEWDATA POSTGRES_USER=$INSTALL_USER ; /usr/local/bin/docker-entrypoint.sh --invalid-arg || true"
echo "running pg_upgrade" echo "running pg_upgrade"
cd /tmp cd /tmp
gosu postgres pg_upgrade --link -b /usr/lib/postgresql/$DATA_VERSION/bin -d $OLDDATA -D $NEWDATA -U $POSTGRES_USER gosu postgres pg_upgrade --link -b /usr/lib/postgresql/$DATA_VERSION/bin -d $OLDDATA -D $NEWDATA -U $INSTALL_USER
cp $OLDDATA/pg_hba.conf $NEWDATA/ cp $OLDDATA/pg_hba.conf $NEWDATA/
mv $NEWDATA/* $PGDATA mv $NEWDATA/* $PGDATA
rm -rf $OLDDATA rm -rf $OLDDATA
rmdir $NEWDATA rmdir $NEWDATA
rm $MIGRATION_MARKER
echo "migration complete" echo "migration complete"
fi fi
fi fi
+8 -16
View File
@@ -1,18 +1,6 @@
#!/bin/bash #!/bin/bash
# Postgres backup/restore hook for the discourse `db` service. Invoked by backupbot-two via: # Postgres backup/restore hook for the discourse `db` service.
# backupbot.backup.pre-hook = "/pg_backup.sh backup"
# backupbot.backup.volumes.postgresql_data.path = "backup.sql"
# backupbot.restore.post-hook = "/pg_backup.sh restore"
# Backup dumps the DB to backup.sql (gzip) inside the postgresql_data volume; backupbot archives it.
# Restore reimports it. Discourse (the rails app + sidekiq) keeps many TCP connections open to the DB
# and reconnects within milliseconds, so a one-shot pg_terminate_backend is NOT enough: restore must
# first block all non-local connections at the pg_hba level (so the app cannot reconnect and interfere
# mid-reimport), then FORCE-drop, recreate, and deterministically reimport the dump, then restore
# pg_hba. (Mirrors the proven matrix-synapse restore hook.) The previous recipe shipped a pg_dump
# backup but NO restore hook — a file-level restore did not reload into the running postgres, so a
# restored backup silently kept the live (un-restored) state. cc-ci caught this: a seeded ci_marker row
# was gone after restore. Same pattern as the immich / mattermost-lts / ghost recipe-PRs.
set -e set -e
@@ -29,8 +17,7 @@ function restore {
cd /var/lib/postgresql/data/ cd /var/lib/postgresql/data/
# Block all non-local connections so the running discourse app + sidekiq cannot reconnect and # Block all non-local connections so the running discourse app + sidekiq cannot reconnect and
# interfere with the drop/recreate/reimport (a one-shot pg_terminate_backend is not enough — the # interfere with the drop/recreate/reimport. Restored on exit.
# app reconnects within ms over TCP). Restored on exit.
restore_hba() { restore_hba() {
cat pg_hba.conf.bak > pg_hba.conf cat pg_hba.conf.bak > pg_hba.conf
rm -f pg_hba.conf.bak rm -f pg_hba.conf.bak
@@ -41,11 +28,16 @@ function restore {
su postgres -c 'pg_ctl reload' su postgres -c 'pg_ctl reload'
trap restore_hba EXIT INT TERM trap restore_hba EXIT INT TERM
# Terminate lingering local sessions, then FORCE-drop + recreate + deterministic reimport. # terminate any lingering local sessions before recreate
# see https://stackoverflow.com/questions/5108876/kill-a-postgresql-session-connection
psql -U "$DB_USER" -d postgres -c \ psql -U "$DB_USER" -d postgres -c \
"SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE datname='${DB_NAME}' AND pid<>pg_backend_pid();" "SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE datname='${DB_NAME}' AND pid<>pg_backend_pid();"
# drop database and then recreate it
psql -U "$DB_USER" -d postgres -c "DROP DATABASE ${DB_NAME} WITH (FORCE);" psql -U "$DB_USER" -d postgres -c "DROP DATABASE ${DB_NAME} WITH (FORCE);"
createdb -U "$DB_USER" "$DB_NAME" createdb -U "$DB_USER" "$DB_NAME"
# reimport data
gunzip -c "$BACKUP_FILE" | psql -U "$DB_USER" -d "$DB_NAME" -1 -v ON_ERROR_STOP=1 -f - gunzip -c "$BACKUP_FILE" | psql -U "$DB_USER" -d "$DB_NAME" -1 -v ON_ERROR_STOP=1 -f -
} }