| 1 | #!/usr/bin/env bash | |
| 2 | # | |
| 3 | # Dogfood — backup, restore, and the restore *rehearsal* (spec §10, M6). | |
| 4 | # | |
| 5 | # ./scripts/backup.sh backup [DIR] pg_dump + a snapshot of the repo volume | |
| 6 | # ./scripts/backup.sh restore DIR restore both into the live stack | |
| 7 | # ./scripts/backup.sh rehearse [DIR] restore into a throwaway database and | |
| 8 | # prove the result is usable | |
| 9 | # | |
| 10 | # Spec §10: | |
| 11 | # | |
| 12 | # > Backups: nightly `pg_dump` plus a filesystem-level snapshot of | |
| 13 | # > `/srv/repos`, both offsite. Test restores quarterly. Note that the database | |
| 14 | # > and the repo volume can drift out of sync during a restore; | |
| 15 | # > `dogfood-admin reindex --all` exists to reconcile them, and that recovery | |
| 16 | # > path should be exercised at least once before the first real user. | |
| 17 | # | |
| 18 | # `rehearse` is that exercise, and it is the reason this script exists rather | |
| 19 | # than a line in a runbook. It restores into a *separate* database, runs the | |
| 20 | # reconciling reindex against it, and checks the result — so the recovery path | |
| 21 | # is proven without touching production. | |
| 22 | # | |
| 23 | # The two artefacts are captured in this order, deliberately: | |
| 24 | # | |
| 25 | # 1. the repository volume, then | |
| 26 | # 2. the database. | |
| 27 | # | |
| 28 | # A repository object that exists on disk but not in the index is invisible and | |
| 29 | # fixable with `reindex`. A row that points at an object which was never captured | |
| 30 | # is a broken page with no recovery. Capturing storage first makes the drift fall | |
| 31 | # on the recoverable side. | |
| 32 | ||
| 33 | set -euo pipefail | |
| 34 | ||
| 35 | ROOT="" | |
| 36 | COMPOSE=(docker compose -f "$ROOT/docker/compose.yaml") | |
| 37 | PG_IMAGE="postgres:17-alpine" | |
| 38 | VOLUME="dogfood_repos" | |
| 39 | ||
| 40 | c_red() { printf '\033[31m%s\033[0m\n' "$*" >&2; } | |
| 41 | c_green() { printf '\033[32m%s\033[0m\n' "$*"; } | |
| 42 | c_dim() { printf '\033[2m%s\033[0m\n' "$*"; } | |
| 43 | ||
| 44 | env_get() { sed -n "s/^$1=//p" "$ROOT/.env" | head -1; } | |
| 45 | ||
| 46 | require_env() { | |
| 47 | [[ -f "$ROOT/.env" ]] || { c_red "no .env at $ROOT/.env"; exit 1; } | |
| 48 | DATABASE_URL="" | |
| 49 | [[ -n "$DATABASE_URL" ]] || { c_red "DATABASE_URL is not set in .env"; exit 1; } | |
| 50 | } | |
| 51 | ||
| 52 | # ─── backup ────────────────────────────────────────────────────────────────── | |
| 53 | ||
| 54 | cmd_backup() { | |
| 55 | require_env | |
| 56 | local dir="" | |
| 57 | mkdir -p "$dir" | |
| 58 | ||
| 59 | # 1. Repository storage first (see the note at the top). | |
| 60 | c_dim "snapshotting the repository volume…" | |
| 61 | docker run --rm \ | |
| 62 | -v "$VOLUME":/srv/repos:ro \ | |
| 63 | -v "$dir":/backup \ | |
| 64 | "$PG_IMAGE" \ | |
| 65 | tar -C /srv/repos -czf /backup/repos.tar.gz . | |
| 66 | ||
| 67 | # 2. The database. `--no-owner` and `--no-acl` so a restore does not depend | |
| 68 | # on the role names of the instance it came from. | |
| 69 | c_dim "dumping the database…" | |
| 70 | docker run --rm -v "$dir":/backup "$PG_IMAGE" \ | |
| 71 | pg_dump --no-owner --no-acl --format=custom \ | |
| 72 | --file=/backup/dogfood.dump "$DATABASE_URL" | |
| 73 | ||
| 74 | # A manifest, so a restore can tell what it is holding without opening it. | |
| 75 | cat > "$dir/manifest.txt" <<EOF | |
| 76 | dogfood backup | |
| 77 | taken_at: | |
| 78 | host: | |
| 79 | repos_bytes: | |
| 80 | dump_bytes: | |
| 81 | note: storage was captured before the database, so any drift is recoverable | |
| 82 | with 'dogfood-admin reindex --all'. | |
| 83 | EOF | |
| 84 | ||
| 85 | c_green "backup written to $dir" | |
| 86 | cat "$dir/manifest.txt" | |
| 87 | } | |
| 88 | ||
| 89 | # ─── restore ───────────────────────────────────────────────────────────────── | |
| 90 | ||
| 91 | cmd_restore() { | |
| 92 | require_env | |
| 93 | local dir="" | |
| 94 | [[ -f "$dir/dogfood.dump" && -f "$dir/repos.tar.gz" ]] \ | |
| 95 | || { c_red "$dir does not look like a Dogfood backup"; exit 1; } | |
| 96 | ||
| 97 | c_red "This overwrites the live database and repository volume." | |
| 98 | read -rp "Type RESTORE to continue: " confirm | |
| 99 | [[ "$confirm" == "RESTORE" ]] || { echo "aborted"; exit 1; } | |
| 100 | ||
| 101 | c_dim "stopping the stack…" | |
| 102 | "" down | |
| 103 | ||
| 104 | c_dim "restoring the repository volume…" | |
| 105 | docker run --rm -v "$VOLUME":/srv/repos -v "$dir":/backup:ro "$PG_IMAGE" \ | |
| 106 | sh -c 'rm -rf /srv/repos/* && tar -C /srv/repos -xzf /backup/repos.tar.gz' | |
| 107 | ||
| 108 | c_dim "restoring the database…" | |
| 109 | docker run --rm -v "$dir":/backup:ro "$PG_IMAGE" \ | |
| 110 | pg_restore --clean --if-exists --no-owner --no-acl \ | |
| 111 | --dbname "$DATABASE_URL" /backup/dogfood.dump | |
| 112 | ||
| 113 | c_dim "starting the stack…" | |
| 114 | "" up -d | |
| 115 | ||
| 116 | # The whole point of §10's note: after a restore the two halves may disagree, | |
| 117 | # and this is what reconciles them. | |
| 118 | c_dim "reconciling the index with storage…" | |
| 119 | docker compose -f "$ROOT/docker/compose.yaml" exec -T worker \ | |
| 120 | dogfood-admin reindex --all | |
| 121 | ||
| 122 | c_green "restore complete" | |
| 123 | } | |
| 124 | ||
| 125 | # ─── rehearsal ─────────────────────────────────────────────────────────────── | |
| 126 | ||
| 127 | cmd_rehearse() { | |
| 128 | require_env | |
| 129 | local dir="" | |
| 130 | ||
| 131 | # No backup named: take one now, so the rehearsal exercises the real path | |
| 132 | # end to end rather than an artefact somebody prepared by hand. | |
| 133 | if [[ -z "$dir" ]]; then | |
| 134 | dir="/backup" | |
| 135 | c_dim "no backup given; taking one into $dir" | |
| 136 | cmd_backup "$dir" >/dev/null | |
| 137 | fi | |
| 138 | ||
| 139 | # Not `local`: the EXIT trap below runs after this function's frame is gone, | |
| 140 | # and a `local` would be unbound by then — which under `set -u` turns a | |
| 141 | # successful rehearsal into a non-zero exit during cleanup. | |
| 142 | pg_name="dogfood-restore-rehearsal-$$" | |
| 143 | repo_vol="dogfood-rehearsal-repos-$$" | |
| 144 | worker_name="dogfood-rehearsal-worker-$$" | |
| 145 | ||
| 146 | cleanup() { | |
| 147 | c_dim "cleaning up the rehearsal environment…" | |
| 148 | docker rm -f "$worker_name" >/dev/null 2>&1 || true | |
| 149 | docker rm -f "$pg_name" >/dev/null 2>&1 || true | |
| 150 | docker volume rm "$repo_vol" >/dev/null 2>&1 || true | |
| 151 | } | |
| 152 | trap cleanup EXIT | |
| 153 | ||
| 154 | c_dim "starting a throwaway database…" | |
| 155 | docker run -d --name "$pg_name" \ | |
| 156 | -e POSTGRES_PASSWORD=rehearse -e POSTGRES_DB=dogfood \ | |
| 157 | "$PG_IMAGE" >/dev/null | |
| 158 | ||
| 159 | # Wait for it rather than sleeping a guess. | |
| 160 | for _ in ; do | |
| 161 | docker exec "$pg_name" pg_isready -U postgres >/dev/null 2>&1 && break | |
| 162 | sleep 1 | |
| 163 | done | |
| 164 | docker exec "$pg_name" pg_isready -U postgres >/dev/null \ | |
| 165 | || { c_red "the rehearsal database never became ready"; exit 1; } | |
| 166 | ||
| 167 | local url="postgres://postgres:rehearse@127.0.0.1:5432/dogfood" | |
| 168 | ||
| 169 | c_dim "restoring the dump…" | |
| 170 | docker cp "$dir/dogfood.dump" "$pg_name":/tmp/dogfood.dump | |
| 171 | docker exec "$pg_name" pg_restore --clean --if-exists --no-owner --no-acl \ | |
| 172 | --dbname "$url" /tmp/dogfood.dump | |
| 173 | ||
| 174 | c_dim "restoring the repository volume…" | |
| 175 | docker volume create "$repo_vol" >/dev/null | |
| 176 | docker run --rm -v "$repo_vol":/srv/repos -v "$dir":/backup:ro "$PG_IMAGE" \ | |
| 177 | tar -C /srv/repos -xzf /backup/repos.tar.gz | |
| 178 | ||
| 179 | # ── the checks that make this a rehearsal rather than a copy ────────────── | |
| 180 | c_dim "checking the restored database…" | |
| 181 | ||
| 182 | local checks_failed=0 | |
| 183 | check() { | |
| 184 | local label="$1" sql="$2" | |
| 185 | local out | |
| 186 | out="" | |
| 187 | if [[ "$out" == "t" || "$out" == "0" ]]; then | |
| 188 | printf ' \033[32m✓\033[0m %s\n' "$label" | |
| 189 | else | |
| 190 | printf ' \033[31m✗\033[0m %s (got: %s)\n' "$label" "$out" | |
| 191 | checks_failed=$((checks_failed + 1)) | |
| 192 | fi | |
| 193 | } | |
| 194 | ||
| 195 | check "schema is present" \ | |
| 196 | "SELECT to_regclass('public.repos') IS NOT NULL" | |
| 197 | check "migrations are recorded" \ | |
| 198 | "SELECT count(*) = 0 FROM _sqlx_migrations WHERE NOT success" | |
| 199 | check "no change points at a missing repository" \ | |
| 200 | "SELECT count(*) FROM changes c LEFT JOIN repos r ON r.id = c.repo_id WHERE r.id IS NULL" | |
| 201 | check "no revision points at a missing change" \ | |
| 202 | "SELECT count(*) FROM revisions v LEFT JOIN changes c ON c.id = v.change_id_fk WHERE c.id IS NULL" | |
| 203 | check "no comment is anchored to a missing revision" \ | |
| 204 | "SELECT count(*) FROM comments m | |
| 205 | WHERE m.anchor_revision IS NOT NULL | |
| 206 | AND NOT EXISTS (SELECT 1 FROM revisions v WHERE v.id = m.anchor_revision)" | |
| 207 | ||
| 208 | # Every repository row must have storage behind it. This is the drift §10 | |
| 209 | # warns about, and the one the reindex exists to reconcile. | |
| 210 | c_dim "checking that every repository row has storage…" | |
| 211 | local missing=0 total=0 | |
| 212 | while read -r id; do | |
| 213 | [[ -n "$id" ]] || continue | |
| 214 | total=$((total + 1)) | |
| 215 | local hex="" | |
| 216 | if ! docker run --rm -v "$repo_vol":/srv/repos:ro "$PG_IMAGE" \ | |
| 217 | test -d "/srv/repos//.git"; then | |
| 218 | missing=$((missing + 1)) | |
| 219 | c_dim " missing storage for repo $id" | |
| 220 | fi | |
| 221 | done < | |
| 222 | ||
| 223 | printf ' %s %d of %d repositories have storage\n' \ | |
| 224 | "" \ | |
| 225 | "$((total - missing))" "$total" | |
| 226 | ||
| 227 | if [[ $missing -gt 0 ]]; then | |
| 228 | c_dim " (recoverable: 'dogfood-admin reindex --all' reconciles the index" | |
| 229 | c_dim " with storage, but it cannot invent objects that were not captured)" | |
| 230 | fi | |
| 231 | ||
| 232 | # The reconciling reindex itself, against the rehearsal environment. This is | |
| 233 | # the §10 recovery path, actually run — queue the work, then run a worker | |
| 234 | # long enough to drain it, then check that it drained cleanly. | |
| 235 | # | |
| 236 | # `--entrypoint` is required: the worker image's entrypoint is the worker | |
| 237 | # daemon, so passing `dogfood-admin …` as the command would start the | |
| 238 | # daemon with those words as arguments and never return. | |
| 239 | c_dim "queueing 'dogfood-admin reindex --all' against the restore…" | |
| 240 | if docker run --rm --network "container:$pg_name" \ | |
| 241 | --entrypoint dogfood-admin \ | |
| 242 | -v "$repo_vol":/srv/repos \ | |
| 243 | -e DATABASE_URL="$url" -e REPO_ROOT=/srv/repos \ | |
| 244 | dogfood-worker reindex --all; then | |
| 245 | printf ' \033[32m✓\033[0m reindex queued\n' | |
| 246 | else | |
| 247 | printf ' \033[31m✗\033[0m reindex could not be queued\n' | |
| 248 | checks_failed=$((checks_failed + 1)) | |
| 249 | fi | |
| 250 | ||
| 251 | c_dim "running a worker against the restore until the queue drains…" | |
| 252 | docker run -d --name "$worker_name" --network "container:$pg_name" \ | |
| 253 | -v "$repo_vol":/srv/repos \ | |
| 254 | -e DATABASE_URL="$url" -e REPO_ROOT=/srv/repos \ | |
| 255 | dogfood-worker >/dev/null | |
| 256 | ||
| 257 | local drained=0 | |
| 258 | for _ in ; do | |
| 259 | local queued | |
| 260 | queued=" | |
| 261 | " | |
| 262 | if [[ "$queued" == "0" ]]; then drained=1; break; fi | |
| 263 | sleep 1 | |
| 264 | done | |
| 265 | ||
| 266 | local failed | |
| 267 | failed=" | |
| 268 | " | |
| 269 | ||
| 270 | docker logs "$worker_name" 2>&1 | grep -E '"level":"(ERROR|WARN)"' | head -5 || true | |
| 271 | docker rm -f "$worker_name" >/dev/null 2>&1 || true | |
| 272 | ||
| 273 | if [[ "$drained" == "1" && "$failed" == "0" ]]; then | |
| 274 | printf ' \033[32m✓\033[0m the queue drained with no failed jobs\n' | |
| 275 | else | |
| 276 | printf ' \033[31m✗\033[0m reindex did not complete (drained=%s failed=%s)\n' \ | |
| 277 | "$drained" "$failed" | |
| 278 | checks_failed=$((checks_failed + 1)) | |
| 279 | fi | |
| 280 | ||
| 281 | echo | |
| 282 | if [[ $checks_failed -eq 0 ]]; then | |
| 283 | c_green "restore rehearsal passed — this backup is restorable" | |
| 284 | else | |
| 285 | c_red "restore rehearsal FAILED $checks_failed check(s)" | |
| 286 | exit 1 | |
| 287 | fi | |
| 288 | } | |
| 289 | ||
| 290 | case "" in | |
| 291 | backup) shift; cmd_backup "$@" ;; | |
| 292 | restore) shift; cmd_restore "$@" ;; | |
| 293 | rehearse) shift; cmd_rehearse "$@" ;; | |
| 294 | *) | |
| 295 | sed -n '2,30p' "" | sed 's/^# \{0,1\}//' | |
| 296 | exit 1 | |
| 297 | ;; | |
| 298 | esac |
298 lines · Shell