diff --git a/.dockerignore b/.dockerignore index 45fb491..33d0164 100644 --- a/.dockerignore +++ b/.dockerignore @@ -9,3 +9,8 @@ tasks/ docker-compose*.yml Dockerfile .github/ + +# Nightly pg_dump output (scripts/pg-backup.sh). Root-only on the production +# host, so a build context that includes it fails with permission denied, and +# the dumps are gigabytes that must never enter an image layer. +backups/ diff --git a/DEPLOY.md b/DEPLOY.md index b90861b..6542a31 100644 --- a/DEPLOY.md +++ b/DEPLOY.md @@ -946,15 +946,32 @@ no previous key to fall back to — is still unrecoverable. scripts, both exercised end-to-end before landing (backup → archive verification → restore → inventory gates): -- `scripts/pg-backup.sh` — run from host cron. Dumps custom-format into the +- `scripts/pg-backup.sh` — run by a host systemd timer. Dumps custom-format into the `./backups` mount the compose file already provides, verifies the archive - with `pg_restore --list` **before** it gets its final name (a dump that - cannot be listed is not a backup), `chmod 600`s it, then ages out completed - dumps older than `RETENTION_DAYS` (default 14). Partials are never deleted at - any age; ones older than a day are warned about, because a partial is the - corpse of a failed run and worth seeing. Install: - - 17 2 * * * /opt/tidepool/scripts/pg-backup.sh >> /opt/tidepool/backups/backup.log 2>&1 + with a full `pg_restore -f /dev/null` read **before** it gets its final name + (a dump that cannot be read end to end is not a backup; `--list` reads only + the TOC and would pass a truncated dump), `chmod 600`s it, then ages out completed + dumps older than `RETENTION_DAYS` (default 14). A failed run keeps only the + most recent failed partial, under the fixed name + `backups/tidepool-last-failed.dump.partial`, so retries overwrite it rather + than pile up ~2GB files; partials are not aged out, and ones older than a day + are warned about, because a partial is the corpse of a failed run and worth + seeing. The production host has no cron + daemon, so it runs from the systemd units in `scripts/systemd/` (02:17 UTC, + as root, output in the journal). A failed run retries every 5 minutes, up to + 4 starts in 2 hours, which covers a boot-time catch-up run that fires before + the containers are up. The unit runs a root-owned copy at + `/usr/local/sbin/tidepool-pg-backup`, not the checkout's script: a checkout + writable by a non-root user should not supply code that runs as root. + Install, from `/opt/tidepool`, and re-run the `install` line after any change + to `scripts/pg-backup.sh`: + + sudo install -m 755 scripts/pg-backup.sh /usr/local/sbin/tidepool-pg-backup + sudo cp scripts/systemd/tidepool-pg-backup.{service,timer} /etc/systemd/system/ + sudo systemctl daemon-reload + sudo systemctl enable --now tidepool-pg-backup.timer + sudo systemctl start tidepool-pg-backup.service # first backup now + journalctl -u tidepool-pg-backup.service # its output **Provisioning: `chmod 700 /opt/tidepool/backups` once, by hand.** The script can only fix the files it creates. Every dump in that directory contains the @@ -1002,8 +1019,8 @@ material anywhere — a genuinely fresh install — is allowed to mint on an unproven key. **Retention and drill cadence interact, and not in your favour.** Retention -ages dumps out on `pg-backup.sh`'s own criteria, which is `pg_restore --list` -succeeding; a dump that lists but would fail the drill therefore keeps +ages dumps out on `pg-backup.sh`'s own criteria, which is a full `pg_restore` +read succeeding; a dump that reads cleanly but would fail the drill therefore keeps refreshing the retention window for up to `RETENTION_DAYS` while the last *drilled* dump ages out from under it. The drill is the deeper check and nothing runs it automatically: **run it after the first backup, after any @@ -1013,7 +1030,7 @@ restore, and quarterly.** copy of `.env` (password manager or sealed storage, not this server), and `touch backups/.env-backed-up` after each copy — `pg-backup.sh` warns on every run where `.env` is newer than that marker, so a rotated or edited KEK that -was never re-escrowed shows up in the backup log instead of in an incident. +was never re-escrowed shows up in the backup's journal output instead of in an incident. Residual limits, stated rather than implied: dumps live on the same host they protect (no offsite replication of `./backups` yet — copying them into the diff --git a/scripts/pg-backup.sh b/scripts/pg-backup.sh index 8db4c43..7de3cc1 100755 --- a/scripts/pg-backup.sh +++ b/scripts/pg-backup.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # pg-backup.sh — nightly logical backup of the production Postgres. # -# Runs on the HOST (cron), not in compose: the postgres container already +# Runs on the HOST (systemd timer), not in compose: the postgres container already # mounts ./backups:/backups, so pg_dump writes inside the container and the # file lands in $COMPOSE_DIR/backups on the host. Custom format (-Fc) so a # restore can be selective and parallel. @@ -10,7 +10,7 @@ # live outside Postgres. A database backup without the KEK restores # ciphertext nobody can open — DEPLOY.md "Backup and restore" says where the # KEK copy must live. This script only refuses to let that be forgotten -# silently: it warns (to stderr and the log) if .env has no offsite marker. +# silently: it warns (to stderr, i.e. the journal) if .env has no offsite marker. # # That warning fires on ANY change to .env, not only a KEK change, and that is # deliberate — do not "fix" it into comparing KEK values. The script cannot @@ -19,8 +19,11 @@ # unrelated env edit is the correct price for never missing the one edit that # makes every dump on this host unopenable. # -# Cron example (02:17 daily, as the user that owns /opt/tidepool): -# 17 2 * * * /opt/tidepool/scripts/pg-backup.sh >> /opt/tidepool/backups/backup.log 2>&1 +# Scheduled by scripts/systemd/tidepool-pg-backup.timer (02:17 UTC daily, as +# root); output goes to the journal. The unit runs a root-owned copy installed +# at /usr/local/sbin/tidepool-pg-backup, not this file, so re-install after +# editing it. Nothing here depends on the script's own location: paths come +# from COMPOSE_DIR. Install steps: DEPLOY.md "Backup and restore". set -euo pipefail @@ -35,15 +38,37 @@ DUMP="tidepool-${STAMP}.dump" echo "[$(date -u +%FT%TZ)] backup starting: ${DUMP}" +# A failed run's partial moves to one fixed name, so the timer's retries (up to +# 4 a night) overwrite a single ~2GB corpse instead of stacking timestamped +# ones. The host can mv it: the mount is the same files, and this runs as root. +PARTIAL="${COMPOSE_DIR}/backups/${DUMP}.partial" +LAST_FAILED="${COMPOSE_DIR}/backups/tidepool-last-failed.dump.partial" +COMPLETED=0 +keep_last_failed_partial() { + if [[ "${COMPLETED}" -ne 1 && -f "${PARTIAL}" ]]; then + mv -f "${PARTIAL}" "${LAST_FAILED}" + # Same secrets as a completed dump; pg_dump leaves it world-readable. + chmod 600 "${LAST_FAILED}" + echo "[$(date -u +%FT%TZ)] FAILED: partial kept as ${LAST_FAILED}" >&2 + fi +} +trap keep_last_failed_partial EXIT +# bash skips the EXIT trap on a signal it has no trap for; systemd stops and +# timeouts send SIGTERM, so turn it into an exit the EXIT trap sees. +trap 'exit 143' TERM +trap 'exit 130' INT + # Dump inside the container onto the shared mount. --no-owner/--no-acl so a # drill restore into a scratch container needs no matching roles. docker exec "${CONTAINER}" pg_dump -Fc --no-owner --no-acl \ -U "${DB_USER}" -d "${DB_NAME}" -f "/backups/${DUMP}.partial" # Verify the archive is readable BEFORE it gets the real name: a dump that -# pg_restore cannot list is not a backup, and finding that out during an -# incident is the failure mode this file exists to prevent. -docker exec "${CONTAINER}" pg_restore --list "/backups/${DUMP}.partial" > /dev/null +# pg_restore cannot read end to end is not a backup, and finding that out +# during an incident is the failure mode this file exists to prevent. A full +# read to /dev/null, not --list: --list reads only the TOC, so a dump +# truncated in its data blocks would pass. +docker exec "${CONTAINER}" pg_restore -f /dev/null "/backups/${DUMP}.partial" docker exec "${CONTAINER}" mv "/backups/${DUMP}.partial" "/backups/${DUMP}" # 600 inside the container, which is 600 on the host mount: this file holds # the plaintext service-actor PEM plus every sealed blob in the database, and @@ -53,11 +78,12 @@ docker exec "${CONTAINER}" chmod 600 "/backups/${DUMP}" SIZE="$(du -h "${COMPOSE_DIR}/backups/${DUMP}" | cut -f1)" echo "[$(date -u +%FT%TZ)] backup verified: ${DUMP} (${SIZE})" +COMPLETED=1 -# Retention: age out completed dumps only, never the log. Partials are NEVER -# deleted at any age — a partial is the corpse of a failed run, and the second -# find only warns about the ones old enough (>24h, so not this run's) to be -# certainly dead. -mmin +1440 rather than -mtime +1, which rounds down to whole +# Retention: age out completed dumps only. Partials are never deleted by age — +# a failed run keeps only its most recent partial, as tidepool-last-failed +# (see the trap above), and the second find only warns about partials old +# enough (>24h) to be certainly dead. -mmin +1440 rather than -mtime +1, which rounds down to whole # days and would not fire until the partial was nearly two days old. find "${COMPOSE_DIR}/backups" -name 'tidepool-*.dump' -mtime "+${RETENTION_DAYS}" -delete find "${COMPOSE_DIR}/backups" -name 'tidepool-*.dump.partial' -mmin +1440 -print | while read -r stale; do diff --git a/scripts/pg-restore-drill.sh b/scripts/pg-restore-drill.sh index 8611aa9..700cffc 100755 --- a/scripts/pg-restore-drill.sh +++ b/scripts/pg-restore-drill.sh @@ -39,8 +39,11 @@ if [[ -z "${DUMP}" || ! -f "${DUMP}" ]]; then fi echo "drilling restore of: ${DUMP}" +# -v removes the anonymous data volume the postgres image declares. Without it +# every drill leaves the restored production dataset on disk after the +# container is gone. cleanup() { - docker rm -f "${DRILL_NAME}" > /dev/null 2>&1 || + docker rm -f -v "${DRILL_NAME}" > /dev/null 2>&1 || echo "WARNING: could not remove drill container ${DRILL_NAME} — it may still hold a copy of production data; remove it by hand" >&2 } trap cleanup EXIT diff --git a/scripts/systemd/tidepool-pg-backup.service b/scripts/systemd/tidepool-pg-backup.service new file mode 100644 index 0000000..6c30ea7 --- /dev/null +++ b/scripts/systemd/tidepool-pg-backup.service @@ -0,0 +1,17 @@ +[Unit] +Description=Tidepool nightly Postgres backup +Requires=docker.service +After=docker.service +# Persistent=true on the timer fires a missed run at boot, possibly before the +# postgres container is up: retry, but stop after 4 starts in 2 hours so a +# persistent failure ends as a failed unit instead of looping. +StartLimitIntervalSec=2h +StartLimitBurst=4 + +[Service] +Type=oneshot +# A root-owned installed copy, not the checkout: a checkout writable by a +# non-root user should not supply code that runs as root. +ExecStart=/usr/local/sbin/tidepool-pg-backup +Restart=on-failure +RestartSec=5min diff --git a/scripts/systemd/tidepool-pg-backup.timer b/scripts/systemd/tidepool-pg-backup.timer new file mode 100644 index 0000000..e8ad1b7 --- /dev/null +++ b/scripts/systemd/tidepool-pg-backup.timer @@ -0,0 +1,9 @@ +[Unit] +Description=Run tidepool-pg-backup.service nightly + +[Timer] +OnCalendar=*-*-* 02:17:00 UTC +Persistent=true + +[Install] +WantedBy=timers.target