diff --git a/bootstrap.sh b/bootstrap.sh index 884d706..2f05723 100755 --- a/bootstrap.sh +++ b/bootstrap.sh @@ -560,18 +560,18 @@ phase_12_snapshot() { # stale golden image worse than no golden image at all. # # Creating it unconditionally meant every full run produced another one, so - # the §6.1 instruction to delete it after the first drill-rebuild.sh was - # undone by the next bootstrap. The condition here is the creation-side half - # of that instruction, and the one phase 12 can actually observe: an - # automatic backup existing is exactly the state in which the substitute is - # no longer needed. + # deleting it once was undone by the next bootstrap. The condition here is + # the creation-side half of that same problem, and the one phase 12 can + # actually observe: an automatic backup existing is exactly the state in + # which the substitute is no longer needed. drill-restore.sh holds the + # deletion-side half — it removes this image once a quarterly run proves an + # automatic backup exists by restoring from one instead (§6.1). if hcloud image list --type backup -o noheader -o columns=created_from \ | grep -qFx "$FLIT_SERVER"; then info "automatic backups exist for $FLIT_SERVER; no golden snapshot needed (§6.1)" return 0 fi hcloud server create-image --type snapshot --description "$FLIT_NAME golden $(date -I)" "$FLIT_SERVER" - warn "delete this snapshot once drill-rebuild.sh has passed once (§6.1) — it bills monthly and is read once" } phase_13_drills() { diff --git a/drill-rebuild.sh b/drill-rebuild.sh index 9fef84d..bc96e3f 100755 --- a/drill-rebuild.sh +++ b/drill-rebuild.sh @@ -122,5 +122,3 @@ fi DRILL_RC=0 log "rebuild drill PASSED in $(( ($(date +%s) - START) / 60 )) minutes; no drift" -log "reminder: once this has passed, delete the golden snapshot (§6.1) —" -log "the scripts are the golden image now, and it bills monthly for one use." diff --git a/drill-restore.sh b/drill-restore.sh index f556cec..9067a5d 100755 --- a/drill-restore.sh +++ b/drill-restore.sh @@ -144,3 +144,25 @@ DRILL_RC=0 mins=$(( ($(date +%s) - START) / 60 )) log "restore drill PASSED in ${mins} minutes" log "that number is your real recovery time; §8.3 is the procedure a human follows." + +# The golden snapshot is a build-time stand-in for an automatic backup that +# does not exist yet (phase 12, bootstrap.sh). A run that reaches this point +# without --from-snapshot has just restored from a real automatic backup, +# which is proof one now exists — the condition phase 12 checks before ever +# creating another snapshot. Nothing else reads this image again, and it +# bills per GB every month for a use that cannot recur, so it is retired here +# rather than left for an operator to remember (§6.1). Reuses the same query +# --from-snapshot uses above so the two cannot pick different images. +if [[ $FROM_SNAPSHOT -eq 0 ]]; then + golden=$(hcloud image list --type snapshot -o noheader -o columns=id,description \ + | grep "$FLIT_NAME golden" | tail -1 | awk '{print $1}') + if [[ -n "$golden" ]]; then + if hcloud image delete "$golden" >/dev/null; then + log "deleted golden snapshot $golden — this run proved an automatic backup now exists (§6.1)" + else + warn "could not delete golden snapshot $golden — remove it by hand (§6.1); it costs a monthly charge, not a broken recovery" + fi + fi + # No golden snapshot found is the normal steady state once this has run + # once; nothing to log. +fi diff --git a/flit-spec.md b/flit-spec.md index d767de8..4c6c938 100644 --- a/flit-spec.md +++ b/flit-spec.md @@ -839,24 +839,27 @@ Manual Hetzner snapshots are full compressed images with no deduplication and no automatic retention — a snapshot taken one month still bills months later unless deleted. Do not use manual snapshots as a substitute for the rolling backup. -#### The golden snapshot is temporary — delete it +#### The golden snapshot is temporary — deleted automatically Phase 12 takes one, and phase 13's restore drill uses it because no automatic backup exists yet -(§12.1). **After the first successful `drill-rebuild.sh`, delete it.** +(§12.1). It is used exactly once: the first quarterly `drill-restore.sh` run made without +`--from-snapshot` restores from a real automatic backup instead, which is proof that backup now +exists, and `drill-restore.sh` deletes the golden snapshot at the end of that same passing run — +the deletion-side half of the condition `phase_12_snapshot()` checks on the creation side. `phase_12_snapshot()` skips entirely once `hcloud image list --type backup` reports a backup -created from `$FLIT_SERVER`, which is the creation-side half of that same instruction: an -automatic backup existing is exactly the state in which the substitute is no longer needed. -Without that condition the phase created another image on every full run, so deleting one per -this section was undone by the next `bootstrap.sh`. - -It is used exactly once. Quarterly drills use the most recent automatic backup instead, so -nothing reads it again — while it bills per GB every month and diverges further from reality as -the machine changes. A months-old golden image is worse than no golden image, because it invites -restoring from it. - -Once the rebuild drill has passed, **the scripts are the golden image**. That is §10's entire -argument against Docker, and keeping a stale snapshot around quietly contradicts it. +created from `$FLIT_SERVER`, which is the creation-side half of that condition: an automatic +backup existing is exactly the state in which the substitute is no longer needed. Without that +condition the phase created another image on every full run, so a deletion elsewhere was undone +by the next `bootstrap.sh`. + +Deletion is best-effort and never fails an already-passing drill — a failed `hcloud image delete` +warns instead, since the cost of a leftover snapshot is a monthly charge, not a broken recovery. +A months-old golden image is worse than no golden image, because it invites restoring from it and +diverges further from reality as the machine changes. + +Once an automatic backup exists, **the scripts are the golden image**. That is §10's entire +argument against Docker, and a stale snapshot lingering would quietly contradict it. #### A snapshot clone inherits the imaged server's tailnet identity