#!/usr/bin/env bash # 30-tailscale.sh — join the tailnet. Runs as root on the server. # 30-tailscale.sh # # Must complete, and its marker must exist, before 40-firewall.sh will run. set -uo pipefail : "${FLIT_NAME:=flit}" U="${FLIT_USER:-dev}" : "${FLIT_IN_DRILL:=0}" AUTHKEY=${1:?usage: 30-tailscale.sh } MARKER="/var/lib/$FLIT_NAME/tailscale-up" SERIAL_FILE="/var/lib/$FLIT_NAME/server-serial" die() { echo "error: $*" >&2; exit 1; } # jq parses every tailscale status below, and 10-packages.sh — which installs # it — runs a phase LATER than this script does, so the numbers on these files # do not describe the order bootstrap.sh actually calls them in. Depending on a # neighbour having run is what broke here; ask the system instead (§5.3), and # install what is missing. if ! command -v jq >/dev/null 2>&1; then DEBIAN_FRONTEND=noninteractive apt-get update -qq \ && DEBIAN_FRONTEND=noninteractive apt-get install -y -qq jq \ || die "could not install jq, which this script needs to read tailscale status" fi if ! command -v tailscale >/dev/null 2>&1; then curl -fsSL https://tailscale.com/install.sh | sh || die "tailscale install failed" fi # Guard on observed state, never a sentinel (§5.3). state=$(tailscale status --json 2>/dev/null | jq -r '.BackendState // "Stopped"') if [[ "$state" != "Running" ]]; then tailscale up --authkey="$AUTHKEY" --ssh=false --accept-routes=false \ || die "tailscale up failed" fi [[ "$(tailscale status --json | jq -r .BackendState)" == "Running" ]] \ || die "tailscale is not Running after 'up'" # §4.2: a node key that expires becomes unreachable while public SSH is # closed — a scheduled lockout, roughly 180 days out, with no warning. Skipped # under FLIT_IN_DRILL: a drill node is created seconds before this runs and # destroyed when the drill ends, and disabling expiry is a manual admin-console # action against a node that does not exist until the drill creates it — the # assertion is unsatisfiable there by construction, not weaker. Production # still enforces it unconditionally. expiry=$(tailscale status --json | jq -r '.Self.KeyExpiry // empty') if [[ "$FLIT_IN_DRILL" != 1 && -n "$expiry" && "$expiry" != "null" ]]; then days=$(( ( $(date -d "$expiry" +%s) - $(date +%s) ) / 86400 )) cat >&2 <"$MARKER" # hcloud server create-image (phase 12) snapshots this host including # /var/lib/tailscale/tailscaled.state (§6.1) — a server later created from # that image would otherwise boot holding this host's tailscale identity. # Install a oneshot that resets /var/lib/tailscale on any host whose DMI # serial does not match the one recorded below, ordered # Before=tailscaled.service and pulled into the same boot transaction via # WantedBy=tailscaled.service, so the reset is guaranteed to run before # tailscaled reads the state it would strip, rather than racing whatever runs # cloud-init. # # The unit has no User=, unlike the timer units in 70-timers.sh, because # resetting /var/lib/tailscale needs root. That is why the copy it runs lives # under /usr/local/sbin rather than in the delivered tree: $U can write # ~/$FLIT_NAME, this box runs Claude Code against untrusted repositories # (§5.7), and a root unit executing a $U-writable path is root for anything # that can write it. GUARD_UNIT="${FLIT_NAME}-tailscale-clone-guard" GUARD_BIN="/usr/local/sbin/${FLIT_NAME}-reset-tailscale-clone" install -m 0755 -o root -g root \ "/home/$U/$FLIT_NAME/server/bin/reset-tailscale-clone.sh" "$GUARD_BIN" \ || die "could not install the clone guard to $GUARD_BIN" cat >"/etc/systemd/system/${GUARD_UNIT}.service" <"$SERIAL_FILE" 2>/dev/null \ || echo "warning: could not read product_serial; the clone guard has nothing to compare against yet" >&2 echo "tailnet joined as $(cat "$MARKER"), key expiry disabled" echo "clone guard installed: ${GUARD_UNIT}.service"