From 086aefda1cc82d230e671acec12b4636465abcc2 Mon Sep 17 00:00:00 2001 From: dietrich ayala Date: Fri, 7 Aug 2026 10:54:05 +0200 Subject: [PATCH] let a restore drill reach its instance by tailnet address too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Third site of the same defect. On the restore path DRILL_IP holds a MagicDNS name, so a laptop that cannot complete that name spends all 40 attempts and then dies blaming the ephemeral Tailscale key or MagicDNS propagation — neither of which is wrong, and neither of which is the problem when the node is up and answering on its tailnet address. The name is still tried first, because when it works it proves the resolution path the drill exists to prove. Only after that attempt fails does it ask tailscale for the address, which does not go through DNS. Succeeding that way warns and rewrites DRILL_IP, so drill_ssh() and every later step follow. --- lib/drill-common.sh | 23 ++++++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/lib/drill-common.sh b/lib/drill-common.sh index 3c18f54..8688bd0 100644 --- a/lib/drill-common.sh +++ b/lib/drill-common.sh @@ -170,7 +170,7 @@ fault here — set HCLOUD_LOCATION to one that has it and re-run."; } DRILL_IP="$public_ip" info "waiting for ssh on $DRILL_IP" fi - local attempt err="" + local attempt err="" ts_ip="" for attempt in $(seq 1 40); do # Keep the last failure rather than discarding it, same reasoning as # bootstrap.sh's wait_for_ssh(): "unreachable" alone does not say whether @@ -181,6 +181,27 @@ fault here — set HCLOUD_LOCATION to one that has it and re-run."; } info "reachable after ${attempt} attempt(s)" return 0 fi + # A name that will not resolve is not a node that is down. On the restore + # path DRILL_IP holds a MagicDNS name, so a laptop that cannot complete + # that name — macOS on a phone hotspot ends up with no search domain for + # the tailnet suffix — spends all 40 attempts and then dies blaming the + # ephemeral Tailscale key or MagicDNS propagation, neither of which is + # wrong. tailscale knows the address without going through DNS, so it is + # asked before an attempt is written off. The name is still tried first: + # when it works it proves the resolution path the drill is meant to prove. + if [[ "$DRILL_IP" == "$DRILL_INSTANCE" ]] \ + && command -v tailscale >/dev/null 2>&1 \ + && ts_ip=$(tailscale ip -4 "$DRILL_INSTANCE" 2>/dev/null) && [[ -n "$ts_ip" ]] \ + && err=$(ssh -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o ConnectTimeout=5 \ + "$FLIT_USER@$ts_ip" true 2>&1); then + warn "'$DRILL_INSTANCE' does not resolve on this network; reaching it at $ts_ip instead. +The drill continues against that address. A name that will not resolve here is +not a node that is down, and every later step goes through drill_ssh(), which +uses whatever this resolved to." + DRILL_IP="$ts_ip" + info "reachable after ${attempt} attempt(s)" + return 0 + fi # Same fast-fail as bootstrap.sh's wait_for_ssh(): waiting cannot fix a # host-key mismatch, so do not spend 400s discovering one 10s at a time. if grep -q 'REMOTE HOST IDENTIFICATION HAS CHANGED' <<<"$err"; then -- 2.51.2