diff --git a/src/admin/backfill.rs b/src/admin/backfill.rs index 7974cd5..ccda7e7 100644 --- a/src/admin/backfill.rs +++ b/src/admin/backfill.rs @@ -555,9 +555,21 @@ async fn run_pipelined_resolve_and_fetch( if cancelled.load(Ordering::Relaxed) { return None; } - let result = - profile::resolve_pds_endpoint(&state.http, &state.config.plc_url, &did) - .await; + // Bound the entire resolution of one DID (DNS, connect, and + // any rate-limit retry loop) so a single stuck DID can never + // hang the resolver stream. On expiry the DID is skipped. + let result = match tokio::time::timeout( + crate::http_retry::RESOLVE_DEADLINE, + profile::resolve_pds_endpoint(&state.http, &state.config.plc_url, &did), + ) + .await + { + Ok(result) => result, + Err(_) => Err(AppError::Internal(format!( + "PDS resolution timed out after {}s", + crate::http_retry::RESOLVE_DEADLINE.as_secs() + ))), + }; Some((did, result)) } }) diff --git a/src/http_retry.rs b/src/http_retry.rs index c26255c..7e5cdcb 100644 --- a/src/http_retry.rs +++ b/src/http_retry.rs @@ -2,11 +2,16 @@ use std::time::Duration; /// Per-request timeout for outbound atproto network fetches (relay, PLC/DID /// resolution, PDS reads). Bounds a single HTTP attempt so a host that connects -/// but never responds fails instead of stalling a backfill job forever. This is -/// deliberately a per-request timeout, not a wall-clock deadline over a retry -/// loop, so rate-limit backoffs are left intact. +/// but never responds fails instead of stalling a backfill job forever. pub const REQUEST_TIMEOUT: Duration = Duration::from_secs(30); +/// Overall deadline for resolving a single DID's PDS endpoint. Unlike +/// [`REQUEST_TIMEOUT`], this bounds the *entire* resolution of one DID — +/// including DNS, connection setup, and any rate-limit retry/backoff loop — so +/// a single stuck DID can never hang the resolver stream. On expiry the DID is +/// treated as a resolution failure and skipped, and the backfill moves on. +pub const RESOLVE_DEADLINE: Duration = Duration::from_secs(30); + /// Parse rate-limit sleep duration from response headers. /// Checks `RateLimit-Reset` (Unix timestamp, used by XRPC servers) first, /// then `retry-after` (seconds), defaulting to 5s.