From fbdffbe3dcc0388121a948957f8c63d7f3274c27 Mon Sep 17 00:00:00 2001 From: zzstoatzz Date: Mon, 6 Apr 2026 16:26:21 -0500 Subject: [PATCH] mark DB success on did_cache hits isDbHealthy() is a 30s freshness check on last_db_success, but the markDbSuccess() call sites only fire on cache misses + the 10-min GC tick. with a hot did_cache in steady state, miss rate can dip for 30+ seconds, leaving the health flag stale and tripping k8s liveness probes even though the relay is healthy. cache hits use DB-derived data, so they're a valid signal that the data path is functioning. mark success on the fast path. cost is one clock_gettime + atomic store per ingestion event (~20 ns vDSO). the GC tick still provides real DB liveness as a backstop. Co-Authored-By: Claude Opus 4.6 (1M context) --- src/event_log.zig | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/src/event_log.zig b/src/event_log.zig index 7e65814..da3e4ce 100644 --- a/src/event_log.zig +++ b/src/event_log.zig @@ -432,8 +432,15 @@ pub const DiskPersist = struct { /// resolve a DID to a numeric UID. creates a new account row on first encounter. /// matches indigo's Relay.DidToUid → Account.UID mapping. pub fn uidForDid(self: *DiskPersist, did: []const u8) !u64 { - // fast path: check in-memory cache - if (self.did_cache.get(did)) |uid| return uid; + // fast path: check in-memory cache. mark DB success even on hits — + // a hot cache means the DB-derived data path is functioning, which + // is what /_readyz cares about. without this, the 30s health window + // depends entirely on cache misses + the 10-min GC tick, which can + // gap during steady-state and trip k8s liveness probes. + if (self.did_cache.get(did)) |uid| { + self.markDbSuccess(); + return uid; + } // check database if (try self.db.rowUnsafe( -- 2.51.2