diff --git a/server/src/crate_server/handlers/health.gleam b/server/src/crate_server/handlers/health.gleam index f0721f3..f8b5229 100644 --- a/server/src/crate_server/handlers/health.gleam +++ b/server/src/crate_server/handlers/health.gleam @@ -1,7 +1,10 @@ //// Deployment probes: liveness proves the process can serve HTTP, while -//// readiness also proves the appview backing store remains available. +//// readiness reports on the appview's backing store. Only proof that the +//// store is unusable gates traffic; a store whose state is merely unknown +//// still leaves every route that never touches it serving. import crate_server/readiness.{type Check} +import gleam/json import wisp.{type Response} pub fn liveness() -> Response { @@ -9,8 +12,18 @@ pub fn liveness() -> Response { } pub fn readiness(check: Check) -> Response { - case readiness.is_ready(check) { - True -> wisp.json_response("{\"status\":\"ready\"}", 200) - False -> wisp.json_response("{\"status\":\"not ready\"}", 503) + case readiness.status(check) { + readiness.Ready -> wisp.json_response("{\"status\":\"ready\"}", 200) + readiness.Unknown(reason) -> reported("unknown", reason, 200) + readiness.Down(reason) -> reported("not ready", reason, 503) } } + +fn reported(status: String, reason: String, code: Int) -> Response { + json.object([ + #("status", json.string(status)), + #("reason", json.string(reason)), + ]) + |> json.to_string + |> wisp.json_response(code) +} diff --git a/server/src/crate_server/readiness.gleam b/server/src/crate_server/readiness.gleam index 435ea8e..9482027 100644 --- a/server/src/crate_server/readiness.gleam +++ b/server/src/crate_server/readiness.gleam @@ -20,6 +20,17 @@ import pog const cache_ttl_seconds = 5 +/// Three-valued because a boolean gate answers "unknown" and "unusable" the +/// same way, and this gate is what Caddy drops the only upstream on: a +/// database that is merely slow, busy or unreachable must stay `Unknown`, so +/// every route that never touches Postgres keeps serving. Only a completed +/// query proving the schema unusable is `Down`. +pub type Readiness { + Ready + Unknown(reason: String) + Down(reason: String) +} + /// Every store's own required tables, assembled here so the list stays /// exhaustive while each module stays the only place its table names live. pub fn required_tables() -> List(String) { @@ -51,12 +62,12 @@ pub fn schema_check_sql() -> String { } pub type Check { - Check(run: fn() -> Bool) + Check(run: fn() -> Readiness) } type State { State( - last: Option(#(Bool, Int)), + last: Option(#(Readiness, Int)), // Generation and start time of the refresh in flight; the generation keeps // a late reply from an abandoned attempt out of the current one. in_flight: Option(#(Int, Int)), @@ -66,8 +77,8 @@ type State { } type Msg { - Probe(Subject(Bool)) - ProbeResult(Int, Bool) + Probe(Subject(Readiness)) + ProbeResult(Int, Readiness) } // How long the actor waits on a fresh result before answering from cache. @@ -89,24 +100,37 @@ const refresh_overrun_seconds = 10 // answer forever, silently. const max_staleness_seconds = 30 +const never_probed = "no probe has completed yet" + pub fn ready() -> Check { - Check(fn() { True }) + Check(fn() { Ready }) } -pub fn unavailable() -> Check { - Check(fn() { False }) +pub fn postgres(conn: pog.Connection) -> Check { + cached(fn() { probe_schema(conn) }, clock.now_seconds) } -pub fn postgres(conn: pog.Connection) -> Check { - cached(fn() { schema_is_ready(conn) }, clock.now_seconds) +// Zero rows is proof the schema is unusable; an error is only ever an absence +// of proof. +fn probe_schema(conn: pog.Connection) -> Readiness { + let row = decode.success(Nil) + case + pog.query(schema_check_sql()) |> pog.returning(row) |> pog.execute(conn) + { + Ok(pog.Returned(rows: [_, ..], ..)) -> Ready + Ok(pog.Returned(rows: [], ..)) -> + Down("a required table is missing, or the role lacks CRUD on it") + Error(error) -> + Unknown("the schema check did not complete: " <> string.inspect(error)) + } } /// Coalesces calls through one actor. It spawns a worker for `probe` rather /// than running it, so a slow probe never wedges its mailbox: an unfinished -/// probe is "unknown" and answered from the last known state, and the worker's +/// probe is `Unknown` and answered from the last known state, and the worker's /// reply folds in later. That reply must arrive through the actor's selector; /// a bare receive loses it to the built-in catch-all. -pub fn cached(probe: fn() -> Bool, now: fn() -> Int) -> Check { +pub fn cached(probe: fn() -> Readiness, now: fn() -> Int) -> Check { let assert Ok(started) = actor.new_with_initialiser(1000, fn(subject) { let worker_replies = process.new_subject() @@ -128,28 +152,29 @@ pub fn cached(probe: fn() -> Bool, now: fn() -> Int) -> Check { |> actor.start let subject = started.data Check(fn() { - parallel.try_call(subject, call_budget_ms, Probe) |> result.unwrap(False) + parallel.try_call(subject, call_budget_ms, Probe) + |> result.unwrap(Unknown("the readiness actor did not answer in time")) }) } -pub fn is_ready(check: Check) -> Bool { +pub fn status(check: Check) -> Readiness { check.run() } fn handle( state: State, msg: Msg, - probe: fn() -> Bool, + probe: fn() -> Readiness, now: fn() -> Int, ) -> actor.Next(State, Msg) { case msg { Probe(reply) -> { - let #(available, next) = respond(state, probe, now) - process.send(reply, available) + let #(current, next) = respond(state, probe, now) + process.send(reply, current) actor.continue(next) } - ProbeResult(generation, available) -> - actor.continue(accept(state, generation, available, now)) + ProbeResult(generation, current) -> + actor.continue(accept(state, generation, current, now)) } } @@ -158,21 +183,21 @@ fn handle( fn accept( state: State, generation: Int, - available: Bool, + current: Readiness, now: fn() -> Int, ) -> State { case state.in_flight { - Some(#(current, _)) if current == generation -> - State(..state, last: Some(#(available, now())), in_flight: None) + Some(#(live, _)) if live == generation -> + State(..state, last: Some(#(current, now())), in_flight: None) _ -> state } } fn respond( state: State, - probe: fn() -> Bool, + probe: fn() -> Readiness, now: fn() -> Int, -) -> #(Bool, State) { +) -> #(Readiness, State) { case state.in_flight { Some(#(_, started_at)) -> case now() - started_at > refresh_overrun_seconds { @@ -187,7 +212,7 @@ fn respond( } } -fn needs_refresh(last: Option(#(Bool, Int)), now: fn() -> Int) -> Bool { +fn needs_refresh(last: Option(#(Readiness, Int)), now: fn() -> Int) -> Bool { case last { None -> True Some(#(_, checked_at)) -> now() - checked_at >= cache_ttl_seconds @@ -196,22 +221,22 @@ fn needs_refresh(last: Option(#(Bool, Int)), now: fn() -> Int) -> Bool { // The best answer available without waiting on the database, vouching for // nothing older than `max_staleness_seconds`. -fn fallback(last: Option(#(Bool, Int)), now: fn() -> Int) -> Bool { +fn fallback(last: Option(#(Readiness, Int)), now: fn() -> Int) -> Readiness { case last { - Some(#(available, checked_at)) -> + Some(#(current, checked_at)) -> case now() - checked_at > max_staleness_seconds { - True -> False - False -> available + True -> Unknown("the last completed probe is too old to vouch for") + False -> current } - None -> False + None -> Unknown(never_probed) } } fn refresh( state: State, - probe: fn() -> Bool, + probe: fn() -> Readiness, now: fn() -> Int, -) -> #(Bool, State) { +) -> #(Readiness, State) { let generation = state.next_generation let replies = state.worker_replies process.spawn_unlinked(fn() { @@ -225,22 +250,12 @@ fn refresh( ) // Only ProbeResult lands on `replies`; the catch-all keeps the match total. case process.receive(state.worker_replies, local_wait_ms) { - Ok(ProbeResult(reply_generation, available)) + Ok(ProbeResult(reply_generation, current)) if reply_generation == generation -> #( - available, - State(..started, last: Some(#(available, now())), in_flight: None), + current, + State(..started, last: Some(#(current, now())), in_flight: None), ) Ok(_) | Error(Nil) -> #(fallback(state.last, now), started) } } - -fn schema_is_ready(conn: pog.Connection) -> Bool { - let row = decode.success(Nil) - case - pog.query(schema_check_sql()) |> pog.returning(row) |> pog.execute(conn) - { - Ok(pog.Returned(rows: [_, ..], ..)) -> True - _ -> False - } -} diff --git a/server/test/health_test.gleam b/server/test/health_test.gleam index 60931ea..049d934 100644 --- a/server/test/health_test.gleam +++ b/server/test/health_test.gleam @@ -1,5 +1,6 @@ -//// Liveness only proves that the HTTP process can answer. Readiness also -//// proves the appview's configured backing store remains usable. +//// Liveness only proves that the HTTP process can answer. Readiness reports +//// on the appview's configured backing store, and gates traffic only on +//// proof that store is unusable. import crate_server/handlers/health import crate_server/parallel @@ -38,10 +39,33 @@ pub fn readiness_is_ok_when_the_appview_store_is_available_test() { } pub fn readiness_is_unavailable_when_the_appview_store_is_down_test() { - let response = health.readiness(readiness.unavailable()) + let response = + health.readiness(check_of(readiness.Down("known_users is missing"))) assert response.status == 503 assert string.contains(simulate.read_body(response), "not ready") + assert string.contains(simulate.read_body(response), "known_users is missing") +} + +// A boolean gate answers "unknown" and "unusable" alike, and Caddy drops its +// only upstream on the answer: a busy or unreachable database must not black +// out the routes that never touch it. +pub fn readiness_does_not_gate_traffic_on_an_unknown_store_test() { + let response = + health.readiness(check_of(readiness.Unknown("connection unavailable"))) + + assert response.status == 200 + assert string.contains(simulate.read_body(response), "unknown") + assert string.contains(simulate.read_body(response), "connection unavailable") +} + +pub fn a_probe_that_cannot_reach_the_database_stays_unknown_test() { + let check = + readiness.cached(fn() { readiness.Unknown("connection unavailable") }, fn() { + 0 + }) + + assert readiness.status(check) == readiness.Unknown("connection unavailable") } pub fn health_routes_keep_liveness_and_readiness_separate_test() { @@ -51,15 +75,24 @@ pub fn health_routes_keep_liveness_and_readiness_separate_test() { "https://resolver.test", "http://localhost:8080", ) - let context = support.stub_context(cfg) + let ctx = support.stub_context(cfg) - let liveness = - router.handle_request(simulate.request(http.Get, "/healthz"), context) - let readiness = - router.handle_request(simulate.request(http.Get, "/readyz"), context) + let live = router.handle_request(simulate.request(http.Get, "/healthz"), ctx) + let ready = router.handle_request(simulate.request(http.Get, "/readyz"), ctx) + + assert live.status == 200 + assert ready.status == 200 +} - assert liveness.status == 200 - assert readiness.status == 200 +fn check_of(current: readiness.Readiness) -> readiness.Check { + readiness.Check(fn() { current }) +} + +fn unknown(current: readiness.Readiness) -> Bool { + case current { + readiness.Unknown(_) -> True + readiness.Ready | readiness.Down(_) -> False + } } pub fn cached_readiness_coalesces_probes_inside_its_ttl_test() { @@ -68,14 +101,14 @@ pub fn cached_readiness_coalesces_probes_inside_its_ttl_test() { readiness.cached( fn() { process.send(calls, Nil) - True + readiness.Ready }, fn() { 100 }, ) list.repeat(Nil, 3) |> list.each(fn(_) { - assert readiness.is_ready(check) + assert readiness.status(check) == readiness.Ready }) assert support.drain_count(calls) == 1 @@ -84,53 +117,56 @@ pub fn cached_readiness_coalesces_probes_inside_its_ttl_test() { pub fn slow_failed_probe_is_cached_from_completion_test() { let calls = process.new_subject() let #(advance, now) = test_clock([100, 101, 105]) + let down = readiness.Down("known_users is missing") let check = readiness.cached( fn() { advance() process.send(calls, Nil) - False + down }, now, ) - assert !readiness.is_ready(check) - assert !readiness.is_ready(check) + assert readiness.status(check) == down + assert readiness.status(check) == down assert support.drain_count(calls) == 1 } -pub fn probe_timeout_before_any_success_reports_not_ready_test() { +pub fn probe_timeout_before_any_result_reports_unknown_test() { let calls = process.new_subject() let check = readiness.cached( fn() { process.send(calls, Nil) process.sleep(700) - True + readiness.Ready }, fn() { 0 }, ) - assert !readiness.is_ready(check) + // Nothing has been proven either way yet, so the gate must not close. + assert unknown(readiness.status(check)) process.sleep(900) - assert readiness.is_ready(check) + assert readiness.status(check) == readiness.Ready assert support.drain_count(calls) == 1 } pub fn probe_timeout_falls_back_to_last_known_state_test() { let seen = counter() let #(_advance, now) = test_clock([0, 10, 15, 16, 20]) + let down = readiness.Down("known_users is missing") let check = readiness.cached( fn() { case seen() { - 0 -> True + 0 -> readiness.Ready _ -> { process.sleep(700) - False + down } } }, @@ -138,15 +174,15 @@ pub fn probe_timeout_falls_back_to_last_known_state_test() { ) // Fast probe: a known-good state. - assert readiness.is_ready(check) + assert readiness.status(check) == readiness.Ready // TTL lapsed, refresh started but slow. Timing out must fall back to the - // last known state rather than report "not ready". - assert readiness.is_ready(check) + // last known state rather than invent one. + assert readiness.status(check) == readiness.Ready // The slow probe finishes and fails, which becomes the reported state. process.sleep(900) - assert !readiness.is_ready(check) + assert readiness.status(check) == down } type CountState { @@ -258,17 +294,17 @@ pub fn schema_probe_requires_existence_and_operation_privileges_test() { }) } -pub fn a_wedged_probe_eventually_reports_not_ready_test() { +pub fn a_wedged_probe_eventually_stops_vouching_test() { let seen = counter() let #(_advance, now) = test_clock([0, 0, 10, 10, 15, 25, 25, 35]) let check = readiness.cached( fn() { case seen() { - 0 -> True + 0 -> readiness.Ready _ -> { process.sleep_forever() - True + readiness.Ready } } }, @@ -276,15 +312,16 @@ pub fn a_wedged_probe_eventually_reports_not_ready_test() { ) // A known-good state. - assert readiness.is_ready(check) + assert readiness.status(check) == readiness.Ready // The refresh hangs, but the cached state is inside its staleness ceiling // and stays trusted. - assert readiness.is_ready(check) + assert readiness.status(check) == readiness.Ready // Past the ceiling, a database that hangs rather than errors must stop - // advertising as ready even though its probe never finished. - assert !readiness.is_ready(check) + // advertising as ready. It is unknown, not proven unusable, so it still + // must not close the gate. + assert unknown(readiness.status(check)) } pub fn a_crashing_worker_does_not_permanently_disable_probing_test() { @@ -295,16 +332,16 @@ pub fn a_crashing_worker_does_not_permanently_disable_probing_test() { fn() { case seen() { 0 -> panic as "the database connection was severed mid-query" - _ -> True + _ -> readiness.Ready } }, now, ) // The first probe's worker crashes before answering. - assert !readiness.is_ready(check) + assert unknown(readiness.status(check)) // Probing must recover once the crashed attempt has overrun, not stay wedged // behind a worker that died without replying. - assert readiness.is_ready(check) + assert readiness.status(check) == readiness.Ready }