diff --git a/docs/deployment.md b/docs/deployment.md index 0b507394..85389cf4 100644 --- a/docs/deployment.md +++ b/docs/deployment.md @@ -249,10 +249,13 @@ and the expiry it is measured against. Three consecutive failures raise that zone to `Warning`; inside three days of expiry it is `Critical` whether or not anything has failed, because at that point expiry itself is the risk. `CertificateFleet::alert_level` reports the worst level any zone is at and -`zone_alert_levels` says which zone it came from. Nothing pages on this yet — -see "alert on the things that fail quietly" in [deploy](../plan/deploy.md). -An expired certificate is a total outage for every hostname in that zone, -agents included. +`zone_alert_levels` says which zone it came from. `GET /health` carries both, +under `certificates`: `worst` is the summary an alarm expression matches on, +`zones` names each zone and its own level. `infra/pds/monitoring.tf` is what +reads it — a Route53 health check string-matching `worst`, and a CloudWatch +alarm on that check — so a zone above `ok` and a server that stopped +answering raise the same alarm. An expired certificate is a total outage for +every hostname in that zone, agents included. **The zone's `CAA` record is written by the server, on every start.** It is published before the first order and restated on each start, naming the diff --git a/infra/pds/monitoring.tf b/infra/pds/monitoring.tf new file mode 100644 index 00000000..cc102e13 --- /dev/null +++ b/infra/pds/monitoring.tf @@ -0,0 +1,78 @@ +# The alarm behind `GET /health`. +# +# Why a Route53 health check and not a CloudWatch metric filter: the signal +# this alarm is about -- how close each zone's certificate is to expiring +# with a failing renewal -- lives in the body of `GET /health` +# (`crates/didbot-serve/src/health.rs`), which the server answers over the +# public listener the internet already reaches. The instance's own log stays +# on the instance, read over `journalctl -u didbot-pds` (docs/operations.md). +# A Route53 health check is the mechanism in this account that reads an HTTP +# response body and publishes the result to CloudWatch, so it is the shorter +# path from the fact to the alarm: one check plus one alarm, against a log +# pipeline that would have to be shipped, parsed and paid for first. +# +# The check therefore covers two failures with one signal -- a server that +# stops answering, and a server that answers with a zone above `ok` -- and +# both want the same response from an operator: go and look at this host. +resource "aws_route53_health_check" "pds" { + type = "HTTPS_STR_MATCH" + fqdn = var.root_zone + port = 443 + resource_path = "/health" + request_interval = 30 + failure_threshold = 3 + measure_latency = false + + # `worst` is `didbot_serve::CertificateHealth`'s single-field summary: it + # is `ok` only while every configured zone is, so the absence of this + # exact string is a zone at `warning` or `critical`. Matching the summary + # rather than a per-zone entry is what keeps this one check correct for a + # deployment that adds a second zone later. + search_string = "\"worst\":\"ok\"" + + tags = { + Name = "didbot-pds-health" + } +} + +# Route53 publishes health check metrics into us-east-1 and nowhere else, so +# the alarm that reads them is created there whatever `var.aws_region` is. +provider "aws" { + alias = "us_east_1" + region = "us-east-1" +} + +resource "aws_cloudwatch_metric_alarm" "pds_health" { + provider = aws.us_east_1 + + alarm_name = "didbot-pds-health" + alarm_description = join(" ", [ + "The PDS answered GET /health without every configured zone's", + "certificate at `ok`, or stopped answering it. Renewal starts 30 days", + "before expiry and an expired certificate is a total outage for every", + "hostname in that zone; see docs/deployment.md.", + ]) + + namespace = "AWS/Route53" + metric_name = "HealthCheckStatus" + dimensions = { + HealthCheckId = aws_route53_health_check.pds.id + } + + # `Minimum` over a 60-second period: the health checkers report from + # several regions, and one region that cannot reach a healthy host is what + # `failure_threshold` on the check itself already absorbs. Two periods so + # a single deploy restart does not raise it. + statistic = "Minimum" + period = 60 + evaluation_periods = 2 + threshold = 1 + comparison_operator = "LessThanThreshold" + + # Missing data means the checkers reported nothing, which is not evidence + # the deployment is fine. + treat_missing_data = "breaching" + + alarm_actions = var.alarm_sns_topic_arns + ok_actions = var.alarm_sns_topic_arns +} diff --git a/infra/pds/variables.tf b/infra/pds/variables.tf index ec0b44e6..8c884d83 100644 --- a/infra/pds/variables.tf +++ b/infra/pds/variables.tf @@ -116,3 +116,13 @@ variable "acme_environment" { error_message = "acme_environment must be \"staging\" or \"production\"." } } + +# Where the health alarm sends its state changes. Empty by default: the alarm +# is still created and still changes state in the console with no topic here, +# and a topic is per-operator -- an email subscription has to be confirmed by +# the human receiving it, which is not something a `terraform apply` can do. +variable "alarm_sns_topic_arns" { + description = "SNS topic ARNs the didbot-pds health alarm notifies on ALARM and OK. Must be topics in us-east-1, where Route53 publishes health check metrics." + type = list(string) + default = [] +} diff --git a/plan/deploy.md b/plan/deploy.md index 8fe2fde1..d34c3b5f 100644 --- a/plan/deploy.md +++ b/plan/deploy.md @@ -115,8 +115,10 @@ own bucket and distribution to provision — see that epic's open items. logic now -- `didbot_tls::renew::RenewalTracker::alert_level` climbs from `Ok` to `Warning` after repeated failures and to `Critical` inside a fixed window of actual expiry, tested without a network in - `crates/didbot-tls/src/renew.rs` -- but it only reaches `tracing` today, - not [alerts](alerts.md)'s channel. + `crates/didbot-tls/src/renew.rs`, and `GET /health`'s `certificates` + carries every zone's level so `infra/pds/monitoring.tf`'s Route53 health + check and CloudWatch alarm can read it. The other three failures above, + and [alerts](alerts.md)'s own channel, are what is left here. - [ ] **Debugging without a redeploy.** Log level changeable at runtime, and a way to ask why one request was refused that does not require [ops-dashboard](ops-dashboard.md) to exist yet. @@ -226,8 +228,9 @@ that had not landed as this revision was written. [deployment](../docs/deployment.md): the existing certificate keeps being served, the failure is logged with the zone and the expiry it is measured against, and the level escalates before expiry rather than at - it. Nothing pages on it — "alert on the things that fail quietly" - above is where that stays. + it. `GET /health` reports the same levels, and `infra/pds/monitoring.tf` + alarms on them — see "alert on the things that fail quietly" above for + the failures that channel still covers. - [x] **The certificate hot-reload seam, and TLS termination itself.** `didbot-tls` obtains a wildcard-covering certificate over ACME DNS-01 (`instant-acme`, on `ring` — its `hyper-rustls` feature is deliberately