diff --git a/crates/didbot-pds/tests/durability.rs b/crates/didbot-pds/tests/durability.rs index 5646d739..3c0b3d14 100644 --- a/crates/didbot-pds/tests/durability.rs +++ b/crates/didbot-pds/tests/durability.rs @@ -872,6 +872,81 @@ fn a_log_that_is_mostly_history_is_rewritten_as_state() { let _ = std::fs::remove_dir_all(&dir); } +/// A log that reached its budget is under it again after the restart that +/// compacts it. +/// +/// This is what `--log-budget` is worth having: the budget bounds the file, +/// the compaction bounds it against the state, and the deployment that ran +/// out of budget on history rather than on state takes its room back by being +/// restarted. What it costs is a restart -- the compaction is the one at +/// startup -- which is why the number a deployment picks has to be one its +/// state fits inside and not merely one its history reaches slowly. +#[test] +fn a_log_that_filled_its_budget_writes_again_once_it_is_compacted() { + let dir = scratch("budget-compaction"); + + // Churn, so the log is mostly history: twelve provisioned, eleven gone. + let full = { + let (durable, pds) = boot(&dir, None); + let mut dids = Vec::new(); + for n in 0..12 { + dids.push( + pds.provision(request(&format!("churn{n}"))) + .expect("provision") + .account + .did, + ); + } + for did in &dids[1..] { + pds.delete(did.as_str()).expect("delete"); + } + durable.wal().sync().expect("a sync"); + durable.wal().len() + }; + + // A budget this log is already at. Nothing is written under it before the + // restart, so this is the deployment an operator finds: serving, and + // refusing every write. + { + let (durable, pds) = boot(&dir, None); + let compacted = durable.wal().len(); + assert!( + compacted < full, + "the restart did not compact the log: {full} bytes stayed {compacted}" + ); + // Measured against the log as it stood before the compaction, which + // is the size the budget was reached at. + durable.wal().set_capacity(Some(compacted)); + let refusal = pds + .provision(request("afterwards")) + .expect_err("a log at its budget provisioned an account"); + assert!( + matches!( + refusal, + ProvisionError::Store(_) | ProvisionError::Record(_) + ), + "a log at its budget refused with {refusal:?} rather than for want of room" + ); + } + + // The same budget, one restart later: the compaction already happened + // above, so what this asserts is that the budget is a bound the state + // fits inside rather than one the deployment cannot come back from. + let (durable, pds) = boot(&dir, None); + let budget = durable.wal().len() + 64 * 1024; + durable.wal().set_capacity(Some(budget)); + pds.provision(request("afterwards")) + .expect("a compacted log under the same budget refused a write"); + assert!( + durable.wal().len() <= budget, + "the log holds {} bytes of its {budget} byte budget", + durable.wal().len() + ); + assert_eq!(pds.accounts().len(), 2, "the survivor and the new account"); + + let _ = std::fs::remove_dir_all(&dir); +} + /// A compaction rewrites the log from live state rather than replaying it, so /// every fact that lives on an account rather than in an entry of its own has /// to be carried across by hand. A freeze is one of those, and losing it is diff --git a/crates/didbot-serve/src/bin/didbot-pds.rs b/crates/didbot-serve/src/bin/didbot-pds.rs index 458d6a20..28a7c3ad 100644 --- a/crates/didbot-serve/src/bin/didbot-pds.rs +++ b/crates/didbot-serve/src/bin/didbot-pds.rs @@ -2129,13 +2129,16 @@ mod config_tests { // these are the shell-side placeholders `sed` then replaces, and a // new one has to be given a value here or this test fails rather // than skipping it. - const PLACEHOLDERS: [(&str, &str); 6] = [ + const PLACEHOLDERS: [(&str, &str); 7] = [ ("ZONE_PLACEHOLDER", "agents.quernstone.example"), ("OWNER_PLACEHOLDER", "did:web:owner.example"), ("ACME_ENVIRONMENT_PLACEHOLDER", "staging"), ("ROUTE53_ZONE_ID_PLACEHOLDER", "Z0QUERNSTONE"), ("RECORD_TARGET_PLACEHOLDER", "203.0.113.47"), ("RELAY_HOSTNAME_PLACEHOLDER", "relay.quernstone.example"), + // A tenth of the default 20 GiB data volume, which is what + // `local.log_budget_bytes` computes -- see infra/pds/ec2.tf. + ("LOG_BUDGET_PLACEHOLDER", "2147483648"), ]; let template = include_str!("../../../../infra/pds/templates/user_data.sh.tftpl"); @@ -2200,6 +2203,27 @@ mod config_tests { so an operator on the instance cannot open it" ); + // The log shares its volume with the blob store, so the size it + // stops at is a number this deployment chooses rather than the + // size of the disk. A unit that dropped the flag would take the + // binary's own default -- no budget -- and the first symptom + // would be a volume with no room for a blob. + let budget = args + .iter() + .position(|arg| arg == "--log-budget") + .and_then(|at| args.get(at + 1)) + .unwrap_or_else(|| { + panic!( + "the unit stopped passing --log-budget, so this deployment's log \ + grows until the volume it shares with the blob store is full: {args:?}" + ) + }); + assert_eq!( + budget.parse::().ok(), + Some(2_147_483_648), + "--log-budget is {budget}, which is not the substituted value" + ); + let relay = args .iter() .position(|arg| arg == "--relay-hostname") @@ -2217,13 +2241,23 @@ mod config_tests { (false, None) => {} } - if let Err(err) = parse_args(args.clone().into_iter()) { - panic!( + let parsed = match parse_args(args.clone().into_iter()) { + Ok(parsed) => parsed, + Err(err) => panic!( "the deployment's unit runs `didbot-pds {}`, which this binary refuses: \ {err}", args.join(" ") - ); - } + ), + }; + // Parsing is half of it: the budget has to arrive as the number + // the unit wrote, since that is the one this run refuses writes + // past. + assert_eq!( + parsed.log_budget, + Some(2_147_483_648), + "the unit's --log-budget reached this binary as {:?}", + parsed.log_budget + ); } } } diff --git a/docs/deployment.md b/docs/deployment.md index f6c4fe89..8ee23f19 100644 --- a/docs/deployment.md +++ b/docs/deployment.md @@ -379,6 +379,34 @@ the `Dockerfile` rather than left to `useradd`, and which also re-homes a restored snapshot whose files were written by a different uid. +### The log's budget + +`--log-budget` is the number of bytes the log may reach; past it an append is +refused before a byte is written, the file stands exactly where it was, and +the write comes back as `507 StorageFull`. `infra/pds/variables.tf`'s +`log_budget_bytes` sets it, and takes a tenth of `data_volume_size_gb` unless +a number is pinned — **2 GiB on the default 20 GiB volume**. The remaining +nine tenths are the blob store's, which shares this volume and grows in +megabytes at a time where the log grows in hundreds of bytes. + +What has to fit inside the number is the deployment's *state*, not its +history. Every start rewrites the log as one entry per live thing once the +history has run to `didbot_pds::COMPACT_RATIO` times the state, which at a few +hundred bytes an entry puts millions of live records inside 2 GiB. A released +name, a deleted account, an overwritten record and a reissued credential are +all history a compaction drops, so a log that reached its budget on churn is +under it again after a restart — `didbot-pds/tests/durability.rs` asserts that +round trip. + +An operator sees this in three places. The startup line `the write-ahead log +will refuse writes past its budget` carries the `budget` and the bytes `held`, +on every boot. `compacted the write-ahead log` on a later boot says the +rewrite ran and how many entries it kept. And a deployment at its budget +answers `507` whose `message` names the bytes held, the budget, and the bytes +the refused entry wanted. A budget a compaction cannot bring the log below is +a deployment whose state alone exceeds the number: raise `log_budget_bytes`, +or raise `data_volume_size_gb` and with it the tenth this derives. + ## Secrets, and where each lives Three, and the process holds all three itself: diff --git a/infra/pds/ec2.tf b/infra/pds/ec2.tf index c20bc38f..739dbc2b 100644 --- a/infra/pds/ec2.tf +++ b/infra/pds/ec2.tf @@ -64,6 +64,7 @@ resource "aws_instance" "pds" { acme_environment = var.acme_environment record_target = aws_eip.pds.public_ip relay_hostname = var.relay_hostname + log_budget_bytes = local.log_budget_bytes }) # user_data changes (a new image tag, a config change) should not silently @@ -117,6 +118,10 @@ locals { # which user_data.sh.tftpl resolves at boot rather than this file guessing # the instance family. ecr_registry = element(split("/", var.container_image), 0) + # A tenth of the data volume unless a number was pinned. Derived rather than + # defaulted in variables.tf so that resizing the volume moves the log's + # budget with it: a variable's default cannot read another variable. + log_budget_bytes = coalesce(var.log_budget_bytes, floor(var.data_volume_size_gb * 1024 * 1024 * 1024 / 10)) } # The data directory's volume: the write-ahead log, the blob store and the diff --git a/infra/pds/templates/user_data.sh.tftpl b/infra/pds/templates/user_data.sh.tftpl index a6041b56..3c9c7048 100644 --- a/infra/pds/templates/user_data.sh.tftpl +++ b/infra/pds/templates/user_data.sh.tftpl @@ -135,6 +135,17 @@ ExecStartPre=-/usr/bin/docker rm -f didbot-pds # enough to hold on the order of a million request lines at the ~250 bytes # json-file's envelope makes of one. # +# `--log-budget` is the size the write-ahead log stops at. The log shares +# /data with the blob store (see infra/pds/ec2.tf's `aws_ebs_volume.data`), +# and past this many bytes an append is refused before it is written: the +# deployment answers `StorageFull` with a log that is byte for byte what it +# was, rather than discovering the size of the volume with a partial frame in +# the file. Every start rewrites the log as one entry per live thing once the +# history has run to four times the state (`didbot_pds::COMPACT_RATIO`), so +# what has to fit under this number is what the deployment holds and not what +# it has done. The value comes from infra/pds/variables.tf's +# `log_budget_bytes`, a tenth of the data volume. +# # `--estop-socket` puts the e-stop admin socket on the mounted data volume, # where it is `/data/pds/estop.sock` on this host and reachable from a break- # glass SSH or SSM session without entering the container. That socket is the @@ -162,6 +173,7 @@ ExecStart=/usr/bin/docker run --rm --name didbot-pds \ --owner OWNER_PLACEHOLDER \ --port 443 \ --data /data \ + --log-budget LOG_BUDGET_PLACEHOLDER \ --tls acme \ --acme-environment ACME_ENVIRONMENT_PLACEHOLDER \ --route53-zone-id ROUTE53_ZONE_ID_PLACEHOLDER \ @@ -189,6 +201,7 @@ sed -i \ -e "s|RECORD_TARGET_PLACEHOLDER|${record_target}|" \ -e "s|ACME_ENVIRONMENT_PLACEHOLDER|${acme_environment}|" \ -e "s|RELAY_HOSTNAME_PLACEHOLDER|${relay_hostname}|" \ + -e "s|LOG_BUDGET_PLACEHOLDER|${log_budget_bytes}|" \ /etc/systemd/system/didbot-pds.service systemctl daemon-reload diff --git a/infra/pds/variables.tf b/infra/pds/variables.tf index 706823b4..096111d7 100644 --- a/infra/pds/variables.tf +++ b/infra/pds/variables.tf @@ -62,6 +62,29 @@ variable "data_volume_size_gb" { default = 20 } +# What the write-ahead log may grow to before the server refuses writes. The +# log shares its 20 GiB volume with the blob store, and the blob store is the +# half that grows in megabytes at a time -- so the log is held to a tenth of +# the volume and the remainder is left to blobs. See docs/deployment.md's +# "The log's budget". +variable "log_budget_bytes" { + description = <<-EOT + Bytes the write-ahead log may reach before writes are refused with + StorageFull, passed to the server as --log-budget. Null takes a tenth of + data_volume_size_gb, so the default 20 GiB volume gives the log 2 GiB and + leaves the rest to blob bytes. Every start rewrites the log as one entry + per live thing, so what has to fit inside this number is the deployment's + state and not its history: 2 GiB is millions of live entries. Set a + number to pin it, in bytes. + EOT + type = number + default = null + validation { + condition = var.log_budget_bytes == null || coalesce(var.log_budget_bytes, 0) >= 1048576 + error_message = "log_budget_bytes must be at least 1048576 (1 MiB)." + } +} + variable "vpc_id" { description = "VPC to deploy into." type = string