From 5c0916a36fdab7d1b7cec0ffb5e6cf53bc97b06e Mon Sep 17 00:00:00 2001 From: "@permadeath.com" Date: Tue, 18 Aug 2026 17:42:16 -0400 Subject: [PATCH] fix(match-launch)!: size a match up to sixteen vCPUs, and round memory up MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Eight human seats demand 8500 milli-vCPU. The tier table stopped at 8192, so the pick fell off the end of it and served eight client JVMs on eight vCPUs without saying so. Fargate's 16 vCPU tier is in the table now. The table also carries each tier's memory step, which is what the old code got wrong in the other direction: memory was a spelled-out match over 1024-steps, and 24576 MiB — a legal 8192-tier figure, and exactly what eight client heaps want — fell through its catch-all to 20480. Less than was asked for is the one direction rounding must never go. `fit_memory` rounds up onto the tier's step, caps before it rounds so it cannot overflow, and `TaskSize::memory` is an owned String because the legal values are a range per tier rather than a list worth spelling out. Breaking for any caller of `TaskSize`, which is `aws::run_task` and nothing else; a match of any shape that already fitted keeps the size it had. Co-Authored-By: Claude Opus 5 (1M context) --- plan/match-launch.md | 23 ++++ services/api/src/matches/aws.rs | 2 +- services/api/src/matches/sizing.rs | 177 +++++++++++++++++++++-------- 3 files changed, 154 insertions(+), 48 deletions(-) diff --git a/plan/match-launch.md b/plan/match-launch.md index 6d11a9e..a87a302 100644 --- a/plan/match-launch.md +++ b/plan/match-launch.md @@ -16,6 +16,12 @@ Fargate, which also exercised arena's never-run upload path for the first time. ## Still open +- [ ] **What 16 vCPU costs is not modelled.** A full eight-seat lobby now + asks for the top Fargate tier, which is four times the two-vCPU shape + every match ran on until PvP. Each launch logs the size it chose; + nothing yet adds them up. [cost-model](cost-model.md) is where that + belongs, and eight seats is the shape that makes it worth doing. + ## Done - [x] **The skeleton.** The Cargo workspace and `headquarters-api`: config @@ -50,3 +56,20 @@ Fargate, which also exercised arena's never-run upload path for the first time. Ordering mattered: arena had to drop its own fallback first (arena #43), since an un-updated arena against a manifest with no signing targets has no upload route at all. Both were deployed on 2026-08-08, arena first. +- [x] **Sizing reaches eight human seats.** `task_size` adds a vCPU per + client JVM, so eight humans demand 8500 milli-vCPU — past the 8192 + tier, and the tier table stopped there. The pick fell off the end of + the table and served eight clients on eight vCPUs without saying so. + Fargate's 16 vCPU tier is now in the table, and the table carries each + tier's memory *step* as well as its range: 4 vCPU takes 1 GiB + increments, 8 vCPU 4 GiB, 16 vCPU 8 GiB. + + The step is what the old code got wrong in the other direction too. + Memory was a spelled-out match over 1024-steps with a catch-all arm, + and 24576 MiB — a legal 8192-tier figure, and exactly what eight + client heaps want — had no arm of its own, so it fell through to + 20480. Less than was asked for is the one direction rounding must + never go. `fit_memory` rounds up onto the tier's step instead, caps + before it rounds so it cannot overflow, and `TaskSize::memory` is an + owned `String` because the legal values are a range per tier rather + than a list short enough to spell out. diff --git a/services/api/src/matches/aws.rs b/services/api/src/matches/aws.rs index c5ff541..c86d922 100644 --- a/services/api/src/matches/aws.rs +++ b/services/api/src/matches/aws.rs @@ -179,7 +179,7 @@ impl Aws { // match asks for more only for its own task. let overrides = TaskOverride::builder() .cpu(size.cpu) - .memory(size.memory) + .memory(&size.memory) .container_overrides( ContainerOverride::builder() .name("arena") diff --git a/services/api/src/matches/sizing.rs b/services/api/src/matches/sizing.rs index 70bc25e..2235c8f 100644 --- a/services/api/src/matches/sizing.rs +++ b/services/api/src/matches/sizing.rs @@ -29,18 +29,65 @@ //! numbers get refined - from real matches, not a laptop stand-in. /// What RunTask is asked for. Strings because the ECS API takes strings. +/// +/// `memory` is owned rather than `&'static str`: the legal values are a +/// range per tier, not a short list, and spelling them all out was already +/// wrong once — 24576 MiB, which eight clients want, fell through a match +/// arm to 20480. #[derive(Debug, PartialEq)] pub struct TaskSize { pub cpu: &'static str, - pub memory: &'static str, + pub memory: String, } -/// Fargate's valid CPU tiers and each tier's memory range, MiB. Fargate -/// rejects combinations outside these, so the pick snaps into them. -const TIERS: &[(u32, &str, u32, u32)] = &[ - (2048, "2048", 4096, 16384), - (4096, "4096", 8192, 30720), - (8192, "8192", 16384, 61440), +/// One Fargate CPU tier: its size in Fargate units, the string RunTask +/// wants, and the memory range and step it accepts. Fargate rejects any +/// combination outside these, so the pick snaps into them. +/// +/// The step is per tier and is not decoration: 4 vCPU takes memory in 1 GiB +/// increments, 8 vCPU in 4 GiB, and 16 vCPU in 8 GiB. Asking 8 vCPU for +/// 18432 MiB is as invalid as asking it for 4096. +struct Tier { + units: u32, + cpu: &'static str, + memory_min: u32, + memory_max: u32, + memory_step: u32, +} + +const TIERS: &[Tier] = &[ + Tier { + units: 2048, + cpu: "2048", + memory_min: 4096, + memory_max: 16384, + memory_step: 1024, + }, + Tier { + units: 4096, + cpu: "4096", + memory_min: 8192, + memory_max: 30720, + memory_step: 1024, + }, + Tier { + units: 8192, + cpu: "8192", + memory_min: 16384, + memory_max: 61440, + memory_step: 4096, + }, + // 16 vCPU, and the reason this tier exists here: eight human seats ask + // for 8500 milli-vCPU, which is past what the 8192 tier covers. Without + // it the pick fell off the end of the table and served an eight-client + // match on eight vCPUs, silently. + Tier { + units: 16384, + cpu: "16384", + memory_min: 32768, + memory_max: 122880, + memory_step: 8192, + }, ]; // Production-anchored coefficients, in milli-vCPU. See the module doc for @@ -61,6 +108,11 @@ const FLOOR_MEMORY_MIB: u32 = 8192; /// The size a match of this shape wants. Never below the task definition's /// own two-vCPU shape: the definition is the floor, this only asks up. +/// +/// The top tier is a ceiling, not a promise: a shape that wants more than +/// 16 vCPU gets 16 and runs slower, the same way the old top tier behaved. +/// Nothing here refuses to launch — what stops a lobby growing past what +/// Fargate can serve belongs in the lobby, not in the sizer. pub fn task_size(humans: u32, bots: u32, units: u32) -> TaskSize { let mut demand = BASE_MILLI + HUMAN_MILLI * humans; if bots > 0 { @@ -75,44 +127,30 @@ pub fn task_size(humans: u32, bots: u32, units: u32) -> TaskSize { let memory_wanted = (BASE_MEMORY_MIB + HUMAN_MEMORY_MIB * humans).max(FLOOR_MEMORY_MIB); // Smallest tier that covers the CPU demand (1024 Fargate units ≈ 1000 - // milli-vCPU), with the memory clamped into that tier's legal range. - for (units_cpu, cpu, memory_min, memory_max) in TIERS { - if demand <= units_cpu * 1000 / 1024 { - let memory = memory_wanted.clamp(*memory_min, *memory_max); - return TaskSize { - cpu, - memory: memory_str(memory), - }; - } - } - let (_, cpu, memory_min, memory_max) = TIERS[TIERS.len() - 1]; + // milli-vCPU), with the memory snapped into what that tier accepts. + let tier = TIERS + .iter() + .find(|tier| demand <= tier.units * 1000 / 1024) + .unwrap_or(&TIERS[TIERS.len() - 1]); TaskSize { - cpu, - memory: memory_str(memory_wanted.clamp(memory_min, memory_max)), + cpu: tier.cpu, + memory: fit_memory(memory_wanted, tier).to_string(), } } -/// Fargate memory values are 1024-steps; the wanted figure rounds up. -fn memory_str(mib: u32) -> &'static str { - // A static str keeps TaskSize Copy-ish and the call sites allocation - // free; the set of values Fargate accepts is small enough to spell out. - match mib.div_ceil(1024) * 1024 { - 0..=4096 => "4096", - 5120 => "5120", - 6144 => "6144", - 7168 => "7168", - 8192 => "8192", - 9216 => "9216", - 10240 => "10240", - 11264 => "11264", - 12288 => "12288", - 13312 => "13312", - 14336 => "14336", - 15360 => "15360", - 16384 => "16384", - // The 8192 tier steps by 4096, not 1024. - _ => "20480", - } +/// The wanted memory as a figure this tier actually accepts: rounded *up* +/// to its step, so nothing is quietly served less than it asked for, and +/// held inside the tier's range. +/// +/// The ceiling is applied before the rounding rather than after. Rounding +/// first is what a reader expects and it overflows on a large enough +/// figure; capping first cannot, and costs nothing, because every tier's +/// maximum is itself a multiple of its step — so a capped value is already +/// on the step and the rounding leaves it alone. +fn fit_memory(wanted: u32, tier: &Tier) -> u32 { + let capped = wanted.min(tier.memory_max); + let stepped = capped.div_ceil(tier.memory_step) * tier.memory_step; + stepped.max(tier.memory_min) } #[cfg(test)] @@ -152,22 +190,67 @@ mod tests { assert!(size.memory.parse::().unwrap() >= 16384); } + /// Eight human seats: 8500 milli-vCPU, past what the 8192 tier covers. + /// Before the 16384 tier existed this fell off the end of the table and + /// served eight client JVMs on eight vCPUs without saying so. + #[test] + fn eight_humans_take_the_sixteen_vcpu_tier() { + let size = task_size(8, 0, 16); + assert_eq!(size.cpu, "16384"); + assert_eq!( + size.memory, "32768", + "eight client heaps want 24576 MiB, and the tier's floor is what \ + it can actually be given" + ); + } + + /// The arithmetic bug the eight-seat case turned up: 24576 MiB is a + /// legal 8192-tier value (that tier steps by 4096) and the old spelled- + /// out table had no arm for it, so it fell through to 20480 — less than + /// was asked for, which is the one direction rounding must never go. + #[test] + fn memory_rounds_up_into_the_tier_step_never_down() { + let eight = &TIERS[2]; + assert_eq!(eight.cpu, "8192"); + assert_eq!(fit_memory(24576, eight), 24576, "an exact step is kept"); + assert_eq!(fit_memory(20481, eight), 24576, "a part step rounds up"); + assert_eq!(fit_memory(1, eight), eight.memory_min); + assert_eq!(fit_memory(u32::MAX, eight), eight.memory_max); + } + /// Memory follows the client heaps and stays inside the chosen tier's - /// legal range for every shape the catalog can produce. + /// legal range — and on its step — for every shape a lobby can now + /// produce, eight seats included. #[test] fn memory_lands_in_the_tier_range() { - for humans in 0..=5 { - for bots in 0..=5 { + for humans in 0..=8 { + for bots in 0..=8 { let size = task_size(humans, bots, 29); let cpu: u32 = size.cpu.parse().unwrap(); let memory: u32 = size.memory.parse().unwrap(); - let (_, _, memory_min, memory_max) = - TIERS.iter().find(|(units, ..)| *units == cpu).unwrap(); + let tier = TIERS.iter().find(|tier| tier.cpu == size.cpu).unwrap(); assert!( - memory >= *memory_min && memory <= *memory_max, + memory >= tier.memory_min && memory <= tier.memory_max, "humans={humans} bots={bots}: {cpu} cpu with {memory} MiB" ); + assert_eq!( + memory % tier.memory_step, + 0, + "humans={humans} bots={bots}: {memory} MiB is not on the \ + {} MiB step {cpu} cpu takes", + tier.memory_step + ); } } } + + /// Every tier's own range has to be expressible on its own step, or + /// `fit_memory`'s clamp could hand back a figure Fargate rejects. + #[test] + fn every_tier_range_sits_on_its_step() { + for tier in TIERS { + assert_eq!(tier.memory_min % tier.memory_step, 0, "{}", tier.cpu); + assert_eq!(tier.memory_max % tier.memory_step, 0, "{}", tier.cpu); + } + } } -- 2.51.2