diff --git a/docs/incident-2026-05-30-arena-bloat.md b/docs/incident-2026-05-30-arena-bloat.md new file mode 100644 index 0000000..65cbe9f --- /dev/null +++ b/docs/incident-2026-05-30-arena-bloat.md @@ -0,0 +1,65 @@ +# memory floor after 18:40Z burst — engineer response 2026-05-30 + +**Punchline:** the post-burst RSS floor is glibc per-thread-arena fragmentation, not a leak. +The scavenger-handoff doc's mechanism is right in spirit but wrong on two facts: zlay uses +**glibc `c_allocator`, not `SmpAllocator`**, and we **already call `malloc_trim(0)`** every +10 min. So the fix isn't a custom `madvise` scavenger — it's (1) see all arenas, (2) cap them. + +## what was actually wrong with the diagnosis + +- **allocator**: `src/main.zig:167` and `broadcaster.zig:161` use `std.heap.c_allocator` + (glibc malloc). The `relay_malloc_*` metrics are literally glibc `mallinfo` fields. +- **trim already runs**: `gcLoop` (`main.zig:477`) calls `malloc_trim(0)` every 10 min and it + is *not* recovering this memory — because `malloc_trim` only returns the contiguous free + *top* of each arena. Free chunks stranded below a live chunk stay put. Classic sticky + fragmentation, worst-cased by a burst interleaving short/long-lived allocs across thousands + of per-thread arenas. +- **the dashboard was half-blind**: `mallinfo()` reports the **main arena only** + (`broadcaster.zig:1243`). With ~2670 worker threads glibc spreads allocations across many + per-thread arenas, so the `relay_malloc_free_bytes` / `relay_malloc_arena_bytes` numbers in + the handoff table understate the true heap. The 2.4 GiB delta likely lives in arenas the old + metrics never counted. + +## change shipped (code) + +Added all-arena gauges via `malloc_info()` (covers every arena, not just main): + +- `relay_malloc_system_bytes` — total bytes from the OS across all arenas +- `relay_malloc_free_total_bytes` — free bytes held in all arenas (`fast`+`rest`), not returned +- `relay_malloc_arena_count` — live arena count (the knob `MALLOC_ARENA_MAX` controls) + +Validated end-to-end: parser unit-tested against sample + **real glibc `malloc_info` output** +in a debian container (16 threads → 4 arenas, top-level totals parsed exactly). Existing +`relay_malloc_*` (main-arena) gauges kept for dashboard continuity. + +Watch `relay_malloc_free_total_bytes` and `relay_malloc_arena_count` across the next burst — +that's the previously-invisible fragmentation. + +## operator action — reversible, no rebuild + +Set **`MALLOC_ARENA_MAX`** in the zlay deployment env. It's read by glibc at libc init; fewer +arenas = far less fragmentation surface (the single highest-leverage glibc RSS knob for +thread-heavy services). Start at `4`, watch `relay_malloc_arena_count` settle there and +`relay_malloc_free_total_bytes` after a burst; if throughput holds, try `2`. + +```yaml + env: + - name: MALLOC_ARENA_MAX + value: "4" +``` + +Trade-off: fewer arenas means more malloc-lock contention. The new `arena_count` / +`free_total` gauges + existing throughput metrics tell us if we went too tight — back off to a +higher value if `fr/s` dips. Fully reversible, no binary change. + +## if glibc tuning isn't enough + +Swap to **jemalloc** or **mimalloc** (`LD_PRELOAD` or link) — both ship background purger +threads, the real analog to Go's scavenger the handoff wanted. Bigger change; only if +`ARENA_MAX` + watching the new metrics shows the floor still ratchets. + +## still open + +What *caused* the 5.7 GiB transient burst — no slow-consumer / queue-full / backpressure / +decode-error detector fired. Separate puzzle, likely needs per-consumer queue-depth +attribution. The arena cap shrinks the *aftermath*; it doesn't explain the *spike*. diff --git a/src/broadcaster.zig b/src/broadcaster.zig index 01030c5..a81ed54 100644 --- a/src/broadcaster.zig +++ b/src/broadcaster.zig @@ -1261,6 +1261,27 @@ fn appendProcMetrics(w: *Io.Writer, io: Io) void { \\relay_malloc_mmap_bytes {d} \\ , .{ arena, in_use, free_bytes, mmap_bytes }) catch {}; + + // all-arena heap picture via malloc_info() (mallinfo above is main-arena only). + // with ~2670 worker threads glibc spreads allocations across many per-thread + // arenas; post-burst fragmentation lives there, invisible to mallinfo. these + // gauges sum every arena so a rising free_total / arena_count is observable. + if (readMallocInfo()) |info| { + w.print( + \\# TYPE relay_malloc_system_bytes gauge + \\# HELP relay_malloc_system_bytes total bytes obtained from the OS across all arenas + \\relay_malloc_system_bytes {d} + \\ + \\# TYPE relay_malloc_free_total_bytes gauge + \\# HELP relay_malloc_free_total_bytes free bytes held in all arenas (fast+rest), not returned to OS + \\relay_malloc_free_total_bytes {d} + \\ + \\# TYPE relay_malloc_arena_count gauge + \\# HELP relay_malloc_arena_count number of glibc malloc arenas (tune via MALLOC_ARENA_MAX) + \\relay_malloc_arena_count {d} + \\ + , .{ info.system_bytes, info.free_bytes, info.arena_count }) catch {}; + } } // glibc mallinfo struct — fields are c_int (32-bit), we bitcast to u32 for range. @@ -1279,6 +1300,102 @@ const MallInfo = extern struct { const mallinfo = @extern(*const fn () callconv(.c) MallInfo, .{ .name = "mallinfo" }); +// all-arena heap stats, parsed from glibc malloc_info() XML. unlike mallinfo() +// these cover every per-thread arena, not just the main one. +const MallocInfo = struct { + system_bytes: u64, + free_bytes: u64, + arena_count: u32, +}; + +const CFile = opaque {}; +const open_memstream = @extern(*const fn (ptr: *[*c]u8, size: *usize) callconv(.c) ?*CFile, .{ .name = "open_memstream" }); +const fclose = @extern(*const fn (*CFile) callconv(.c) c_int, .{ .name = "fclose" }); +const c_free = @extern(*const fn (?*anyopaque) callconv(.c) void, .{ .name = "free" }); +const malloc_info_fn = @extern(*const fn (options: c_int, stream: *CFile) callconv(.c) c_int, .{ .name = "malloc_info" }); + +fn readMallocInfo() ?MallocInfo { + // comptime if prunes the linux-only externs from analysis on other hosts, + // so `zig build test` on macOS doesn't try to link malloc_info/open_memstream. + return if (comptime builtin.os.tag == .linux) readMallocInfoLinux() else null; +} + +fn readMallocInfoLinux() ?MallocInfo { + var buf_ptr: [*c]u8 = null; + var buf_size: usize = 0; + const stream = open_memstream(&buf_ptr, &buf_size) orelse return null; + _ = malloc_info_fn(0, stream); + _ = fclose(stream); // flushes and populates buf_ptr/buf_size + const p = buf_ptr orelse return null; + defer c_free(p); + const xml = p[0..buf_size]; + return MallocInfo{ + // top-level totals are the LAST occurrence (each emits its own). + .system_bytes = lastSizeAttr(xml, "system type=\"current\"") orelse 0, + .free_bytes = (lastSizeAttr(xml, "total type=\"fast\"") orelse 0) + + (lastSizeAttr(xml, "total type=\"rest\"") orelse 0), + .arena_count = countOccurrences(xml, " 0); + try std.testing.expect(info.arena_count >= 1); +} + +test "lastSizeAttr / countOccurrences against sample XML" { + const xml = + \\ + \\ + \\ + \\ + \\ + \\ + \\ + \\ + \\ + \\ + \\ + \\ + \\ + \\ + \\ + ; + try std.testing.expectEqual(@as(?u64, 18887), lastSizeAttr(xml, "system type=\"current\"")); + try std.testing.expectEqual(@as(?u64, 211), lastSizeAttr(xml, "total type=\"fast\"")); + try std.testing.expectEqual(@as(?u64, 422), lastSizeAttr(xml, "total type=\"rest\"")); + try std.testing.expectEqual(@as(u32, 2), countOccurrences(xml, "