diff --git a/src/index.ts b/src/index.ts index 519937f..76935f1 100644 --- a/src/index.ts +++ b/src/index.ts @@ -5,6 +5,7 @@ * aggregates; it holds no schedule and never talks to the archive itself. */ import { IngestParseError, parseIngestBatch, type SegmentRecord } from "./ingest"; +import page from "./page.html"; interface Env { readonly DB: D1Database; @@ -101,6 +102,11 @@ interface HeatRow { readonly count: number; } +interface RankRow { + readonly nsid: string; + readonly count: number; +} + interface BucketRow { readonly bucket: number; readonly first_idx: number; @@ -112,9 +118,13 @@ interface BucketRow { readonly event_count: number; } +const MAX_TOP = 40; + async function heat(url: URL, env: Env): Promise { const step = Number(url.searchParams.get("step") ?? "10"); + const top = Number(url.searchParams.get("top") ?? "12"); if (!Number.isSafeInteger(step) || step < 1) return json({ error: "step must be a positive integer" }, 400); + if (!Number.isSafeInteger(top) || top < 1 || top > MAX_TOP) return json({ error: `top must be 1..${MAX_TOP}` }, 400); const progressRow = await env.DB.prepare("SELECT MAX(idx) AS max_idx FROM segments").first<{ max_idx: number | null }>(); const maxIdx = progressRow?.max_idx ?? -1; if (Math.floor(maxIdx / step) + 1 > MAX_HEAT_BUCKETS) { @@ -122,22 +132,37 @@ async function heat(url: URL, env: Env): Promise { } const results = await env.DB.batch([ env.DB.prepare( - `SELECT idx / ?1 AS bucket, MIN(idx) AS first_idx, MAX(idx) AS last_idx, MIN(min_seq) AS min_seq, + `SELECT idx / CAST(?1 AS INTEGER) AS bucket, MIN(idx) AS first_idx, MAX(idx) AS last_idx, MIN(min_seq) AS min_seq, MAX(max_seq) AS max_seq, MIN(min_witnessed_at) AS min_witnessed_at, MAX(max_witnessed_at) AS max_witnessed_at, SUM(event_count) AS event_count FROM segments GROUP BY bucket ORDER BY bucket`, ).bind(step), env.DB.prepare( - `SELECT idx / ?1 AS bucket, nsid, SUM(count) AS count - FROM segment_collections GROUP BY bucket, nsid ORDER BY bucket, nsid`, - ).bind(step), + `WITH ranked AS ( + SELECT nsid FROM segment_collections GROUP BY nsid ORDER BY SUM(count) DESC LIMIT CAST(?2 AS INTEGER) + ) + SELECT idx / CAST(?1 AS INTEGER) AS bucket, + CASE WHEN nsid IN (SELECT nsid FROM ranked) THEN nsid ELSE 'other' END AS nsid, + SUM(count) AS count + FROM segment_collections GROUP BY bucket, 2 ORDER BY bucket, 2`, + ).bind(step, top), + env.DB.prepare( + "SELECT nsid, SUM(count) AS count FROM segment_collections GROUP BY nsid ORDER BY count DESC LIMIT CAST(?1 AS INTEGER)", + ).bind(top), ]); - const [buckets, cells] = results; - if (buckets === undefined || cells === undefined) return json({ error: "batch returned fewer results than statements" }, 500); - // SAFETY: the two statements above select exactly the columns of BucketRow and HeatRow. + const [buckets, cells, ranked] = results; + if (buckets === undefined || cells === undefined || ranked === undefined) { + return json({ error: "batch returned fewer results than statements" }, 500); + } + // SAFETY: the statements above select exactly the columns of BucketRow, HeatRow, and RankRow. const bucketRows = buckets.results as BucketRow[]; const heatRows = cells.results as HeatRow[]; - return json({ step, buckets: bucketRows, cells: heatRows }, 200, { "cache-control": "public, max-age=300" }); + const rankRows = ranked.results as RankRow[]; + return json( + { step, top, buckets: bucketRows, cells: heatRows, collections: rankRows }, + 200, + { "cache-control": "public, max-age=300" }, + ); } async function collections(env: Env): Promise { @@ -160,6 +185,8 @@ export default { case "/api/collections": return collections(env); case "/": + return new Response(page, { headers: { "content-type": "text/html; charset=utf-8", "cache-control": "public, max-age=300" } }); + case "/api": return json({ name: "strata", what: "per-segment, per-collection event counts of the stream.waow.tech archive", diff --git a/src/page.html b/src/page.html new file mode 100644 index 0000000..a8e7d95 --- /dev/null +++ b/src/page.html @@ -0,0 +1,342 @@ + + + + + +strata + + + + + +
+

strata

+

the shape of the stream.waow.tech archive, by lexicon — read from the sealed segments, not the process

+
+ +
loading…
+ +
+ + + + +
+ +
+
+ + +
+
fewermore
+
+ x is archive position (segment index; each segment ≈ 3.2M events, ~270 MB). the dates under it are when + stream witnessed the events, not when they happened: the bootstrap era replayed years of history + inside a six-day window from 2026-07-28, so those columns are dense; segments after it are live-era, about + two minutes each. counts come from each segment's collection index; account/identity/sync markers carry no + collection and are excluded. "other" folds every collection outside the top rows. +
+
+ + + +
+ fed hourly by a prefect flow that range-reads segment footers · + api · + source +
+ + + + diff --git a/src/text-modules.d.ts b/src/text-modules.d.ts new file mode 100644 index 0000000..9b4de38 --- /dev/null +++ b/src/text-modules.d.ts @@ -0,0 +1,4 @@ +declare module "*.html" { + const text: string; + export default text; +} diff --git a/wrangler.jsonc b/wrangler.jsonc index eb82fa1..6d8d04a 100644 --- a/wrangler.jsonc +++ b/wrangler.jsonc @@ -2,6 +2,7 @@ "$schema": "node_modules/wrangler/config-schema.json", "name": "strata", "main": "src/index.ts", + "rules": [{ "type": "Text", "globs": ["**/*.html"], "fallthrough": true }], "compatibility_date": "2026-08-01", "observability": { "enabled": true }, "workers_dev": false,