Something went wrong. Try again.
Central platform for European atproto.<cc> country community websites atproto.eu
community atproto
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116// Aggregated "Dutch accounts on AT Protocol" figures for the /stats page.//// Two tiers (see the plan in the workspace, 2026-08-27-nl-stats-aggregator):// - identified: a deduplicated union of DIDs across feeders. A hard floor that only grows.// - estimate: the modelled activity figure, mirrors nl-estimate.ts. Clearly a model.//// The aggregate is produced off-site (on infra we control) and committed here as JSON, same// model as nl-estimate.ts. The host never appears publicly and DID lists never leave it; only// these counts do. This module just types + light-checks the committed JSON at build time.
import raw from './nl-stats.json';
/** Feeders that CONFIRM an account is in the Netherlands, and so contribute DIDs to the * identified union: a .nl handle, a curated starterpack, a declared Sifa location, a company * domain. Each is evidence a person chose. */export type NlStatsSource = 'nlHandle' | 'starterpack' | 'sifaNl' | 'sifaCompany';
/** Signals that do NOT confirm a country and are never unioned into the floor. Dutch is spoken * in Belgium too, so "posted in Dutch" cannot place an account in the Netherlands. */export type NlStatsSignal = 'dutchPoster';
export interface NlStats { /** ISO timestamp of the aggregator run that produced this file. */ generatedAt: string; identified: { /** Unique DIDs across all feeders (the headline floor). Deduplicated, never summed. */ total: number; /** Non-exclusive per-source counts: one DID can appear in several sources. */ bySource: Record<NlStatsSource, number>; /** Per source: DIDs found ONLY by that source. Absent on aggregates written before v2. */ exclusiveBySource?: Record<NlStatsSource, number>; /** The per-source counts added up. Exceeds `total` by exactly the overlap. */ summedSources?: number; }; estimate: { /** Share of sampled Bluesky posts tagged Dutch. */ dutchPostSharePct: number; /** Total Bluesky accounts, the extrapolation anchor. */ blueskyActive: number; /** Length of the sample window, in minutes. */ windowMinutes: number; /** share * blueskyActive * NL_SHARE_OF_DUTCH, rounded. That last multiplier exists * because Dutch is spoken in Belgium too — the same reason `dutchPoster` is a signal * rather than a confirming source. */ modelledNl: number; /** Low end of the modelled range, from the observed day-to-day spread of the share. * Never below `identified.total` — no model may claim fewer than we can point at. * Absent until there are enough daily samples to derive a range honestly. */ modelledLo?: number; /** High end of the modelled range. Absent together with modelledLo. */ modelledHi?: number; /** How many daily share samples the range is based on. */ shareSamples?: number; shareMeanPct?: number; shareStdevPct?: number; }; /** Language signals: reported, never counted as confirmation. `alsoConfirmed` is how many * of them some confirming source independently places in the Netherlands. Absent on * aggregates written before the language/confirmation split. */ signals?: Record<NlStatsSignal, { count: number; alsoConfirmed: number; signalOnly: number }>; /** Provenance / freshness per feeder. `asOf` is null while a feeder is not yet live. */ sources: Record< NlStatsSource | NlStatsSignal, { count: number; asOf: string | null; window?: string; packs?: number } >;}
// Display order for the per-source breakdown.const SOURCES: NlStatsSource[] = ['nlHandle', 'starterpack', 'sifaNl', 'sifaCompany'];
// Guard the committed data: the union must be a real dedupe (>= any single source, since// every source is a subset of the union) and never exceed the summed counts. Fail the build// loudly if a bad aggregate lands, rather than render a wrong headline. Also normalize: a source// added ahead of its feeder (missing from the JSON) is treated as 0 so it renders instead of NaN.function assertValid(s: NlStats): NlStats { const bySource = Object.fromEntries( SOURCES.map((k) => [k, s.identified.bySource[k] ?? 0]), ) as Record<NlStatsSource, number>; s = { ...s, identified: { ...s.identified, bySource } }; const counts = SOURCES.map((k) => bySource[k]); const max = Math.max(0, ...counts); const sum = counts.reduce((a, b) => a + b, 0); if (s.identified.total < max) { throw new Error(`nl-stats: total (${s.identified.total}) below largest source (${max})`); } if (s.identified.total > sum) { throw new Error(`nl-stats: total (${s.identified.total}) above summed sources (${sum})`); } return s;}
export const nlStats: NlStats = assertValid(raw as NlStats);
/** All defined sources, ordered by count (largest first). Shown even at 0 (e.g. a feeder not * yet live), which sort to the end. *//** The overlap between sources: how many DIDs would be double-counted by summing. * Small by design — the detectors find largely different people. */const summedSources = nlStats.identified.summedSources ?? SOURCES.reduce((a, k) => a + nlStats.identified.bySource[k], 0);
export { summedSources };
export const overlapCount = summedSources > 0 ? summedSources - nlStats.identified.total : null;
/** Whether the aggregate carries an honest modelled range. */export const hasRange = typeof nlStats.estimate.modelledLo === 'number' && typeof nlStats.estimate.modelledHi === 'number';
export const activeSources = [...SOURCES].sort( (a, b) => nlStats.identified.bySource[b] - nlStats.identified.bySource[a],);
/** The Dutch-posting signal, or null on pre-split aggregates. Never added to the floor. */export const languageSignal = nlStats.signals?.dutchPoster ?? null;