#!/usr/bin/env python3 """Aggregate the per-country feeders into europe-stats.json, the dataset atproto.eu renders. The Dutch aggregator answers "how many accounts are in the Netherlands". This one answers "how many are in each European country", which is a different problem: the same DID can match several countries at once (a .es handle, a Sifa location in Germany), so per-country counts would sum to more than the number of people. So every DID is attributed to exactly ONE country, by the precedence in scripts/attribution.py: declarations beat inferences, deliberate choices beat incidental ones, and a signal that answers "where are you" beats one that answers "where do you take part". Disagreements at the same rank are recorded as contested rather than broken by a coin flip. Language is not an input here at all. It confirms nothing about a country, which is the whole reason the Dutch floor was rebuilt (see nl-stats-aggregate.py). Inputs, per country cc, in --dir: handles..dids ccTLD handles from the PLC export scan -> strongTld / weakTld sifa-.dids declared Sifa location -> sifaLocation sifa-company-.dids company website ccTLD -> strongTld (an org's market) verifier..dids country-community verifier list members -> verifier starterpack.dids curated pack members (NL only today) -> starterpack Output is COUNTS ONLY. DID lists never leave the host that generates them. Usage: python3 scripts/eu-stats-aggregate.py --dir . --out europe-stats.json python3 scripts/eu-stats-aggregate.py --self-test """ from __future__ import annotations import argparse import datetime import glob import json import os import re import sys sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from attribution import Signal, attribute_all, summarise # noqa: E402 # TLDs whose registrations are commonly English word-play rather than a statement about a # country, so a handle in them is deliberate but ambiguous. Kept, ranked lower. See the # attribution decision: .es is currently the LARGEST bucket in the PLC scan, ahead of .gb and # .de, which is not plausible as a measure of Spanish atproto adoption. WEAK_TLD_COUNTRIES = {"is", "es", "at", "se", "it"} DEFAULT_MAP = os.path.join(os.path.dirname(os.path.abspath(__file__)), "country-tlds.json") def read_dids(path): if not path or not os.path.exists(path): return set() with open(path) as f: return {ln.strip() for ln in f if ln.strip().startswith("did:")} def collect_signals(directory, countries, starterpack_country="nl", abandoned=None, active_dids=None): """Build did -> [Signal] across every country's feeders. `active_dids` (optional): when given, a weak-ccTLD handle (.es/.at/.it/.is/.se — word-play-prone) only counts as a signal if the account is in it, i.e. authored something recently. A dead th.is/foo.es is English word-play, not a person in that country; the data shows these are 96-99% of those countries' totals and 10-33% active vs 73-86% for confirmed. A LIVE weak-ccTLD account still counts (rank 6, never "confirmed"). Strong ccTLDs (.nl/.de/.fr) are never gated. Off by default, so behaviour is unchanged until the activity scan has run. NOTE: activity is aliveness, not nationality -- an active .es account that is not Spanish (e.g. a Hungarian on a .es handle) still passes; catching those is the Jev-QA layer, not this gate. """ abandoned = abandoned or set() signals: dict[str, list[Signal]] = {} def add(did, kind, cc): if did in abandoned: return signals.setdefault(did, []).append(Signal(kind, cc)) for cc in countries: weak = cc in WEAK_TLD_COUNTRIES for did in read_dids(os.path.join(directory, f"handles.{cc}.dids")): if weak and active_dids is not None and did not in active_dids: continue # dead word-play: a weak ccTLD only counts when the account is alive add(did, "weakTld" if weak else "strongTld", cc) for did in read_dids(os.path.join(directory, f"sifa-{cc}.dids")): add(did, "sifaLocation", cc) # A company's website ccTLD says which market it addresses -- an operational tie, unlike # a registered HQ, which is a tax choice and sits far lower in the ladder. for did in read_dids(os.path.join(directory, f"sifa-company-{cc}.dids")): add(did, "companyDomain", cc) # A country community's verifier (Eurosky "Verified by -atmosphe.re", produced by # eu-verifier-lists.py) vouched for the account -- a deliberate, vetted community signal. for did in read_dids(os.path.join(directory, f"verifier.{cc}.dids")): add(did, "verifier", cc) for did in read_dids(os.path.join(directory, "starterpack.dids")): add(did, "starterpack", starterpack_country) return signals def countries_in(directory, tld_map_path=DEFAULT_MAP): """Countries we have any feeder file for, plus every country in the TLD map.""" with open(tld_map_path) as f: known = sorted(set(json.load(f)["tlds"].values())) seen = set(known) for p in glob.glob(os.path.join(directory, "handles.*.dids")): m = re.search(r"handles\.([a-z]{2})\.dids$", p) if m: seen.add(m.group(1)) return sorted(seen) def read_activity(path, window_days=90, as_of=None): """(active, recorded) DID sets from an eu-activity.jsonl recency scan, or (None, None) if absent. active = DIDs whose lastActive is within window_days of as_of (today by default). recorded = DIDs with any activity row (for coverage). Both are the WHOLE scan, independent of attribution, because the weak-ccTLD gate in collect_signals needs the active set BEFORE attribution runs. Dates are ISO YYYY-MM-DD, compared as strings. """ if not path or not os.path.exists(path): return None, None ref = datetime.date.fromisoformat(as_of) if as_of else datetime.date.today() cutoff = (ref - datetime.timedelta(days=window_days)).isoformat() active, recorded = set(), set() with open(path) as f: for line in f: line = line.strip() if not line: continue try: row = json.loads(line) except ValueError: continue did = row.get("did") if not did: continue recorded.add(did) la = row.get("lastActive") if la and la >= cutoff: active.add(did) return active, recorded def build(directory, tld_map_path=DEFAULT_MAP, abandoned=None, emit_dids_to=None, emit_placed_to=None, activity_path=None, active_window_days=90, as_of=None): countries = countries_in(directory, tld_map_path) # Activity first: the weak-ccTLD gate drops dead word-play before attribution even sees it. active_all, recorded_all = read_activity(activity_path, active_window_days, as_of) signals = collect_signals(directory, countries, abandoned=abandoned, active_dids=active_all) attributions = attribute_all(signals) attributed_dids = {did for did, a in attributions.items() if a.country and not a.contested} if active_all is not None: active_dids = active_all & attributed_dids seen = recorded_all & attributed_dids coverage = len(seen) / len(attributed_dids) if attributed_dids else 0.0 else: active_dids, coverage = None, None summary = summarise(attributions, active_dids=active_dids) # One attribution pass, many consumers. Writing attributed..dids lets the national # aggregators read their own row instead of re-deriving it, so the country total on # atproto.nl and the NL row on atproto.eu agree by construction rather than by luck. if emit_dids_to: by_country: dict[str, list[str]] = {} for did, a in attributions.items(): if a.country: by_country.setdefault(a.country, []).append(did) for cc, dids in by_country.items(): path = os.path.join(emit_dids_to, f"attributed.{cc}.dids") tmp = path + ".tmp" with open(tmp, "w") as f: f.write("\n".join(sorted(dids)) + ("\n" if dids else "")) os.replace(tmp, path) # The DIDs behind the published "confirmed, placed in a country" figure, in one flat file. # eu-infra-pds.py intersects it with the accounts on a European PDS, so the site can say how # many European-infrastructure accounts are ALREADY counted under a country and how many are # additional. Without that intersection the two headline numbers cannot be added at all. # # Deliberately NOT named attributed..dids: nl-stats-aggregate.py auto-adopts a file by # that name from the same directory, which would silently move the Dutch floor. This file is # read by eu-infra-pds.py and nothing else. It stays on the private host, like every .dids. if emit_placed_to: placed = sorted(did for did, a in attributions.items() if a.confirmed) tmp = emit_placed_to + ".tmp" with open(tmp, "w") as f: f.write("\n".join(placed) + ("\n" if placed else "")) os.replace(tmp, emit_placed_to) track_active = active_dids is not None per_country = [] for cc, c in summary["countries"].items(): row = { "country": cc, "total": c["total"], "confirmed": c["confirmed"], "byRank": c["byRank"], "bySignal": c["byKind"], } if track_active: row["active"] = c["active"] per_country.append(row) result = { "generatedAt": datetime.datetime.now(datetime.timezone.utc) .strftime("%Y-%m-%dT%H:%M:%SZ"), # People, not rows: every DID counts towards exactly one country, so this is a real # total rather than a sum of overlapping per-country numbers. "people": summary["attributed"], "contested": summary["contested"], "countries": per_country, "method": { "attribution": "one DID, one country, by signal precedence", "languageUsed": False, "weakTldCountries": sorted(WEAK_TLD_COUNTRIES), }, } # identified (people) vs active (a floor on engagement). Both published; the caveat travels with # the number so it never reads as "total users". See the identified-vs-active decision. if track_active: result["activePeople"] = summary["activeTotal"] result["activeMethod"] = { "window": f"authored (post/reply/repost) in the last {active_window_days} days", "asOf": (as_of or datetime.date.today().isoformat()), "signal": "newest getAuthorFeed item", "scanCoverage": round(coverage or 0.0, 4), "note": "reading is invisible on atproto, so active is a floor on engagement -- the real " "community is larger. Counts only attributed accounts scanned so far " "(scanCoverage < 1 means the recency scan is still running).", } return result def _self_test(): import tempfile d = tempfile.mkdtemp() def w(name, dids): with open(os.path.join(d, name), "w") as f: f.write("\n".join(dids) + "\n") w("handles.nl.dids", ["did:plc:a", "did:plc:b"]) w("handles.es.dids", ["did:plc:b", "did:plc:c"]) # b: nl strong vs es weak -> nl wins w("sifa-de.dids", ["did:plc:a"]) # a: sifa location beats a .nl handle w("starterpack.dids", ["did:plc:d"]) w("verifier.be.dids", ["did:plc:e", "did:plc:c"]) # e: be by verifier; c: es weakTld beats verifier? no -> verifier(4) beats weakTld(6) out = build(d) per = {c["country"]: c for c in out["countries"]} assert per["de"]["total"] == 1, out # a -> de by declared location assert per["nl"]["total"] == 2, out # b (strongTld beats weak), d (starterpack) # c: es weakTld (rank 6) vs be verifier (rank 4) -> verifier wins, so c is be, not es. assert per["be"]["total"] == 2, out # e (verifier), c (verifier beats weak .es) assert "es" not in per, out # c moved to be; es had only c assert out["people"] == 5, out # a,b,c,d,e -- no DID counted twice assert out["contested"] == 0, out assert per["be"]["bySignal"].get("verifier") == 2, per["be"] assert sum(c["total"] for c in out["countries"]) == out["people"], out assert out["method"]["languageUsed"] is False # d came from a starterpack: rank 5, so counted but not "confirmed" at rank <= 5? It is 5. assert per["nl"]["confirmed"] >= 1, per["nl"] # --emit-placed writes exactly the DIDs behind the published confirmed total, so # eu-infra-pds.py can subtract the overlap instead of the site double-counting it. placed_path = os.path.join(d, "eu-placed.dids") build(d, emit_placed_to=placed_path) placed = read_dids(placed_path) assert len(placed) == sum(c["confirmed"] for c in out["countries"]), placed assert "did:plc:a" in placed, placed # sifaLocation, rank 1 # weak-ccTLD active-gate: with activity data a DORMANT .es handle is dropped, a LIVE one stays, # and a strong .nl handle is never gated. Without activity the gate is off and both .es count. d2 = tempfile.mkdtemp() w2 = lambda name, dids: open(os.path.join(d2, name), "w").write("\n".join(dids) + "\n") w2("handles.es.dids", ["did:plc:live", "did:plc:dead"]) w2("handles.nl.dids", ["did:plc:nlx"]) today = datetime.date.today().isoformat() with open(os.path.join(d2, "act.jsonl"), "w") as f: f.write(json.dumps({"did": "did:plc:live", "lastActive": today}) + "\n") f.write(json.dumps({"did": "did:plc:dead", "lastActive": "2020-01-01"}) + "\n") gated = {c["country"]: c for c in build(d2, activity_path=os.path.join(d2, "act.jsonl"))["countries"]} assert gated["es"]["total"] == 1 and gated["es"]["active"] == 1, gated # live kept, dead dropped assert gated["nl"]["total"] == 1, gated # strong .nl never gated ungated = {c["country"]: c for c in build(d2)["countries"]} assert ungated["es"]["total"] == 2, ungated # gate off -> both count print("eu-stats self-test OK") def main(): ap = argparse.ArgumentParser() ap.add_argument("--dir", default=".") ap.add_argument("--out", default="europe-stats.json") ap.add_argument("--tld-map", default=DEFAULT_MAP) ap.add_argument("--abandoned", default=None, help="optional .dids of accounts to exclude") ap.add_argument("--emit-dids", default=None, help="directory to write attributed..dids into (stays private)") ap.add_argument("--emit-placed", default=None, help="path for the flat .dids of confirmed, country-placed accounts, for " "eu-infra-pds.py to de-duplicate against (stays private)") ap.add_argument("--activity", default=None, help="optional eu-activity.jsonl (eu-activity-scan.py) to add per-country " "active- counts alongside the identified totals") ap.add_argument("--active-window", type=int, default=90, help="days: an account is active if it authored within this window (default 90)") ap.add_argument("--as-of", default=None, help="reference date YYYY-MM-DD (default: today)") ap.add_argument("--self-test", action="store_true") args = ap.parse_args() if args.self_test: _self_test() return # Honor the shared exclusion list by default (same file nl-stats-aggregate defaults to), so the # ONE cross-country attribution pass here is the single place abandoned/misattributed accounts # are dropped -- and every site reading europe-stats.json + attributed..dids gets the same # excluded set. INTERIM: the deeper single-source refactor (nl-stats reads this output instead # of re-deriving its own floor) is decisions/2026-09-24-single-source-attribution.md. # Two exclusion sources, unioned: abandoned.dids (collector-owned dead repos -- nl-lexicons.py # OVERWRITES it each run) and struck.dids (manual/Jev-QA country-misattribution strikes -- never # auto-written, so they must live in their own file or the collector's overwrite would wipe them). if args.abandoned: abandoned = read_dids(args.abandoned) else: abandoned = read_dids(os.path.join(args.dir, "abandoned.dids")) | \ read_dids(os.path.join(args.dir, "struck.dids")) out = build(args.dir, args.tld_map, abandoned, emit_dids_to=args.emit_dids, emit_placed_to=args.emit_placed, activity_path=args.activity, active_window_days=args.active_window, as_of=args.as_of) with open(args.out, "w") as f: f.write(json.dumps(out, indent=2) + "\n") top = ", ".join(f"{c['country']}={c['total']}" for c in out["countries"][:8]) print(f"Wrote {args.out}: {out['people']} people across {len(out['countries'])} countries " f"({out['contested']} contested); {top}") if __name__ == "__main__": main()