Something went wrong. Try again.
Central platform for European atproto.<cc> country community websites atproto.eu
community atproto
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341#!/usr/bin/env python3"""Aggregate the per-country feeders into europe-stats.json, the dataset atproto.eu renders.
The Dutch aggregator answers "how many accounts are in the Netherlands". This one answers"how many are in each European country", which is a different problem: the same DID can matchseveral countries at once (a .es handle, a Sifa location in Germany), so per-country countswould sum to more than the number of people.
So every DID is attributed to exactly ONE country, by the precedence in scripts/attribution.py:declarations beat inferences, deliberate choices beat incidental ones, and a signal that answers"where are you" beats one that answers "where do you take part". Disagreements at the same rankare recorded as contested rather than broken by a coin flip.
Language is not an input here at all. It confirms nothing about a country, which is the wholereason the Dutch floor was rebuilt (see nl-stats-aggregate.py).
Inputs, per country cc, in --dir: handles.<cc>.dids ccTLD handles from the PLC export scan -> strongTld / weakTld sifa-<cc>.dids declared Sifa location -> sifaLocation sifa-company-<cc>.dids company website ccTLD -> strongTld (an org's market) verifier.<cc>.dids country-community verifier list members -> verifier starterpack.dids curated pack members (NL only today) -> starterpack
Output is COUNTS ONLY. DID lists never leave the host that generates them.
Usage: python3 scripts/eu-stats-aggregate.py --dir . --out europe-stats.json python3 scripts/eu-stats-aggregate.py --self-test"""from __future__ import annotations
import argparseimport datetimeimport globimport jsonimport osimport reimport sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))from attribution import Signal, attribute_all, summarise # noqa: E402
# TLDs whose registrations are commonly English word-play rather than a statement about a# country, so a handle in them is deliberate but ambiguous. Kept, ranked lower. See the# attribution decision: .es is currently the LARGEST bucket in the PLC scan, ahead of .gb and# .de, which is not plausible as a measure of Spanish atproto adoption.WEAK_TLD_COUNTRIES = {"is", "es", "at", "se", "it"}
DEFAULT_MAP = os.path.join(os.path.dirname(os.path.abspath(__file__)), "country-tlds.json")
def read_dids(path): if not path or not os.path.exists(path): return set() with open(path) as f: return {ln.strip() for ln in f if ln.strip().startswith("did:")}
def collect_signals(directory, countries, starterpack_country="nl", abandoned=None, active_dids=None): """Build did -> [Signal] across every country's feeders.
`active_dids` (optional): when given, a weak-ccTLD handle (.es/.at/.it/.is/.se — word-play-prone) only counts as a signal if the account is in it, i.e. authored something recently. A dead th.is/foo.es is English word-play, not a person in that country; the data shows these are 96-99% of those countries' totals and 10-33% active vs 73-86% for confirmed. A LIVE weak-ccTLD account still counts (rank 6, never "confirmed"). Strong ccTLDs (.nl/.de/.fr) are never gated. Off by default, so behaviour is unchanged until the activity scan has run. NOTE: activity is aliveness, not nationality -- an active .es account that is not Spanish (e.g. a Hungarian on a .es handle) still passes; catching those is the Jev-QA layer, not this gate. """ abandoned = abandoned or set() signals: dict[str, list[Signal]] = {}
def add(did, kind, cc): if did in abandoned: return signals.setdefault(did, []).append(Signal(kind, cc))
for cc in countries: weak = cc in WEAK_TLD_COUNTRIES for did in read_dids(os.path.join(directory, f"handles.{cc}.dids")): if weak and active_dids is not None and did not in active_dids: continue # dead word-play: a weak ccTLD only counts when the account is alive add(did, "weakTld" if weak else "strongTld", cc) for did in read_dids(os.path.join(directory, f"sifa-{cc}.dids")): add(did, "sifaLocation", cc) # A company's website ccTLD says which market it addresses -- an operational tie, unlike # a registered HQ, which is a tax choice and sits far lower in the ladder. for did in read_dids(os.path.join(directory, f"sifa-company-{cc}.dids")): add(did, "companyDomain", cc) # A country community's verifier (Eurosky "Verified by <cc>-atmosphe.re", produced by # eu-verifier-lists.py) vouched for the account -- a deliberate, vetted community signal. for did in read_dids(os.path.join(directory, f"verifier.{cc}.dids")): add(did, "verifier", cc)
for did in read_dids(os.path.join(directory, "starterpack.dids")): add(did, "starterpack", starterpack_country) return signals
def countries_in(directory, tld_map_path=DEFAULT_MAP): """Countries we have any feeder file for, plus every country in the TLD map.""" with open(tld_map_path) as f: known = sorted(set(json.load(f)["tlds"].values())) seen = set(known) for p in glob.glob(os.path.join(directory, "handles.*.dids")): m = re.search(r"handles\.([a-z]{2})\.dids$", p) if m: seen.add(m.group(1)) return sorted(seen)
def read_activity(path, window_days=90, as_of=None): """(active, recorded) DID sets from an eu-activity.jsonl recency scan, or (None, None) if absent.
active = DIDs whose lastActive is within window_days of as_of (today by default). recorded = DIDs with any activity row (for coverage). Both are the WHOLE scan, independent of attribution, because the weak-ccTLD gate in collect_signals needs the active set BEFORE attribution runs. Dates are ISO YYYY-MM-DD, compared as strings. """ if not path or not os.path.exists(path): return None, None ref = datetime.date.fromisoformat(as_of) if as_of else datetime.date.today() cutoff = (ref - datetime.timedelta(days=window_days)).isoformat() active, recorded = set(), set() with open(path) as f: for line in f: line = line.strip() if not line: continue try: row = json.loads(line) except ValueError: continue did = row.get("did") if not did: continue recorded.add(did) la = row.get("lastActive") if la and la >= cutoff: active.add(did) return active, recorded
def build(directory, tld_map_path=DEFAULT_MAP, abandoned=None, emit_dids_to=None, emit_placed_to=None, activity_path=None, active_window_days=90, as_of=None): countries = countries_in(directory, tld_map_path) # Activity first: the weak-ccTLD gate drops dead word-play before attribution even sees it. active_all, recorded_all = read_activity(activity_path, active_window_days, as_of) signals = collect_signals(directory, countries, abandoned=abandoned, active_dids=active_all) attributions = attribute_all(signals) attributed_dids = {did for did, a in attributions.items() if a.country and not a.contested} if active_all is not None: active_dids = active_all & attributed_dids seen = recorded_all & attributed_dids coverage = len(seen) / len(attributed_dids) if attributed_dids else 0.0 else: active_dids, coverage = None, None summary = summarise(attributions, active_dids=active_dids)
# One attribution pass, many consumers. Writing attributed.<cc>.dids lets the national # aggregators read their own row instead of re-deriving it, so the country total on # atproto.nl and the NL row on atproto.eu agree by construction rather than by luck. if emit_dids_to: by_country: dict[str, list[str]] = {} for did, a in attributions.items(): if a.country: by_country.setdefault(a.country, []).append(did) for cc, dids in by_country.items(): path = os.path.join(emit_dids_to, f"attributed.{cc}.dids") tmp = path + ".tmp" with open(tmp, "w") as f: f.write("\n".join(sorted(dids)) + ("\n" if dids else "")) os.replace(tmp, path)
# The DIDs behind the published "confirmed, placed in a country" figure, in one flat file. # eu-infra-pds.py intersects it with the accounts on a European PDS, so the site can say how # many European-infrastructure accounts are ALREADY counted under a country and how many are # additional. Without that intersection the two headline numbers cannot be added at all. # # Deliberately NOT named attributed.<cc>.dids: nl-stats-aggregate.py auto-adopts a file by # that name from the same directory, which would silently move the Dutch floor. This file is # read by eu-infra-pds.py and nothing else. It stays on the private host, like every .dids. if emit_placed_to: placed = sorted(did for did, a in attributions.items() if a.confirmed) tmp = emit_placed_to + ".tmp" with open(tmp, "w") as f: f.write("\n".join(placed) + ("\n" if placed else "")) os.replace(tmp, emit_placed_to)
track_active = active_dids is not None per_country = [] for cc, c in summary["countries"].items(): row = { "country": cc, "total": c["total"], "confirmed": c["confirmed"], "byRank": c["byRank"], "bySignal": c["byKind"], } if track_active: row["active"] = c["active"] per_country.append(row)
result = { "generatedAt": datetime.datetime.now(datetime.timezone.utc) .strftime("%Y-%m-%dT%H:%M:%SZ"), # People, not rows: every DID counts towards exactly one country, so this is a real # total rather than a sum of overlapping per-country numbers. "people": summary["attributed"], "contested": summary["contested"], "countries": per_country, "method": { "attribution": "one DID, one country, by signal precedence", "languageUsed": False, "weakTldCountries": sorted(WEAK_TLD_COUNTRIES), }, } # identified (people) vs active (a floor on engagement). Both published; the caveat travels with # the number so it never reads as "total users". See the identified-vs-active decision. if track_active: result["activePeople"] = summary["activeTotal"] result["activeMethod"] = { "window": f"authored (post/reply/repost) in the last {active_window_days} days", "asOf": (as_of or datetime.date.today().isoformat()), "signal": "newest getAuthorFeed item", "scanCoverage": round(coverage or 0.0, 4), "note": "reading is invisible on atproto, so active is a floor on engagement -- the real " "community is larger. Counts only attributed accounts scanned so far " "(scanCoverage < 1 means the recency scan is still running).", } return result
def _self_test(): import tempfile
d = tempfile.mkdtemp()
def w(name, dids): with open(os.path.join(d, name), "w") as f: f.write("\n".join(dids) + "\n")
w("handles.nl.dids", ["did:plc:a", "did:plc:b"]) w("handles.es.dids", ["did:plc:b", "did:plc:c"]) # b: nl strong vs es weak -> nl wins w("sifa-de.dids", ["did:plc:a"]) # a: sifa location beats a .nl handle w("starterpack.dids", ["did:plc:d"]) w("verifier.be.dids", ["did:plc:e", "did:plc:c"]) # e: be by verifier; c: es weakTld beats verifier? no -> verifier(4) beats weakTld(6)
out = build(d) per = {c["country"]: c for c in out["countries"]} assert per["de"]["total"] == 1, out # a -> de by declared location assert per["nl"]["total"] == 2, out # b (strongTld beats weak), d (starterpack) # c: es weakTld (rank 6) vs be verifier (rank 4) -> verifier wins, so c is be, not es. assert per["be"]["total"] == 2, out # e (verifier), c (verifier beats weak .es) assert "es" not in per, out # c moved to be; es had only c assert out["people"] == 5, out # a,b,c,d,e -- no DID counted twice assert out["contested"] == 0, out assert per["be"]["bySignal"].get("verifier") == 2, per["be"] assert sum(c["total"] for c in out["countries"]) == out["people"], out assert out["method"]["languageUsed"] is False # d came from a starterpack: rank 5, so counted but not "confirmed" at rank <= 5? It is 5. assert per["nl"]["confirmed"] >= 1, per["nl"]
# --emit-placed writes exactly the DIDs behind the published confirmed total, so # eu-infra-pds.py can subtract the overlap instead of the site double-counting it. placed_path = os.path.join(d, "eu-placed.dids") build(d, emit_placed_to=placed_path) placed = read_dids(placed_path) assert len(placed) == sum(c["confirmed"] for c in out["countries"]), placed assert "did:plc:a" in placed, placed # sifaLocation, rank 1
# weak-ccTLD active-gate: with activity data a DORMANT .es handle is dropped, a LIVE one stays, # and a strong .nl handle is never gated. Without activity the gate is off and both .es count. d2 = tempfile.mkdtemp() w2 = lambda name, dids: open(os.path.join(d2, name), "w").write("\n".join(dids) + "\n") w2("handles.es.dids", ["did:plc:live", "did:plc:dead"]) w2("handles.nl.dids", ["did:plc:nlx"]) today = datetime.date.today().isoformat() with open(os.path.join(d2, "act.jsonl"), "w") as f: f.write(json.dumps({"did": "did:plc:live", "lastActive": today}) + "\n") f.write(json.dumps({"did": "did:plc:dead", "lastActive": "2020-01-01"}) + "\n") gated = {c["country"]: c for c in build(d2, activity_path=os.path.join(d2, "act.jsonl"))["countries"]} assert gated["es"]["total"] == 1 and gated["es"]["active"] == 1, gated # live kept, dead dropped assert gated["nl"]["total"] == 1, gated # strong .nl never gated ungated = {c["country"]: c for c in build(d2)["countries"]} assert ungated["es"]["total"] == 2, ungated # gate off -> both count print("eu-stats self-test OK")
def main(): ap = argparse.ArgumentParser() ap.add_argument("--dir", default=".") ap.add_argument("--out", default="europe-stats.json") ap.add_argument("--tld-map", default=DEFAULT_MAP) ap.add_argument("--abandoned", default=None, help="optional .dids of accounts to exclude") ap.add_argument("--emit-dids", default=None, help="directory to write attributed.<cc>.dids into (stays private)") ap.add_argument("--emit-placed", default=None, help="path for the flat .dids of confirmed, country-placed accounts, for " "eu-infra-pds.py to de-duplicate against (stays private)") ap.add_argument("--activity", default=None, help="optional eu-activity.jsonl (eu-activity-scan.py) to add per-country " "active-<window> counts alongside the identified totals") ap.add_argument("--active-window", type=int, default=90, help="days: an account is active if it authored within this window (default 90)") ap.add_argument("--as-of", default=None, help="reference date YYYY-MM-DD (default: today)") ap.add_argument("--self-test", action="store_true") args = ap.parse_args()
if args.self_test: _self_test() return
# Honor the shared exclusion list by default (same file nl-stats-aggregate defaults to), so the # ONE cross-country attribution pass here is the single place abandoned/misattributed accounts # are dropped -- and every site reading europe-stats.json + attributed.<cc>.dids gets the same # excluded set. INTERIM: the deeper single-source refactor (nl-stats reads this output instead # of re-deriving its own floor) is decisions/2026-09-24-single-source-attribution.md. # Two exclusion sources, unioned: abandoned.dids (collector-owned dead repos -- nl-lexicons.py # OVERWRITES it each run) and struck.dids (manual/Jev-QA country-misattribution strikes -- never # auto-written, so they must live in their own file or the collector's overwrite would wipe them). if args.abandoned: abandoned = read_dids(args.abandoned) else: abandoned = read_dids(os.path.join(args.dir, "abandoned.dids")) | \ read_dids(os.path.join(args.dir, "struck.dids")) out = build(args.dir, args.tld_map, abandoned, emit_dids_to=args.emit_dids, emit_placed_to=args.emit_placed, activity_path=args.activity, active_window_days=args.active_window, as_of=args.as_of) with open(args.out, "w") as f: f.write(json.dumps(out, indent=2) + "\n") top = ", ".join(f"{c['country']}={c['total']}" for c in out["countries"][:8]) print(f"Wrote {args.out}: {out['people']} people across {len(out['countries'])} countries " f"({out['contested']} contested); {top}")
if __name__ == "__main__": main()