Something went wrong. Try again.
Central platform for European atproto.<cc> country community websites atproto.eu
community atproto
Something went wrong. Try again.
Python
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233#!/usr/bin/env python3"""Aggregate the per-country feeders into europe-stats.json, the dataset atproto.eu renders.
The Dutch aggregator answers "how many accounts are in the Netherlands". This one answers"how many are in each European country", which is a different problem: the same DID can matchseveral countries at once (a .es handle, a Sifa location in Germany), so per-country countswould sum to more than the number of people.
So every DID is attributed to exactly ONE country, by the precedence in scripts/attribution.py:declarations beat inferences, deliberate choices beat incidental ones, and a signal that answers"where are you" beats one that answers "where do you take part". Disagreements at the same rankare recorded as contested rather than broken by a coin flip.
Language is not an input here at all. It confirms nothing about a country, which is the wholereason the Dutch floor was rebuilt (see nl-stats-aggregate.py).
Inputs, per country cc, in --dir: handles.<cc>.dids ccTLD handles from the PLC export scan -> strongTld / weakTld sifa-<cc>.dids declared Sifa location -> sifaLocation sifa-company-<cc>.dids company website ccTLD -> strongTld (an org's market) verifier.<cc>.dids country-community verifier list members -> verifier starterpack.dids curated pack members (NL only today) -> starterpack
Output is COUNTS ONLY. DID lists never leave the host that generates them.
Usage: python3 scripts/eu-stats-aggregate.py --dir . --out europe-stats.json python3 scripts/eu-stats-aggregate.py --self-test"""from __future__ import annotations
import argparseimport datetimeimport globimport jsonimport osimport reimport sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))from attribution import Signal, attribute_all, summarise # noqa: E402
# TLDs whose registrations are commonly English word-play rather than a statement about a# country, so a handle in them is deliberate but ambiguous. Kept, ranked lower. See the# attribution decision: .es is currently the LARGEST bucket in the PLC scan, ahead of .gb and# .de, which is not plausible as a measure of Spanish atproto adoption.WEAK_TLD_COUNTRIES = {"is", "es", "at", "se", "it"}
DEFAULT_MAP = os.path.join(os.path.dirname(os.path.abspath(__file__)), "country-tlds.json")
def read_dids(path): if not path or not os.path.exists(path): return set() with open(path) as f: return {ln.strip() for ln in f if ln.strip().startswith("did:")}
def collect_signals(directory, countries, starterpack_country="nl", abandoned=None): """Build did -> [Signal] across every country's feeders.""" abandoned = abandoned or set() signals: dict[str, list[Signal]] = {}
def add(did, kind, cc): if did in abandoned: return signals.setdefault(did, []).append(Signal(kind, cc))
for cc in countries: for did in read_dids(os.path.join(directory, f"handles.{cc}.dids")): add(did, "weakTld" if cc in WEAK_TLD_COUNTRIES else "strongTld", cc) for did in read_dids(os.path.join(directory, f"sifa-{cc}.dids")): add(did, "sifaLocation", cc) # A company's website ccTLD says which market it addresses -- an operational tie, unlike # a registered HQ, which is a tax choice and sits far lower in the ladder. for did in read_dids(os.path.join(directory, f"sifa-company-{cc}.dids")): add(did, "companyDomain", cc) # A country community's verifier (Eurosky "Verified by <cc>-atmosphe.re", produced by # eu-verifier-lists.py) vouched for the account -- a deliberate, vetted community signal. for did in read_dids(os.path.join(directory, f"verifier.{cc}.dids")): add(did, "verifier", cc)
for did in read_dids(os.path.join(directory, "starterpack.dids")): add(did, "starterpack", starterpack_country) return signals
def countries_in(directory, tld_map_path=DEFAULT_MAP): """Countries we have any feeder file for, plus every country in the TLD map.""" with open(tld_map_path) as f: known = sorted(set(json.load(f)["tlds"].values())) seen = set(known) for p in glob.glob(os.path.join(directory, "handles.*.dids")): m = re.search(r"handles\.([a-z]{2})\.dids$", p) if m: seen.add(m.group(1)) return sorted(seen)
def build(directory, tld_map_path=DEFAULT_MAP, abandoned=None, emit_dids_to=None, emit_placed_to=None): countries = countries_in(directory, tld_map_path) signals = collect_signals(directory, countries, abandoned=abandoned) attributions = attribute_all(signals) summary = summarise(attributions)
# One attribution pass, many consumers. Writing attributed.<cc>.dids lets the national # aggregators read their own row instead of re-deriving it, so the country total on # atproto.nl and the NL row on atproto.eu agree by construction rather than by luck. if emit_dids_to: by_country: dict[str, list[str]] = {} for did, a in attributions.items(): if a.country: by_country.setdefault(a.country, []).append(did) for cc, dids in by_country.items(): path = os.path.join(emit_dids_to, f"attributed.{cc}.dids") tmp = path + ".tmp" with open(tmp, "w") as f: f.write("\n".join(sorted(dids)) + ("\n" if dids else "")) os.replace(tmp, path)
# The DIDs behind the published "confirmed, placed in a country" figure, in one flat file. # eu-infra-pds.py intersects it with the accounts on a European PDS, so the site can say how # many European-infrastructure accounts are ALREADY counted under a country and how many are # additional. Without that intersection the two headline numbers cannot be added at all. # # Deliberately NOT named attributed.<cc>.dids: nl-stats-aggregate.py auto-adopts a file by # that name from the same directory, which would silently move the Dutch floor. This file is # read by eu-infra-pds.py and nothing else. It stays on the private host, like every .dids. if emit_placed_to: placed = sorted(did for did, a in attributions.items() if a.confirmed) tmp = emit_placed_to + ".tmp" with open(tmp, "w") as f: f.write("\n".join(placed) + ("\n" if placed else "")) os.replace(tmp, emit_placed_to)
per_country = [] for cc, c in summary["countries"].items(): per_country.append({ "country": cc, "total": c["total"], "confirmed": c["confirmed"], "byRank": c["byRank"], "bySignal": c["byKind"], })
return { "generatedAt": datetime.datetime.now(datetime.timezone.utc) .strftime("%Y-%m-%dT%H:%M:%SZ"), # People, not rows: every DID counts towards exactly one country, so this is a real # total rather than a sum of overlapping per-country numbers. "people": summary["attributed"], "contested": summary["contested"], "countries": per_country, "method": { "attribution": "one DID, one country, by signal precedence", "languageUsed": False, "weakTldCountries": sorted(WEAK_TLD_COUNTRIES), }, }
def _self_test(): import tempfile
d = tempfile.mkdtemp()
def w(name, dids): with open(os.path.join(d, name), "w") as f: f.write("\n".join(dids) + "\n")
w("handles.nl.dids", ["did:plc:a", "did:plc:b"]) w("handles.es.dids", ["did:plc:b", "did:plc:c"]) # b: nl strong vs es weak -> nl wins w("sifa-de.dids", ["did:plc:a"]) # a: sifa location beats a .nl handle w("starterpack.dids", ["did:plc:d"]) w("verifier.be.dids", ["did:plc:e", "did:plc:c"]) # e: be by verifier; c: es weakTld beats verifier? no -> verifier(4) beats weakTld(6)
out = build(d) per = {c["country"]: c for c in out["countries"]} assert per["de"]["total"] == 1, out # a -> de by declared location assert per["nl"]["total"] == 2, out # b (strongTld beats weak), d (starterpack) # c: es weakTld (rank 6) vs be verifier (rank 4) -> verifier wins, so c is be, not es. assert per["be"]["total"] == 2, out # e (verifier), c (verifier beats weak .es) assert "es" not in per, out # c moved to be; es had only c assert out["people"] == 5, out # a,b,c,d,e -- no DID counted twice assert out["contested"] == 0, out assert per["be"]["bySignal"].get("verifier") == 2, per["be"] assert sum(c["total"] for c in out["countries"]) == out["people"], out assert out["method"]["languageUsed"] is False # d came from a starterpack: rank 5, so counted but not "confirmed" at rank <= 5? It is 5. assert per["nl"]["confirmed"] >= 1, per["nl"]
# --emit-placed writes exactly the DIDs behind the published confirmed total, so # eu-infra-pds.py can subtract the overlap instead of the site double-counting it. placed_path = os.path.join(d, "eu-placed.dids") build(d, emit_placed_to=placed_path) placed = read_dids(placed_path) assert len(placed) == sum(c["confirmed"] for c in out["countries"]), placed assert "did:plc:a" in placed, placed # sifaLocation, rank 1 print("eu-stats self-test OK")
def main(): ap = argparse.ArgumentParser() ap.add_argument("--dir", default=".") ap.add_argument("--out", default="europe-stats.json") ap.add_argument("--tld-map", default=DEFAULT_MAP) ap.add_argument("--abandoned", default=None, help="optional .dids of accounts to exclude") ap.add_argument("--emit-dids", default=None, help="directory to write attributed.<cc>.dids into (stays private)") ap.add_argument("--emit-placed", default=None, help="path for the flat .dids of confirmed, country-placed accounts, for " "eu-infra-pds.py to de-duplicate against (stays private)") ap.add_argument("--self-test", action="store_true") args = ap.parse_args()
if args.self_test: _self_test() return
abandoned = read_dids(args.abandoned) if args.abandoned else set() out = build(args.dir, args.tld_map, abandoned, emit_dids_to=args.emit_dids, emit_placed_to=args.emit_placed) with open(args.out, "w") as f: f.write(json.dumps(out, indent=2) + "\n") top = ", ".join(f"{c['country']}={c['total']}" for c in out["countries"][:8]) print(f"Wrote {args.out}: {out['people']} people across {len(out['countries'])} countries " f"({out['contested']} contested); {top}")
if __name__ == "__main__": main()