Something went wrong. Try again.
Central platform for European atproto.<cc> country community websites atproto.eu
community atproto
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295#!/usr/bin/env python3# /// script# dependencies = []# ///"""Fold the per-DID facts into the counts-only JSON the site publishes.
Reads nl-did-facts.jsonl (creation dates, handle history, PDS) and nl-lexicons.jsonl(collections per repo), writes nl-facts.json: aggregates only, no DIDs, no handles. Samerule as nl-stats.json -- the per-account files stay on the collector host.
uv run scripts/facts-aggregate.py --dir . --out nl-facts.json uv run scripts/facts-aggregate.py --self-test"""
from __future__ import annotations
import argparseimport jsonimport osimport sysimport timeimport urllib.parsefrom collections import Counter
# Bluesky's own collections answer nothing: nearly every account has them.BSKY_PREFIXES = ("app.bsky.", "chat.bsky.")# Bluesky's hosted PDS fleet is <name>.<region>.host.bsky.network. Counting each shard# separately would say "the community is spread over 40 hosts", which is false: they are# one operator. What matters is hosted-by-Bluesky versus hosted elsewhere.BSKY_PDS_SUFFIX = ".host.bsky.network"# A lexicon used by one or two accounts is noise, and naming it invites over-reading.MIN_LEXICON_ACCOUNTS = 3
# NSIDs answer "which record types exist"; people ask "which apps do they use". Counting# per NSID also distorts the ranking: an app that models a profile as seven record types# gets seven small bars, while a single-record app gets one. Group by authority, and# count each ACCOUNT once per app.## The authority is the first two labels of the NSID, except where a namespace hosts# several distinct products, which is why this is a table rather than a rule.APP_BY_PREFIX = { "id.sifa": "Sifa", "sh.tangled": "Tangled", "com.whtwnd": "WhiteWind", "fyi.unravel.frontpage": "Frontpage", "events.smokesignal": "Smoke Signal", "blue.flashes": "Flashes", "my.skylights": "Skylights", "pub.leaflet": "Leaflet", "app.popsky": "Popsky", "blue.linkat": "Linkat", "blue.badge": "Badge", "social.grain": "Grain", "fm.teal": "teal.fm", "app.graze": "Graze", "chat.roomy": "Roomy", "site.standard": "standard.site", "social.pinksky": "Pinksky", "social.mu": "mu.social", "net.anisota": "Anisota", "tech.tokimeki": "Tokimeki", "com.germnetwork": "Germ", "space.aoi.shiharai": "Shiharai", "com.atproto": "AT Protocol core", # Not an app: a shared community namespace, kept separate so it is not read as one. "community.lexicon": "community lexicons",}
def app_of(nsid: str) -> str: for prefix, label in APP_BY_PREFIX.items(): if nsid == prefix or nsid.startswith(prefix + "."): return label parts = nsid.split(".") return ".".join(reversed(parts[:2])) if len(parts) >= 2 else nsid
def read_jsonl(path: str) -> list[dict]: rows = [] if not path or not os.path.exists(path): return rows with open(path, encoding="utf-8") as fh: for line in fh: line = line.strip() if not line: continue try: rows.append(json.loads(line)) except ValueError: continue # a partial line from a run still in flight return rows
def latest_by_did(rows: list[dict]) -> dict[str, dict]: out: dict[str, dict] = {} for row in rows: if row.get("did"): out[row["did"]] = row # append-only files: the last line is the current one return out
def host_of(endpoint: str | None) -> str | None: if not endpoint: return None try: return urllib.parse.urlparse(endpoint).hostname except ValueError: return None
def is_custom_domain(handle: str | None) -> bool: """A handle the account holder controls, rather than one a PDS handed out.""" if not handle: return False h = handle.lower() return not (h.endswith(".bsky.social") or h == "handle.invalid")
def build(facts: dict[str, dict], lexicons: dict[str, dict], now: str) -> dict: dated = [f for f in facts.values() if f.get("createdAt")] creation_by_month = Counter(f["createdAt"][:7] for f in dated)
hosts = Counter() bsky_hosted = 0 for fact in facts.values(): host = host_of(fact.get("pds")) if not host: continue if host.endswith(BSKY_PDS_SUFFIX) or host == "bsky.social": bsky_hosted += 1 else: hosts[host] += 1
# When each account FIRST took a handle it controls. Domain adoption is a maturity # signal for this community specifically: it is the step from "I have an account" to # "this is my identity", and it is the same step that makes an account detectable # as Dutch by handle. custom_by_month = Counter() custom_now = 0 for fact in facts.values(): handles = fact.get("handles") or [] for entry in handles: if is_custom_domain(entry.get("handle")) and entry.get("at"): custom_by_month[entry["at"][:7]] += 1 break if handles and is_custom_domain(handles[-1].get("handle")): custom_now += 1
checked = [row for row in lexicons.values() if row.get("collections") is not None] lex_counter: Counter[str] = Counter() app_counter: Counter[str] = Counter() beyond_users = 0 for row in checked: beyond = [c for c in row["collections"] if not c.startswith(BSKY_PREFIXES)] if beyond: beyond_users += 1 for nsid in beyond: lex_counter[nsid] += 1 # One account, one vote per app, however many record types it holds. for app in {app_of(nsid) for nsid in beyond}: app_counter[app] += 1
return { "generatedAt": now, "accounts": { "known": len(facts), "dated": len(dated), "reposChecked": len(checked), }, # Every series below covers only the accounts our detectors identified, so it is # a shape, not a census. The site says so where it renders them. "creationByMonth": dict(sorted(creation_by_month.items())), "pds": { "blueskyHosted": bsky_hosted, "elsewhere": sum(hosts.values()), "topOtherHosts": [{"host": h, "accounts": n} for h, n in hosts.most_common(10)], }, "handles": { "customDomainNow": custom_now, "firstCustomDomainByMonth": dict(sorted(custom_by_month.items())), }, "lexicons": { "accountsUsingSomething": beyond_users, "byApp": [ {"app": app, "accounts": n} for app, n in app_counter.most_common() if n >= MIN_LEXICON_ACCOUNTS ], "byNsid": [ {"nsid": nsid, "accounts": n} for nsid, n in lex_counter.most_common() if n >= MIN_LEXICON_ACCOUNTS ], }, }
def self_test() -> int: facts = { "did:plc:a": { "did": "did:plc:a", "createdAt": "2024-11-05T00:00:00.000Z", "handles": [ {"handle": "a.bsky.social", "at": "2024-11-05T00:00:00.000Z"}, {"handle": "a.nl", "at": "2024-12-01T00:00:00.000Z"}, ], "pds": "https://morel.us-east.host.bsky.network", }, "did:plc:b": { "did": "did:plc:b", "createdAt": "2023-05-01T00:00:00.000Z", "handles": [{"handle": "b.bsky.social", "at": "2023-05-01T00:00:00.000Z"}], "pds": "https://pds.example.nl", }, "did:plc:c": {"did": "did:plc:c", "createdAt": None, "handles": [], "pds": None}, } lexicons = { "did:plc:a": {"collections": ["app.bsky.feed.post", "id.sifa.profile", "sh.tangled.repo"]}, "did:plc:b": {"collections": ["app.bsky.feed.post"]}, "did:plc:c": {"collections": None}, # unreadable repo } out = build(facts, lexicons, "2026-08-30T00:00:00Z") assert out["accounts"] == {"known": 3, "dated": 2, "reposChecked": 2}, out["accounts"] assert out["creationByMonth"] == {"2023-05": 1, "2024-11": 1}, out["creationByMonth"] # Bluesky's shards collapse to one operator; a self-hosted PDS is counted separately. assert out["pds"]["blueskyHosted"] == 1 and out["pds"]["elsewhere"] == 1, out["pds"] assert out["pds"]["topOtherHosts"] == [{"host": "pds.example.nl", "accounts": 1}], out["pds"] # 'a' moved to a custom domain in 2024-12; 'b' never did. assert out["handles"] == { "customDomainNow": 1, "firstCustomDomainByMonth": {"2024-12": 1}, }, out["handles"] # One account uses something beyond Bluesky, but each lexicon is below the floor. assert out["lexicons"]["accountsUsingSomething"] == 1, out["lexicons"] assert out["lexicons"]["byNsid"] == [], out["lexicons"] assert out["lexicons"]["byApp"] == [], out["lexicons"]
# Grouping: seven Sifa record types in one repo is one account using Sifa, and the # community namespace is never folded into an app. assert app_of("id.sifa.profile.skill") == "Sifa" assert app_of("community.lexicon.payments.webMonetization") == "community lexicons" assert app_of("net.anisota.beta.game.log") == "Anisota" assert app_of("xyz.unknown.thing.record") == "unknown.xyz" many = { f"did:plc:{i}": { "collections": ["app.bsky.feed.post", "id.sifa.profile.self", "id.sifa.profile.skill"] } for i in range(4) } grouped = build({}, many, "2026-08-30T00:00:00Z")["lexicons"] assert grouped["byApp"] == [{"app": "Sifa", "accounts": 4}], grouped["byApp"] assert {r["nsid"] for r in grouped["byNsid"]} == {"id.sifa.profile.self", "id.sifa.profile.skill"} print("self-test OK") return 0
def main() -> int: ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument("--dir", default=".") ap.add_argument("--facts", default=None) ap.add_argument("--lexicons", default=None) ap.add_argument("--dids", default=None, help="restrict to DIDs in this file (one per line). Lets a per-country facts " "file be sliced out of the EU-wide scan without a separate scan.") ap.add_argument("--out", default="nl-facts.json") ap.add_argument("--self-test", action="store_true") args = ap.parse_args()
if args.self_test: return self_test()
facts = latest_by_did(read_jsonl(args.facts or os.path.join(args.dir, "nl-did-facts.jsonl"))) lexicons = latest_by_did(read_jsonl(args.lexicons or os.path.join(args.dir, "nl-lexicons.jsonl")))
# Slice to one country's attributed DIDs, so be-facts.json / no-facts.json come straight out of # the EU-wide scan (eu-did-facts.jsonl) instead of needing their own crawl. if args.dids: allow = {ln.strip() for ln in open(args.dids, encoding="utf-8") if ln.strip().startswith("did:")} facts = {d: v for d, v in facts.items() if d in allow} lexicons = {d: v for d, v in lexicons.items() if d in allow} out = build(facts, lexicons, time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())) with open(args.out, "w", encoding="utf-8") as fh: fh.write(json.dumps(out, indent=2, ensure_ascii=False) + "\n") print( f"[facts-aggregate] {out['accounts']['known']} accounts, " f"{out['accounts']['dated']} dated, {out['accounts']['reposChecked']} repos, " f"{len(out['lexicons']['byNsid'])} lexicons -> {args.out}", file=sys.stderr, ) return 0
if __name__ == "__main__": raise SystemExit(main())