diff --git a/scripts/nl-sifa-company.py b/scripts/nl-sifa-company.py index 189010e..d137237 100644 --- a/scripts/nl-sifa-company.py +++ b/scripts/nl-sifa-company.py @@ -1,22 +1,23 @@ #!/usr/bin/env python3 """ -nl-sifa-company.py -- Dutch company accounts on atproto, identified via Sifa. +nl-sifa-company.py -- Dutch company/org accounts on atproto, from Sifa's entity DB. -For each Sifa profile that is an organization (`org=true`) recorded as NL-based -- by HQ country -(`orgCountry=NL`, TLD/location-independent) OR by a `.nl` domain (`domainTld=nl`) -- take its -`website` domain and resolve it as an atproto handle. If the domain has an atproto account, that -DID is written to .dids -- feeding the `sifaCompany` source of the NL-stats aggregator -(nl-stats-aggregate.py). Deduped into the identified floor with the other sources. +Reads Sifa's company DATABASE (registry records ingested from Wikidata/GLEIF/ROR/PDL, +plus a curated NL seed), NOT claimed Sifa profiles: every entity recorded as NL-based +whose known domain is bound to a DID:PLC (domain-control verified, bidirectional). +Those DIDs feed the `sifaCompany` source of the NL-stats aggregator (nl-stats-aggregate.py), +deduped into the identified floor with the other sources. -The account's *declared* location (`country=NL`) is intentionally NOT used: `orgCountry`/`domainTld` -are the authoritative "is this an NL company" signals. Both return little until Sifa backfills the -entity HQ-country data, so this runs low for now and grows as that lands. +This answers "which NL companies are on atproto", not "which NL companies use Sifa" -- +so it does not depend on companies signing up. It grows as Sifa's NL entity coverage +and domain->DID bindings grow. -Source: sifa.id public search API (`org`/`website` + `org`/`orgCountry`/`domainTld` filters, -singi-labs/sifa-api#1262 and follow-up). Handle resolution: the public Bluesky AppView. +Source: sifa.id public endpoint GET /api/entities/on-network?country=NL (paginated), +singi-labs/sifa-api#1263. No handle resolution here -- the binding is already proven +server-side. Usage: - python3 nl-sifa-company.py --out-prefix sifa-company [--country NL] [--tld nl] + python3 nl-sifa-company.py --out-prefix sifa-company [--country NL] """ import argparse import json @@ -26,15 +27,15 @@ import urllib.error import urllib.parse import urllib.request -SEARCH = "https://sifa.id/api/accounts/search" -RESOLVE = "https://public.api.bsky.app/xrpc/com.atproto.identity.resolveHandle" +ENDPOINT = "https://sifa.id/api/entities/on-network" +PAGE = 200 def get_json(url, retries=4): for i in range(retries): try: req = urllib.request.Request( - url, headers={"accept": "application/json", "user-agent": "atproto.nl-sifa-company/1.0"} + url, headers={"accept": "application/json", "user-agent": "atproto.nl-sifa-company/2.0"} ) with urllib.request.urlopen(req, timeout=30) as r: return json.loads(r.read().decode("utf-8", "replace")) @@ -48,62 +49,30 @@ def get_json(url, retries=4): return None -def domain_of(website): - if not website: - return None - try: - host = urllib.parse.urlparse(website if "//" in website else "//" + website).hostname - except ValueError: - return None - if not host: - return None - host = host.lower().strip(".") - if host.startswith("www."): - host = host[4:] - return host or None - - def main(): ap = argparse.ArgumentParser() ap.add_argument("--out-prefix", default="sifa-company") ap.add_argument("--dir", default=".") ap.add_argument("--country", default="NL") - ap.add_argument("--tld", default="nl") args = ap.parse_args() - def search_all(extra): - out, cursor = [], None - while True: - q = {"org": "true", "limit": "200", **extra} - if cursor: - q["cursor"] = cursor - j = get_json(f"{SEARCH}?{urllib.parse.urlencode(q)}") - if not j: - break - out += j.get("accounts", []) - cursor = j.get("cursor") - if not cursor: - break - return out - - # NL companies by either authoritative signal, unioned by DID (a company can match both). - seen, accounts = set(), [] - for extra in ({"orgCountry": args.country}, {"domainTld": args.tld}): - for a in search_all(extra): - did = a.get("did") - if did and did not in seen: - seen.add(did) - accounts.append(a) - - dids = set() - for a in accounts: - dom = domain_of(a.get("website")) - if not dom: - continue - j = get_json(f"{RESOLVE}?handle={urllib.parse.quote(dom)}") - did = (j or {}).get("did") - if isinstance(did, str) and did.startswith("did:"): - dids.add(did) + dids, entities = set(), 0 + cursor = None + while True: + q = {"country": args.country, "limit": str(PAGE)} + if cursor: + q["cursor"] = cursor + j = get_json(f"{ENDPOINT}?{urllib.parse.urlencode(q)}") + if not j: + break + for e in j.get("entities", []): + entities += 1 + did = e.get("did") + if isinstance(did, str) and did.startswith("did:"): + dids.add(did) + cursor = j.get("cursor") + if not cursor: + break out = os.path.join(args.dir, f"{args.out_prefix}.dids") tmp = out + ".tmp" @@ -112,7 +81,7 @@ def main(): if dids: f.write("\n") os.replace(tmp, out) - print(f"[sifa-company] {len(dids)} company-domain accounts from {len(accounts)} NL org profiles -> {out}") + print(f"[sifa-company] {len(dids)} NL company domains on atproto (from {entities} entities) -> {out}") if __name__ == "__main__":