Something went wrong. Try again.
Central platform for European atproto.<cc> country community websites atproto.eu
community atproto
Something went wrong. Try again.
Python
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234#!/usr/bin/env python3"""Attribute each DID to exactly ONE country, by a fixed precedence of signals.
Why single-valued: the NL pipeline unions five sources and dedupes by DID, so nothing isdouble-counted within one country. Across 27 countries that breaks -- one DID can match `nl`(Dutch posts), `es` (a .es handle) and `de` (a Sifa location), and per-country counts would thensum to more than the union of people. The fix is not to drop signals but to make attributionsingle-valued for counting.
Ordering principle, in four rules: 1. signals that answer *this* question outrank signals that answer a neighbouring one 2. declarations beat inferences 3. deliberate choices beat incidental ones 4. curated beats automated
Rule 1 does most of the work, and it is NOT the same as ranking by how authoritative a signalfeels. A member record is a stronger declaration than a domain choice, but it answers "where doyou take part", not "where are you" -- someone in Luxembourg with no local events joins theBelgian, French and German communities and is none of those nationalities. Ranking it high wouldinflate the countries that run events and empty the ones that do not, which is worst for exactlythe small countries whose data is already thinnest.
This module is pure: it takes signals in, returns an attribution out. No I/O, no network.
Self-test: python3 scripts/attribution.py --self-test"""from __future__ import annotations
import argparseimport sysfrom collections import Counterfrom dataclasses import dataclass, field
# Rank 1 is strongest. Names are stable identifiers -- they are published per country so a# reader can see whether a count rests on declarations or on guesses.RANKS = { "sifaLocation": 1, # id.sifa.profile.location, isPrimary -- where you live, declared by you "orgHqDeclared": 1, # (unused; see note below) "cityDomain": 2, # .amsterdam / .wien / .cat -- nobody registers these by accident "strongTld": 3, # .nl .de .fr .pl ... -- a personal handle "companyDomain": 3, # an organisation's website ccTLD: the market it addresses "memberRecord": 4, # eu.atcommons.member with exactly one country -- where you take part "verifier": 4, # a country community's verifier vouched (Eurosky "Verified by <cc>-...") "starterpack": 5, # a human vouched "weakTld": 6, # .is .es .at .se .it -- deliberate but word-play-prone "orgCountry": 6, # a registered HQ is a legal/tax choice, not a place of operation "uniqueLang": 7, # nl, pl, hu, el -- inference "sharedLang": 8, # de, fr, sv, ... -- weakest inference}del RANKS["orgHqDeclared"] # kept above only to document that HQ was considered for rank 1
# A signal at this rank or better is presented as "confirmed"; the rest are estimate-layer only.CONFIRMED_MAX_RANK = 5
@dataclass(frozen=True)class Signal: """One piece of evidence that a DID belongs to a country.""" kind: str country: str
@property def rank(self) -> int: try: return RANKS[self.kind] except KeyError: raise ValueError(f"unknown signal kind: {self.kind!r}") from None
@dataclassclass Attribution: did: str country: str | None # None when contested or when there is no signal at all rank: int | None kind: str | None contested: bool = False candidates: tuple[str, ...] = field(default_factory=tuple)
@property def confirmed(self) -> bool: return self.country is not None and self.rank is not None and self.rank <= CONFIRMED_MAX_RANK
def attribute(did: str, signals: list[Signal]) -> Attribution: """Pick the country for a DID, or mark it contested.
Ties at the winning rank that disagree on the country are NOT broken. Two city domains, or a multi-country member record with no Sifa location, give a genuine coin flip -- and a coin flip recorded as fact is worse than an honest "unattributed", because it is indistinguishable from a real answer downstream.
Ties at the winning rank that AGREE are not contested: several sources saying "nl" is corroboration, not conflict. """ if not signals: return Attribution(did, None, None, None) best = min(s.rank for s in signals) top = [s for s in signals if s.rank == best] countries = {s.country for s in top} if len(countries) > 1: return Attribution(did, None, best, None, contested=True, candidates=tuple(sorted(countries))) winner = top[0] return Attribution(did, winner.country, best, winner.kind)
def attribute_all(signals_by_did: dict[str, list[Signal]]) -> dict[str, Attribution]: return {did: attribute(did, sigs) for did, sigs in signals_by_did.items()}
def summarise(attributions: dict[str, Attribution]) -> dict: """Per-country totals plus the rank mix that produced them.
The rank mix is published, not internal: a country resting on ranks 6-8 is visibly weaker evidence than one resting on 1-3, and hiding that would make a guess look like a count. """ per: dict[str, dict] = {} contested = 0 unattributed = 0 for a in attributions.values(): if a.contested: contested += 1 continue if a.country is None: unattributed += 1 continue c = per.setdefault(a.country, {"total": 0, "confirmed": 0, "byRank": Counter(), "byKind": Counter()}) c["total"] += 1 c["confirmed"] += 1 if a.confirmed else 0 c["byRank"][a.rank] += 1 c["byKind"][a.kind] += 1 for c in per.values(): c["byRank"] = dict(sorted(c["byRank"].items())) c["byKind"] = dict(sorted(c["byKind"].items(), key=lambda kv: -kv[1])) return { "countries": dict(sorted(per.items(), key=lambda kv: -kv[1]["total"])), "contested": contested, "unattributed": unattributed, "attributed": sum(c["total"] for c in per.values()), }
def _self_test() -> None: S = Signal # Precedence: location beats a domain beats a member record. a = attribute("did:x", [S("memberRecord", "be"), S("strongTld", "de"), S("sifaLocation", "lu")]) assert (a.country, a.rank, a.kind) == ("lu", 1, "sifaLocation"), a
# The Luxembourg case: joining BE/FR/DE communities must not make someone Belgian. a = attribute("did:lu", [S("memberRecord", "be")]) assert a.country == "be" and a.rank == 4, a a = attribute("did:lu", [S("memberRecord", "be"), S("sifaLocation", "lu")]) assert a.country == "lu", a
# A domain choice outranks a member record: place beats participation. a = attribute("did:d", [S("memberRecord", "be"), S("cityDomain", "nl")]) assert a.country == "nl" and a.rank == 2, a
# orgCountry sits with weak TLDs, not at rank 1: a registered HQ is a tax choice. assert RANKS["orgCountry"] == RANKS["weakTld"] == 6 a = attribute("did:o", [S("orgCountry", "nl"), S("strongTld", "de")]) assert a.country == "de", a
# A country verifier is a deliberate vouch: it beats a starter pack and is "confirmed", # but a declared place (a ccTLD handle) still outranks it -- vouch is not location. a = attribute("did:v", [S("verifier", "be"), S("starterpack", "nl")]) assert a.country == "be" and a.rank == 4 and a.confirmed, a a = attribute("did:v2", [S("verifier", "be"), S("strongTld", "nl")]) assert a.country == "nl" and a.rank == 3, a # Verifier and member record are peer affiliation signals: agreement corroborates, # cross-country disagreement is contested rather than coin-flipped. a = attribute("did:v3", [S("verifier", "be"), S("memberRecord", "be")]) assert a.country == "be" and not a.contested, a a = attribute("did:v4", [S("verifier", "be"), S("memberRecord", "fr")]) assert a.contested and a.candidates == ("be", "fr"), a
# Same-rank disagreement is contested, never a coin flip. a = attribute("did:c", [S("cityDomain", "nl"), S("cityDomain", "be")]) assert a.contested and a.country is None and a.candidates == ("be", "nl"), a
# Same-rank agreement is corroboration, not conflict. a = attribute("did:agree", [S("strongTld", "nl"), S("strongTld", "nl")]) assert not a.contested and a.country == "nl", a
# No signals at all: not attributed, not contested. a = attribute("did:none", []) assert a.country is None and not a.contested and a.rank is None, a
# Language never wins against anything deliberate, and is never "confirmed". a = attribute("did:l", [S("uniqueLang", "nl"), S("weakTld", "es")]) assert a.country == "es" and a.rank == 6, a assert not attribute("did:l2", [S("sharedLang", "de")]).confirmed assert not attribute("did:l3", [S("uniqueLang", "nl")]).confirmed assert attribute("did:s", [S("starterpack", "nl")]).confirmed
# Unknown signal kinds fail loudly rather than silently scoring last. try: attribute("did:bad", [S("vibes", "nl")]) except ValueError: pass else: raise AssertionError("unknown signal kind should raise")
# Summary: per-country totals, rank mix, contested and unattributed kept separate. summary = summarise(attribute_all({ "did:1": [S("sifaLocation", "nl")], "did:2": [S("strongTld", "nl")], "did:3": [S("sharedLang", "de")], "did:4": [S("cityDomain", "nl"), S("cityDomain", "be")], "did:5": [], })) assert summary["countries"]["nl"]["total"] == 2, summary assert summary["countries"]["nl"]["confirmed"] == 2, summary assert summary["countries"]["de"]["confirmed"] == 0, summary assert summary["contested"] == 1 and summary["unattributed"] == 1, summary assert summary["attributed"] == 3, summary
# A DID is never counted twice, which is the whole point of this module. assert sum(c["total"] for c in summary["countries"].values()) == summary["attributed"]
print("attribution self-test: OK")
if __name__ == "__main__": ap = argparse.ArgumentParser() ap.add_argument("--self-test", action="store_true") args = ap.parse_args() if args.self_test: _self_test() else: ap.print_help() sys.exit(1)