Something went wrong. Try again.
Central platform for European atproto.<cc> country community websites atproto.eu
community atproto
Something went wrong. Try again.
Python
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189#!/usr/bin/env python3"""Per-country Bluesky-population estimate, language-weighted, merged into europe-stats.json.
The attributed counts in europe-stats.json are a hard floor (accounts we can place in acountry by a deliberate signal). This adds an ESTIMATE of the real active population percountry, the same kind of modelled number atproto.nl already shows, so the .be/.eu/.noheroes can lead with an estimate instead of the raw floor.
Model: estimate(cc) = K * Σ_lang authors(lang) * weight(cc | lang)
- authors(lang): distinct accounts posting `lang` in the collector window (estimate-nl-users.py `byLang[...].distinct_authors`). null (untracked) langs are skipped. - weight(cc | lang): share of `lang`'s European speakers in country cc (eu-lang-speakers.json). - K: one calibration constant, chosen so estimate(nl) equals the NL model's own figure (nl-stats.json estimate.modelledNl). Every other country is expressed in the same units as the number atproto.nl already publishes, so the four sites agree on method.
LANGUAGE IS NOT COUNTRY. This is the estimate layer only; nothing here is "confirmed" and noDID is placed by language. Each estimate is floored at the country's attributed `total` (anestimate below the count we can actually stand behind would be nonsense). Castilian Spain andPortugal are under-covered (es/pt are not author-tracked); non-EU Balkan/Ukrainian languages areomitted. Both are stated on the page. See eu-lang-speakers.json for the weight provenance.
Usage: python3 scripts/eu-estimate.py --stats europe-stats.json --collector nl-live.json \ --weights scripts/eu-lang-speakers.json --nl-stats nl-stats.json python3 scripts/eu-estimate.py --self-test"""from __future__ import annotations
import argparseimport jsonimport osimport sys
def load_authors(collector_path): """{lang: distinct_authors} for langs the collector tracked (non-null).""" d = json.load(open(collector_path)) out = {} for lang, v in (d.get("byLang") or {}).items(): a = v.get("distinct_authors") if isinstance(a, int) and a > 0: out[lang] = a return out
def raw_by_country(authors, weights): """raw(cc) = Σ_lang authors(lang) * weight(cc|lang), before calibration.""" raw: dict[str, float] = {} for lang, n in authors.items(): for cc, w in (weights.get(lang) or {}).items(): raw[cc] = raw.get(cc, 0.0) + n * w return raw
def estimate(authors, weights, nl_anchor, anchor_cc="nl"): """Calibrated per-country estimate. K makes estimate(anchor_cc) == nl_anchor.""" raw = raw_by_country(authors, weights) base = raw.get(anchor_cc, 0.0) if base <= 0: raise ValueError(f"no language signal for the anchor country {anchor_cc!r}; cannot calibrate") k = nl_anchor / base return {cc: r * k for cc, r in raw.items()}, k
# English-first countries: English is not author-tracked (spoken everywhere), so a# language-weighted estimate can only see a tiny non-English minority (Welsh, Irish) and would# wildly undercount them. We give them NO estimate and leave them out of the aggregate headline# rather than publish a number we know is meaningless. They still show their attributed count.ESTIMATE_EXCLUDE = {"gb", "ie"}
def merge_into_stats(stats, est, exclude=ESTIMATE_EXCLUDE): """Add `estimate` to each covered country (floored at its attributed total) + an aggregate.
Excluded (English-first) countries get no estimate and are left out of peopleEstimate. A country with an attributed floor but no language estimate keeps its floor as the estimate, never less. """ per = {c["country"]: c for c in stats.get("countries", [])} total_est = 0 seen = set() for cc, c in per.items(): seen.add(cc) if cc in exclude: c.pop("estimate", None) continue e = max(int(round(est.get(cc, 0))), int(c.get("total", 0))) # never below the floor c["estimate"] = e total_est += e # A language-only country (estimate but no attributed row) still counts, unless excluded. for cc, val in est.items(): if cc not in seen and cc not in exclude: total_est += int(round(val)) stats["peopleEstimate"] = total_est stats["estimateMethod"] = { "model": "K * Σ_lang authors(lang) * speakerWeight(cc|lang)", "calibratedTo": "atproto.nl modelledNl", "languageUsed": True, "excludes": sorted(exclude), "note": "estimate layer only; language never confirms a country. English-first " "countries (gb, ie) are excluded (English untracked); es/pt under-covered.", } return stats
def _self_test(): weights = { "nl": {"nl": 0.73, "be": 0.27}, "fr": {"fr": 0.89, "be": 0.07}, "de": {"de": 1.0}, } authors = {"nl": 1000, "fr": 2000, "de": 5000, "en": 9999} # en absent from weights -> ignored est, k = estimate(authors, weights, nl_anchor=250000, anchor_cc="nl") # raw(nl) = 1000*0.73 = 730 -> K = 250000/730 assert abs(k - 250000 / 730) < 1e-6, k assert abs(est["nl"] - 250000) < 1, est # anchor reproduced exactly # be = (1000*0.27 + 2000*0.07) * K = (270+140)*K assert abs(est["be"] - 410 * k) < 1e-6, est # de present, en (no weights) contributed nothing assert "de" in est and est["de"] == 5000 * k
stats = {"countries": [ {"country": "nl", "total": 3000}, {"country": "be", "total": 900}, {"country": "de", "total": 7081}, {"country": "es", "total": 11666}, # no language estimate -> floored at total {"country": "gb", "total": 8738}, # English-first -> excluded, no estimate ]} merged = merge_into_stats(stats, est) per = {c["country"]: c for c in merged["countries"]} assert per["nl"]["estimate"] == 250000, per["nl"] assert per["es"]["estimate"] == 11666, per["es"] # floor kept, no lang signal assert per["de"]["estimate"] == max(int(round(5000 * k)), 7081), per["de"] assert "estimate" not in per["gb"], per["gb"] # excluded English-first # peopleEstimate = every covered country's estimate + language-only countries (fr here), # and must NOT include the excluded gb. covered = sum(per[c]["estimate"] for c in ("nl", "be", "de", "es")) + int(round(est["fr"])) assert merged["peopleEstimate"] == covered, (merged["peopleEstimate"], covered) assert merged["estimateMethod"]["languageUsed"] is True assert merged["estimateMethod"]["excludes"] == ["gb", "ie"] print("eu-estimate self-test OK")
def main(): ap = argparse.ArgumentParser() ap.add_argument("--stats", default="europe-stats.json", help="europe-stats.json to merge into") ap.add_argument("--collector", default="nl-live.json", help="collector full-result JSON (byLang)") ap.add_argument("--weights", default=os.path.join(os.path.dirname(os.path.abspath(__file__)), "eu-lang-speakers.json")) ap.add_argument("--nl-stats", default="nl-stats.json", help="reads estimate.modelledNl") ap.add_argument("--nl-anchor", type=int, default=None, help="override the NL anchor directly") ap.add_argument("--out", default=None, help="default: overwrite --stats") ap.add_argument("--self-test", action="store_true") args = ap.parse_args()
if args.self_test: _self_test() return
anchor = args.nl_anchor if anchor is None: anchor = int(json.load(open(args.nl_stats)).get("estimate", {}).get("modelledNl", 0)) if not anchor: print("no NL anchor (estimate.modelledNl missing and --nl-anchor unset)", file=sys.stderr) sys.exit(1)
authors = load_authors(args.collector) weights = json.load(open(args.weights)).get("weights", {}) est, k = estimate(authors, weights, anchor) stats = json.load(open(args.stats)) merge_into_stats(stats, est)
out = args.out or args.stats tmp = out + ".tmp" with open(tmp, "w") as f: f.write(json.dumps(stats, indent=2) + "\n") os.replace(tmp, out) top = sorted(((c["country"], c.get("estimate", 0)) for c in stats["countries"]), key=lambda kv: -kv[1])[:8] pretty = ", ".join(f"{cc}={n:,}" for cc, n in top) print(f"Wrote {out}: K={k:.1f}, peopleEstimate={stats['peopleEstimate']:,}; {pretty}")
if __name__ == "__main__": main()