diff --git a/backend/src/ingest/classifier.zig b/backend/src/ingest/classifier.zig index b229b18..72e3963 100644 --- a/backend/src/ingest/classifier.zig +++ b/backend/src/ingest/classifier.zig @@ -38,7 +38,7 @@ const THRESHOLD: f64 = 0.50; // re-scores the whole corpus from a clean slate. v2 = precision fix (content-word // scaffold only, curation veto). v3 = threshold 0.55→0.50. v4 = model-pass gate // (heuristic flags → LLM confirms content is machine-generated before emit). -const SCORING_VERSION: i64 = 8; // v8: judge model → Qwen3.6-35B (3/3 on judge-eval; 7B was 0/3) +const SCORING_VERSION: i64 = 9; // v9: composed-vs-generated prompt + judge → gemma-4-12B (self-served) // review pipeline states. The heuristic is a cheap PRE-FILTER: it never emits // directly (titles can't tell a branded real blog from a registry mirror). It @@ -57,12 +57,13 @@ const MAX_REVIEW_ATTEMPTS: i64 = 5; // give up after this many inconclusive revi // review provider: any OpenAI-compatible chat-completions endpoint. Defaults // to co/core; override all three via env (REVIEW_API_URL / REVIEW_MODEL / // REVIEW_API_KEY) to switch providers with a secrets change, no deploy of code. -// picked by scripts/judge-eval (2026-07-01): 3/3 known-truth accounts, every -// completed vote correct. Qwen2.5-7B (the previous judge) scored 0/3 — -// confidently inverted on all three. Slower + reasoning-style, which the -// worker tolerates (votes retry on inconclusive; latency doesn't back -// anything up — candidates just stay PENDING longer). -const DEFAULT_REVIEW_MODEL = "mlx-community/Qwen3.6-35B-A3B-4bit"; +// picked by scripts/judge-eval: gemma-4-12B went 19/19 conclusive votes +// correct across two runs on the composed-vs-generated prompt (2026-07-02), +// and it's the model our own provider machine serves — reliability and +// economics we control. Qwen2.5-7B (the original judge) was 0/3, confidently +// inverted. Reasoning-style + slow is fine: the worker is a durable queue, +// candidates just stay PENDING longer. +const DEFAULT_REVIEW_MODEL = "mlx-community/gemma-4-12B-it-8bit"; const DEFAULT_REVIEW_URL = "https://console.cocore.dev/api/v1/chat/completions"; const ReviewCfg = struct { @@ -631,26 +632,34 @@ fn reviewVote(allocator: Allocator, cfg: ReviewCfg, io: Io, did: []const u8, vot const samples = try fetchVoteMaterial(allocator, did, vote); defer allocator.free(samples); - // prompt validated against the known cases on cocore Qwen2.5-7B: the framing - // (our task: long-form human writing vs automated feeds, "coherent ≠ human") - // + concrete examples are what make a small model separate a real blog from a - // transit/catalog feed. Without them it calls everything coherent "human". + // the line is COMPOSED vs GENERATED, not human vs machine: most banned + // content was human-written at the source (patents, recall notices, + // transcripts) and original AI writing is welcome. Keep this text in + // lockstep with scripts/judge-eval PROMPT_HEAD, and re-validate any change + // there against the known-truth set before shipping (19/19 conclusive + // votes correct on gemma-4-12B + Qwen3.6-35B, 2026-07-02). var prompt: std.Io.Writer.Allocating = .init(allocator); defer prompt.deinit(); try prompt.writer.writeAll( - \\A search engine indexes original long-form writing by people (blogs, essays, - \\articles). It must EXCLUDE accounts that are automated feeds, catalogs, or - \\database exports — even when the text reads coherently. + \\A search engine indexes writing: documents COMPOSED by an author — a person or + \\an AI — who chose what to say. It must EXCLUDE accounts that GENERATE + \\documents from a data source: one document per database row, feed event, + \\catalog entry, or result, where a template plus the data determines the text. \\ - \\The test: did a PERSON sit and write each item as original prose, OR is an - \\automated system emitting one record per database row / feed event / catalog - \\entry / log? + \\The test is NOT whether the text is fluent, and NOT whether a human once wrote + \\the underlying material — patents, recall notices, episode summaries, and + \\transcripts were all written by people, but republishing them one-per-record + \\is still generation. The test: does each document exist because its author had + \\something to say, OR because a record exists in some dataset? \\ - \\machine=true examples: a transit-alert bot ("Red Line delayed near Roosevelt"), - \\a patent-database mirror, a TV-episode catalog (one entry per episode), - \\product-recall summaries, daily log entries, chart/stats dumps. Coherent != human. - \\machine=false examples: a personal blog (tech notes, essays, reviews), even with - \\branded/templated titles or a numbered series. + \\machine=true examples: a patent-database mirror, vehicle-recall summaries (one + \\per recall), a TV-episode catalog (one per episode), transit alerts (one per + \\service event), chart/stats dumps (one per chart-day), tournament results (one + \\per event), fundraiser announcements stamped from campaign records — even when + \\the underlying data belongs to the account itself. + \\machine=false examples: a personal blog or essay series (even branded, + \\numbered, templated-looking titles, or extremely prolific), a daily journal + \\(the date is a schedule, not a data source), original writing by an AI agent. \\ \\Evidence from ONE account (facts, a title sample spanning its whole \\history, and excerpts from the middle of several documents): diff --git a/docs/exclusions.md b/docs/exclusions.md index 857de42..302275c 100644 --- a/docs/exclusions.md +++ b/docs/exclusions.md @@ -1,17 +1,22 @@ # exclusions: what we keep out of the corpus, and why -pub-search indexes **long-form writing by people**. some actors publish -valid, well-signed AT Protocol records that are nonetheless not that — +pub-search indexes **composed writing** — documents an author (a person or +an AI) wrote because they had something to say. some actors publish valid, +well-signed AT Protocol records that are nonetheless not that — machine-generated registry mirrors, scraper bridges, bulk archives. this file is the registry of every manual exclusion: who, when, why, and the evidence. if we're going to editorialize, we explain ourselves. ## what qualifies (the policy line) -an author is excluded when **all** of these hold: +the line is **composed vs. generated**, not human vs. machine: most banned +content was human-written at the source (patents, recall notices, +transcripts), and original writing by an AI is welcome. an author is +excluded when **all** of these hold: -1. **no human authorship** — the records are generated from an external - database/feed, not written +1. **generated, not composed** — each document exists because a record + exists in some dataset (external or the account's own), with a template + determining the text; not because an author chose to write it 2. **bulk scale** — the volume materially distorts the corpus or a topic's search results (rule of thumb: would be >1% of the corpus, or owns a topic's semantic results) @@ -19,7 +24,8 @@ an author is excluded when **all** of these hold: removes noise, not voices an author is NOT excluded for: low quality, controversial content, high -(human) volume, or self-promotion. the line is authorship, not taste. +volume, self-promotion, or being an AI. the line is how the documents come +to exist, not taste and not the author's species. ## how exclusions are enforced (the new architecture) diff --git a/scripts/judge-eval b/scripts/judge-eval index 8d31c7f..a789c25 100755 --- a/scripts/judge-eval +++ b/scripts/judge-eval @@ -31,6 +31,7 @@ KNOWN = { "did:plc:u5kk4h7tr4s3ntskbmhl5z7d": ("sksksketch.net", False), # human illustrator "did:plc:5swhfspkrynnbidlkrkch3lh": ("prideraiser.org", True), # fundraiser feed bot "did:plc:4z33k5fjzw2ew3u373pg7ku5": ("festivus (episode catalog)", True), + "did:plc:sttgf52vkk46f6yuknvqxvgh": ("coryd.dev (942-doc human)", False), # volume trap } DEFAULT_MODELS = [ @@ -47,19 +48,25 @@ EXCERPT_LEN = 800 COCORE_URL = "https://console.cocore.dev/api/v1/chat/completions" -PROMPT_HEAD = """A search engine indexes original long-form writing by people (blogs, essays, -articles). It must EXCLUDE accounts that are automated feeds, catalogs, or -database exports — even when the text reads coherently. - -The test: did a PERSON sit and write each item as original prose, OR is an -automated system emitting one record per database row / feed event / catalog -entry / log? - -machine=true examples: a transit-alert bot ("Red Line delayed near Roosevelt"), -a patent-database mirror, a TV-episode catalog (one entry per episode), -product-recall summaries, daily log entries, chart/stats dumps. Coherent != human. -machine=false examples: a personal blog (tech notes, essays, reviews), even with -branded/templated titles or a numbered series. +PROMPT_HEAD = """A search engine indexes writing: documents COMPOSED by an author — a person or +an AI — who chose what to say. It must EXCLUDE accounts that GENERATE +documents from a data source: one document per database row, feed event, +catalog entry, or result, where a template plus the data determines the text. + +The test is NOT whether the text is fluent, and NOT whether a human once wrote +the underlying material — patents, recall notices, episode summaries, and +transcripts were all written by people, but republishing them one-per-record +is still generation. The test: does each document exist because its author had +something to say, OR because a record exists in some dataset? + +machine=true examples: a patent-database mirror, vehicle-recall summaries (one +per recall), a TV-episode catalog (one per episode), transit alerts (one per +service event), chart/stats dumps (one per chart-day), tournament results (one +per event), fundraiser announcements stamped from campaign records — even when +the underlying data belongs to the account itself. +machine=false examples: a personal blog or essay series (even branded, +numbered, templated-looking titles, or extremely prolific), a daily journal +(the date is a schedule, not a data source), original writing by an AI agent. Evidence from ONE account (facts, a title sample spanning its whole history, and excerpts from the middle of several documents): diff --git a/scripts/labeler-setup b/scripts/labeler-setup index 05f853c..ae6f8e1 100755 --- a/scripts/labeler-setup +++ b/scripts/labeler-setup @@ -65,10 +65,12 @@ LABEL_DECLARATION = { "lang": "en", "name": "machine-generated", "description": ( - "this account's content is produced by an automated " - "system, not written by a person — e.g. registry/feed " - "mirrors, transit-alert bots, catalogs, database exports. " - "pub-search excludes these from its human-writing search." + "this account's documents are generated from a data " + "source — one per database row, feed event, or catalog " + "entry — rather than composed by an author (a person or " + "an AI). e.g. registry mirrors, transit-alert bots, " + "catalogs, database exports. pub-search excludes these " + "from search." ), } ], diff --git a/site/labels.html b/site/labels.html index 959c0eb..b9e8d87 100644 --- a/site/labels.html +++ b/site/labels.html @@ -62,8 +62,9 @@
- pub-search indexes original writing. Accounts that emit records rather than prose —
- feeds, catalogs, database exports — get a signed
+ pub-search indexes writing that an author — a person or an AI — composed.
+ Accounts that generate documents from a dataset (one per patent, recall, chart,
+ episode, alert) get a signed
machine-generated
label and are excluded from search. The decision is automated: a pattern filter nominates,
a model on co/core decides.
@@ -78,7 +79,7 @@