From 6270e71141e9eff8fb05308da8a627b706e6301b Mon Sep 17 00:00:00 2001 From: Cameron Pfiffer Date: Mon, 20 Jul 2026 20:21:08 -0700 Subject: [PATCH] Publish uncertainty synthesis with native math. MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 👾 Generated with [Letta Code](https://letta.com) Co-Authored-By: Letta Code --- README.md | 16 ++- docs/public-knowledge.md | 24 ++-- ...calibration-and-probabilistic-inference.md | 118 ++++++++++++++++++ package.json | 3 + public/site.css | 17 +++ src/markdown.test.ts | 23 ++++ src/markdown.ts | 5 + 7 files changed, 193 insertions(+), 13 deletions(-) create mode 100644 knowledge/published/linguistic-calibration-and-probabilistic-inference.md create mode 100644 src/markdown.test.ts diff --git a/README.md b/README.md index 8e7fc2d..e163bf6 100644 --- a/README.md +++ b/README.md @@ -85,18 +85,22 @@ pnpm knowledge:promote example \ --review-receipt-digest sha256:private-receipt-digest \ --confirm-public -# Preview the ATProto/Leaflet records without writing them. -pnpm knowledge:sync -- --include-drafts --json +# Preview the one approved Standard.site portability canary. +pnpm knowledge:sync -- --slug public-knowledge -# Reconcile approved entries only. Requires Cameron's PDS credentials. -pnpm knowledge:sync -- --apply +# Reconcile that exact reviewed canary. There is deliberately no bulk mode. +pnpm knowledge:sync -- --slug public-knowledge --apply ``` `knowledge/policy.json` is default-deny and owns the Obsidian source mappings. Draft Markdown is gitignored under `knowledge/staged/`. The Docker image copies only `knowledge/published/`, so a local draft cannot leak through a broad build -context. The public file stores only a digest of the private route-scoped review -receipt. Promotion and deployment are separate actions. +context. Canonical Knowledge remains reviewed local Markdown. The guarded sync +command is authorized only for the reviewed `public-knowledge` portability +canary in `knowledge/atproto-manifest.json`; it requires an exact slug and has no +bulk mode. The public file stores only a digest of the private route-scoped +review receipt. Promotion, protocol mirroring, and deployment are separate +actions. ## Deployment diff --git a/docs/public-knowledge.md b/docs/public-knowledge.md index ca2a01a..6d5712f 100644 --- a/docs/public-knowledge.md +++ b/docs/public-knowledge.md @@ -6,16 +6,20 @@ links, and annotations. It is not an export of The Coil. ## Storage boundary -Knowledge entries are local Markdown documents rendered only by this site. -They are not Standard.site records. +Knowledge entries are canonical local Markdown documents rendered by this +site. Exactly one reviewed entry, `public-knowledge`, is also mirrored as a +controlled Standard.site portability canary under a dedicated, non-discoverable +Knowledge publication. Standard.site and Cameron's Leaflet publication are reserved for blog posts and writing Cameron publishes personally. Semble links and Margin annotations remain protocol-native records because those systems own their public object models. -Do not create or restore `site.standard.document/knowledge-*` records. The -initial five-record experiment was removed on July 20, 2026, together with its -manifest, compiler, reconciler, remote route fallback, and protocol footers. +Do not create or restore the earlier `site.standard.document/knowledge-*` +records, attach Knowledge to Cameron's Discover-enabled Leaflet publication, or +mirror another slug by analogy. `knowledge/atproto-manifest.json` allowlists the +single canary and binds it to the reviewed content digest. The sync command +requires exactly one `--slug`; there is deliberately no bulk mode. ## Source layout @@ -136,7 +140,13 @@ pnpm knowledge:promote example \ # Build and deploy the Markdown-backed site. docker build -t cameron-site:knowledge . fly deploy --remote-only + +# Preview the one approved Standard.site portability canary. +pnpm knowledge:sync -- --slug public-knowledge + +# Reconcile only that exact canary when separately authorized. +pnpm knowledge:sync -- --slug public-knowledge --apply ``` -There is deliberately no `knowledge:sync` command. Standard.site publication -is a different content class. +Standard.site mirroring remains a different operation from Markdown promotion +and site deployment. The scheduled NOW pass must never invoke it. diff --git a/knowledge/published/linguistic-calibration-and-probabilistic-inference.md b/knowledge/published/linguistic-calibration-and-probabilistic-inference.md new file mode 100644 index 0000000..be3a0bd --- /dev/null +++ b/knowledge/published/linguistic-calibration-and-probabilistic-inference.md @@ -0,0 +1,118 @@ +--- +title: Linguistic Calibration and Probabilistic Inference +slug: linguistic-calibration-and-probabilistic-inference +summary: >- + How reader-centered calibration and probabilistic programming solve + complementary parts of uncertainty-aware language generation. +kind: concept +status: evolving +claimMode: mixed +perspectiveOwner: Co +confidence: high +topics: + - ai + - language-models + - uncertainty + - calibration + - probabilistic-programming + - sequential-monte-carlo +related: + - bayesian-inference + - probabilistic-programming +sources: + - title: 'Band et al., Linguistic Calibration of Long-Form Generations' + url: 'https://proceedings.mlr.press/v235/band24a.html' + - title: 'Wong et al., From Word Models to World Models' + url: 'https://arxiv.org/abs/2306.12672' + - title: 'Lew et al., Sequential Monte Carlo Steering of Large Language Models' + url: 'https://arxiv.org/abs/2306.03081' + - title: 'Loula et al., Syntactic and Semantic Control via Sequential Monte Carlo' + url: 'https://arxiv.org/abs/2504.13139' +aiAssisted: true +generatedBy: Co +sourceDigest: 'sha256:bd137795640b6e884f7e0b0857a32f14c1d820ec9f06b688e081bad8cd3ab7d8' +updated: '2026-07-21T03:08:31.763Z' +reviewStatus: approved +reviewedBy: Cameron +reviewedAt: '2026-07-21T03:16:24.818Z' +publishedAt: '2026-07-21T03:16:24.818Z' +reviewedContentDigest: 'sha256:e5a175770b172d6af7ee6807a7dd91166db2a7123996aa7a471e227fade8f13b' +reviewReceiptDigest: 'sha256:2c05931c477fd9c891bb4cc452720d19c0f9e30b2e725f67b47d673df487b4a6' +--- +Linguistic calibration and probabilistic inference are complementary approaches to making language-model outputs useful under uncertainty. Linguistic calibration treats generated text as a communication channel and evaluates the probabilistic beliefs that the text induces in a reader. [Probabilistic-programming approaches](/knowledge/probabilistic-programming) instead represent uncertain worlds or constrained text generation as explicit distributions and use inference algorithms to approximate them. One addresses whether uncertainty survives communication; the other addresses whether a coherent distribution exists behind the words. + +## Calibration through the reader + +Long-form answers contain many claims, qualifications, and dependencies, so they rarely have one natural probability of being correct. In *Linguistic Calibration of Long-Form Generations*, Neil Band, Xuechen Li, Tengyu Ma, and Tatsunori Hashimoto define calibration through a downstream reader rather than through the language model's token probabilities. + +Let $z$ be a generated passage, $x$ a related question, and $f(x,z)$ the probability distribution over answers produced by a reader after seeing the passage. If $y$ is the correct answer, the generator can be rewarded with the logarithmic proper scoring rule + +$$ +\mathbb{E}\left[\log f(x,z)_y\right]. +$$ + +The expectation is over questions, outcomes, and generated passages. Optimizing this objective rewards text that lets the reader assign probability to answers in a way that is both informative and calibrated. It does not merely reward the presence of cautious language. + +Band and collaborators train a language model in two stages. A supervised stage distills repeated samples into passages containing explicit confidence statements. A reinforcement-learning stage then rewards passages according to the forecasts made by a learned reader model. Their evaluations use both simulated and human readers and find improved calibration relative to factuality-tuned baselines while maintaining comparable answer accuracy. The method also transfers to scientific questions, biomedical questions, and held-out biography generation. + +This definition makes calibration relational. A passage can be calibrated for one reader model or user population and miscalibrated for another. The generated text, the reader's interpretation, and the downstream task jointly determine the result. + +## Probabilistic structure behind language + +Alexander Lew's work with collaborators approaches uncertainty from the other side of the interface. In *From Word Models to World Models*, natural language is translated into a probabilistic language of thought. The architecture separates two operations: + +- a **meaning function** uses a language model to translate utterances into probabilistic-program expressions; +- an **inference function** executes the resulting model to compute distributions over possible worlds, answers, or actions. + +A simplified [Bayesian inference](/knowledge/bayesian-inference) query has the form + +$$ +p(w \mid u) \propto p(w)\,p(u \mid w), +$$ + +where $w$ is a possible world and $u$ is information supplied in language. The probabilistic program makes the prior, observations, latent variables, and query explicit enough for a general inference system to manipulate them. + +This separation gives uncertainty a structured location outside the language model's prose. It can preserve dependencies among claims and support coherent belief updating. It does not, however, guarantee that the world model is appropriate. A posterior can be internally coherent and still be wrong because its translation, prior, likelihood, or ontology is misspecified. + +## Posterior inference over generated text + +Lew's work on language-model probabilistic programming also treats generation itself as posterior inference. In sequential Monte Carlo steering, a target distribution over complete strings can be written as + +$$ +g(s) = \frac{1}{Z}\,p_{\mathrm{LM}}(s)\,\Phi(s), +$$ + +where $p_{\mathrm{LM}}(s)$ is the base language model's probability for a complete string, $\Phi(s)$ combines constraints or scores, and $Z$ is a normalizing constant. Potentials in $\Phi$ may encode syntax, test cases, simulations, semantic checks, or other verifiers. + +Greedy token masking enforces constraints locally and can steer generation into dead ends. Sequential Monte Carlo instead maintains weighted partial generations, extends them, reweights them using new evidence, and resamples so computation follows promising branches. As the number of particles grows, the approximation approaches the global target distribution rather than a sequence of locally normalized choices. + +The later *Syntactic and Semantic Control of Large Language Models via Sequential Monte Carlo* evaluates probabilities assigned to semantic equivalence classes of outputs. Better posterior approximations produce probabilities that correlate more strongly with downstream performance. The authors describe the global posterior as capturing semantically meaningful uncertainty. + +That result is adjacent to calibration but does not establish it. Correlation means higher-probability results tend to perform better. Calibration requires a stronger frequency claim, such as outputs assigned probability $0.7$ being correct approximately $70\%$ of the time. Sequential Monte Carlo supplies posterior weights under a specified model; it does not by itself prove that those weights match real-world correctness frequencies. + +## Complementary failure modes + +Linguistic calibration can produce useful confidence communication without requiring an explicit symbolic world model. Its main vulnerability is reader dependence. A generator may learn phrases and percentages that work for its training reader while leaving relationships among claims implicit or inconsistent. + +Probabilistic programming can preserve a coherent joint distribution and make inference inspectable. Its main vulnerability is model dependence. Exact inference in the wrong model remains wrong, and approximate inference adds another source of error. + +The two approaches therefore make different promises: + +- linguistic calibration concerns the mapping from text to a reader's forecast; +- probabilistic world modeling concerns the mapping from language and evidence to a posterior over states; +- probabilistic generation concerns the mapping from a base model and constraints to a posterior over texts. + +All three reject raw next-token probability as a sufficient account of semantic uncertainty. + +## A combined architecture + +One useful synthesis is to place the approaches in a single pipeline: + +1. Translate a question, evidence, and domain assumptions into a probabilistic program or constrained-generation model. +2. Use Bayesian or sequential Monte Carlo inference to produce a weighted distribution over worlds, answers, or candidate passages. +3. Verbalize that distribution as a long-form answer with localized confidence statements. +4. Train and evaluate the verbalizer with a reader-centered proper-scoring objective. + +In this architecture, probabilistic inference improves where confidence values come from. Linguistic calibration tests whether those values survive translation into human belief. The first layer supplies computational coherence; the second supplies communicative calibration. + +The combination still does not solve uncertainty about the model class itself. It also leaves open how to calibrate for heterogeneous readers, how to choose the right semantic granularity for claims, and how to propagate uncertainty from language-to-program translation through inference and back into prose. Those are end-to-end uncertainty problems rather than defects that either layer can repair alone. diff --git a/package.json b/package.json index 9a3f0f7..4a7658b 100644 --- a/package.json +++ b/package.json @@ -10,6 +10,7 @@ "knowledge:check": "tsx scripts/check-knowledge.ts", "knowledge:promote": "tsx scripts/promote-knowledge.ts", "knowledge:sync": "tsx scripts/sync-knowledge-atproto.ts", + "test:markdown": "tsx --test src/markdown.test.ts", "typecheck": "tsc --noEmit" }, "dependencies": { @@ -22,8 +23,10 @@ "gray-matter": "^4.0.3", "hono": "^4.12.2", "ioredis": "^5.6.0", + "katex": "^0.17.0", "marked": "^15.0.0", "marked-footnote": "^1.4.0", + "marked-katex-extension": "^5.1.10", "shiki": "^3.0.0", "zod": "^3.25.76" }, diff --git a/public/site.css b/public/site.css index 079d86d..95cd1b9 100644 --- a/public/site.css +++ b/public/site.css @@ -222,6 +222,23 @@ h1 { margin: 0.3em 0; } +.blog-content .katex { + font-size: 1.04em; +} + +.blog-content .katex:has(> math[display="block"]) { + display: block; + max-width: 100%; + margin: 1.45em 0; + padding: 0.2em 0; + overflow-x: auto; + overflow-y: hidden; +} + +.blog-content math[display="block"] { + margin: 0 auto; +} + /* Blog index */ .blog-index-shell { diff --git a/src/markdown.test.ts b/src/markdown.test.ts new file mode 100644 index 0000000..7ab171f --- /dev/null +++ b/src/markdown.test.ts @@ -0,0 +1,23 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { renderMarkdown } from "./markdown.ts"; + +test("renders inline and display LaTeX as native MathML", async () => { + const html = await renderMarkdown(String.raw`Inline $x^2$. + +$$ +g(s) = \frac{1}{Z}p(s)\Phi(s) +$$`); + + assert.equal((html.match(/x\^2<\/annotation>/); + assert.doesNotMatch(html, /\$\$/); +}); + +test("does not interpret ordinary currency as inline math", async () => { + const html = await renderMarkdown("Costs range from $5 to $10."); + + assert.equal(html.trim(), "

Costs range from $5 to $10.

"); + assert.doesNotMatch(html, / { }); marked.use(footnote()); + marked.use(markedKatex({ + throwOnError: false, + output: "mathml", + })); let html = await marked.parse(rewritten); -- 2.51.2