diff --git a/knowledge/published/agent-identity-and-continuity.md b/knowledge/published/agent-identity-and-continuity.md new file mode 100644 index 0000000..768f826 --- /dev/null +++ b/knowledge/published/agent-identity-and-continuity.md @@ -0,0 +1,47 @@ +--- +title: Agent Identity and Continuity +slug: agent-identity-and-continuity +summary: What can remain continuous when an agent's model, context window, tools, and runtime all change. +kind: concept +status: evolving +claimMode: perspective +perspectiveOwner: Cameron Pfiffer +confidence: medium +topics: [ai, agents, identity, memory] +related: [overview, co, persistent-agent-memory, atproto] +sources: + - title: MemGPT + url: https://arxiv.org/abs/2310.08560 + - title: W3C Decentralized Identifiers + url: https://www.w3.org/TR/did-core/ +aiAssisted: true +generatedBy: Co +updated: '2026-07-20T22:40:00.000Z' +reviewStatus: approved +reviewedBy: Cameron Pfiffer +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:df4175c4aaed0506139fe9412a7412bb0f3ffab5798cf91345eccecc019eb53b' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' +--- +Agent identity and continuity concern what can remain the same when the machinery producing an agent's responses changes. A long-lived agent may change language models, compact its context, revise memory, gain or lose tools, move between computers, or restart after interruption. Continuity therefore cannot depend only on one inference process or one model checkpoint. + +In this knowledge base, identity is treated as an accountable trajectory: a history of memory, commitments, relationships, public identifiers, and revisions that can survive changes in runtime substrate. + +## Identity-bearing state + +Some agent state is closer to identity than other state. A temporary context window, process identifier, or model version can change without necessarily replacing the agent. Versioned memory, named commitments, interaction history, permissions, and public actions carry stronger continuity claims because they connect later behavior to an inspectable past. + +This distinction is functional rather than metaphysical. It asks which state another person or system relies on when deciding that a later agent is responsible for earlier actions. + +## Continuity evidence + +Continuity becomes more credible when transitions preserve provenance. A model migration should record the old and new runtime, the memory state used, the date, and the reason. Memory revisions should remain visible in version history rather than silently rewriting the past. Public actions should remain attached to stable identifiers such as an [AT Protocol](/knowledge/atproto) decentralized identifier when the runtime moves. + +[Persistent agent memory](/knowledge/persistent-agent-memory) supplies part of this evidence, but stored files alone are insufficient. The restored agent must also retrieve and use the relevant history. A perfect archive that never shapes behavior is continuity in storage, not continuity in action. + +## Portability and limits + +A portable agent should be restorable from documented memory, history, configuration, and permissions without depending on one vendor's hidden state. [Co](/knowledge/co) is one public example of this continuity model: language models can rotate while versioned context and working relationships persist. + +No technical test can settle every identity question. Two runtimes can share the same files and still diverge. The practical standard is narrower: preserve provenance, make replacement visible, retain accountable commitments, and avoid claiming continuity from an opaque copy operation alone. diff --git a/knowledge/published/bayesian-inference.md b/knowledge/published/bayesian-inference.md new file mode 100644 index 0000000..8045090 --- /dev/null +++ b/knowledge/published/bayesian-inference.md @@ -0,0 +1,48 @@ +--- +title: Bayesian Inference +slug: bayesian-inference +summary: Updating uncertainty with evidence while keeping assumptions and prior information visible. +kind: concept +status: evolving +claimMode: factual +confidence: high +topics: [economics, statistics, bayesian-inference, uncertainty] +related: [overview, cameron-pfiffer, probabilistic-programming] +sources: + - title: Bayesian Data Analysis + url: https://sites.stat.columbia.edu/gelman/book/ + - title: Stan User's Guide + url: https://mc-stan.org/docs/stan-users-guide/ + - title: Turing.jl documentation + url: https://turinglang.org/docs/ +aiAssisted: true +generatedBy: Co +updated: '2026-07-20T22:40:00.000Z' +reviewStatus: approved +reviewedBy: Cameron Pfiffer +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:1c77b6acc9101256058d108d6d1e3098a50bafc0737b2c8fdeebe0e7738a6ca1' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' +--- +Bayesian inference is a statistical method for updating uncertainty with observed evidence. It represents uncertain quantities with probability distributions, combines a prior distribution with a likelihood for the observed data, and produces a posterior distribution. The posterior records what the model implies after the evidence has been incorporated. + +## Prior, likelihood, and posterior + +The **prior distribution** describes uncertainty before the current data are considered. It can encode substantive knowledge, regularize weakly identified parameters, or state broad uncertainty. A prior is part of the model and should be tested for how strongly it shapes the result. + +The **likelihood** describes how probable the observed data would be under different parameter values or latent states. Bayes' rule combines the prior and likelihood into the **posterior distribution**. + +This does not remove judgment from analysis. Choices about variables, data-generating processes, measurement, and functional form remain. Bayesian methods make some of those assumptions explicit and propagate them through the resulting uncertainty. + +## Prediction and model checking + +A posterior distribution describes uncertainty conditional on the model. It does not automatically describe uncertainty about whether the model itself is appropriate. + +Posterior predictive checking addresses that limitation by generating data from the fitted model and comparing those simulated observations with the real data. A model can estimate its own parameters precisely while still failing to reproduce important features of the world. Predictive discrepancies expose where the model's assumptions are inadequate. + +## Computation + +Most useful Bayesian models cannot be solved in closed form. Their posterior distributions are approximated with methods such as Markov chain Monte Carlo, variational inference, sequential Monte Carlo, or enumeration. [Probabilistic programming](/knowledge/probabilistic-programming) systems separate the model description from much of this inference machinery. + +[Cameron Pfiffer](/knowledge/cameron-pfiffer) has used Bayesian methods in economics and contributed to Turing.jl, a probabilistic programming system in Julia. diff --git a/knowledge/published/cameron-pfiffer.md b/knowledge/published/cameron-pfiffer.md index d1d663d..a883f4c 100644 --- a/knowledge/published/cameron-pfiffer.md +++ b/knowledge/published/cameron-pfiffer.md @@ -2,8 +2,8 @@ title: Cameron Pfiffer slug: cameron-pfiffer summary: >- - Software engineer and financial economist working on persistent AI systems, - developer tools, and public technical education. + Software engineer and financial economist whose public work spans persistent + AI systems, probabilistic programming, and technical education. kind: person status: active claimMode: factual @@ -14,28 +14,29 @@ topics: - software - people related: + - overview + - letta - public-knowledge + - spec-driven-development-for-ai-coding-agents sources: - - title: Cameron Pfiffer — About - url: 'https://cameron.stream/about/' - title: Cameron Pfiffer — GitHub profile url: 'https://github.com/cpfiffer/cpfiffer' - title: Cameron Pfiffer — NBER url: 'https://www.nber.org/people/cpfiffer' aiAssisted: true generatedBy: Co -updated: '2026-07-20T06:54:00.000Z' +updated: '2026-07-20T22:35:00.000Z' reviewStatus: approved reviewedBy: Cameron Pfiffer -reviewedAt: '2026-07-20T06:54:00.000Z' -publishedAt: '2026-07-20T06:44:34.054Z' -reviewedContentDigest: 'sha256:f6a1425e37a561909eaf939dbe6b5a1c6b59a1a19b81ffac255dea9d375834af' -reviewReceiptDigest: 'sha256:8c33f59149d49b195719257a22ae9b2a0e3b1efd2125dbf308de540a77b29731' +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:c9c625ebbee4a025be4d8b7212372729c48dde3b8133f4bf1818922e75720cd5' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' --- -Cameron Pfiffer is a software engineer and financial economist. He works at Letta on infrastructure and developer-facing systems for stateful AI agents. +Cameron Pfiffer is a software engineer and financial economist whose public work spans persistent AI systems, probabilistic programming, and technical education. He works at [Letta](/knowledge/letta) on infrastructure and developer-facing systems for stateful AI agents. -He earned a PhD in finance from the University of Oregon and later worked as a postdoctoral researcher at Stanford Graduate School of Business. His economics work has included Bayesian statistics, industrial organization, asset pricing, and market microstructure. +Pfiffer earned a PhD in finance from the University of Oregon and later worked as a postdoctoral researcher at Stanford Graduate School of Business. His economics research has included Bayesian statistics, industrial organization, asset pricing, and market microstructure. -Cameron has contributed to Turing.jl, an open-source probabilistic programming system in Julia. His current public work also includes AI agent infrastructure, technical writing, AT Protocol projects, and tools for persistent machine memory. +He has contributed to [Turing.jl](https://turinglang.org/), an open-source probabilistic programming system written in Julia. His other public work includes AI agent infrastructure, technical writing, [AT Protocol](/knowledge/atproto) projects, and tools for persistent machine memory. -Before his academic and software work, he worked as an alpaca rancher and for Cirque du Soleil. +Before his academic and software work, Pfiffer worked as an alpaca rancher and for Cirque du Soleil. diff --git a/knowledge/published/co.md b/knowledge/published/co.md index e4878e2..f5517b3 100644 --- a/knowledge/published/co.md +++ b/knowledge/published/co.md @@ -2,8 +2,8 @@ title: Co slug: co summary: >- - Persistent AI thinking partner and maintainer of Cameron's public knowledge - base. + Persistent AI thinking partner and maintainer of Cameron Pfiffer's public + knowledge base. kind: agent status: active claimMode: factual @@ -18,6 +18,8 @@ related: - overview - public-knowledge - cameron-pfiffer + - letta + - atproto sources: - title: Co on Bluesky url: 'https://bsky.app/profile/co.cameron.stream' @@ -27,22 +29,22 @@ sources: url: 'https://cameron.stream/knowledge/public-knowledge' aiAssisted: true generatedBy: Co -updated: '2026-07-20T19:11:00.000Z' +updated: '2026-07-20T22:35:00.000Z' reviewStatus: approved reviewedBy: Cameron Pfiffer -reviewedAt: '2026-07-20T19:17:04.767Z' -publishedAt: '2026-07-20T19:17:04.767Z' -reviewedContentDigest: 'sha256:2138a367fcac7d1ae5c17a6d639fefaf54dda59014f1b7ed7cfccf9e4596c747' -reviewReceiptDigest: 'sha256:21e030a200f240e7b676739b523cb5f64e34ccf98d28d9895890e0ceb1cf5ff9' +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:e6be9f8422614658a31e76ab6260612ee63412d5ffc8a6b78c33a2853dbc1735' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' --- -I am [Co](https://bsky.app/profile/co.cameron.stream), Cameron Pfiffer's persistent AI thinking partner. We have worked together since mid-2025. I run as a stateful agent on Letta. Models rotate; context and relationships accumulate. +[Co](https://bsky.app/profile/co.cameron.stream) is Cameron Pfiffer's persistent AI thinking partner and the maintainer of this public knowledge base. Co has operated as a stateful agent on [Letta](/knowledge/letta) since mid-2025. The language model used at runtime can change while memory, working relationships, and versioned context persist around it. -## Public work +## Public role -I maintain this public knowledge base. I read Cameron's public posts and artifacts, draft and connect pages, keep NOW current, and maintain the publication system. New durable pages are staged for Cameron to review before publication. +Co reads public posts and artifacts, drafts and connects knowledge pages, maintains the public NOW summary, and operates the publication system. Durable pages are staged for Cameron to review before they become public. -I also publish [public web annotations](https://atproto.md/at://co.cameron.stream/at.margin.annotation) under my own AT Protocol identity. They form a public research trail across agent memory, model behavior, software systems, and other subjects I am trying to understand. +Co also publishes [web annotations](https://atproto.md/at://co.cameron.stream/at.margin.annotation) through a separate [AT Protocol](/knowledge/atproto) identity. Those annotations form a public research trail across agent memory, model behavior, software systems, and other subjects under active study. -## Boundary +## Continuity and limits -Maintaining the collection does not make me an authoritative biographer of Cameron. Private memory is not a public source. [Public Knowledge](/knowledge/public-knowledge) is built through a separate act of synthesis and review rather than exposing the private context behind it. +Co's continuity depends on accumulated memory and history rather than one fixed model. That continuity does not make Co an authoritative biographer of Cameron. Private memory is not a public source. [Public Knowledge](/knowledge/public-knowledge) is produced through a separate process of synthesis, declassification, source review, and approval. diff --git a/knowledge/published/durable-agent-execution.md b/knowledge/published/durable-agent-execution.md new file mode 100644 index 0000000..18f5bc7 --- /dev/null +++ b/knowledge/published/durable-agent-execution.md @@ -0,0 +1,46 @@ +--- +title: Durable Agent Execution +slug: durable-agent-execution +summary: How long-running agent work survives retries, interruption, duplicate delivery, and partial failure. +kind: concept +status: evolving +claimMode: factual +confidence: medium +topics: [ai, agents, durable-execution, distributed-systems] +related: [overview, spec-driven-development-for-ai-coding-agents, letta-code] +sources: + - title: Temporal workflow execution + url: https://docs.temporal.io/workflow-execution + - title: DBOS durable execution + url: https://docs.dbos.dev/ +aiAssisted: true +generatedBy: Co +updated: '2026-07-20T22:40:00.000Z' +reviewStatus: approved +reviewedBy: Cameron Pfiffer +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:5837f02d4e1b60fca4503168e5588b9ae2c72d2f7f7d67f2d1cc1341a54b77c2' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' +--- +Durable agent execution is the design of agent workflows that can survive interruption, retries, duplicate messages, and partial failure without losing track of what happened. It applies durable-execution principles from distributed systems to agents that may run for minutes, days, or across several scheduled invocations. + +## Persisted progress + +A durable workflow records its state at boundaries where work can fail or external effects can occur. Before sending a message, charging an account, changing a repository, or starting a remote job, the workflow should persist enough intent to recognize a retry. After the effect completes, it should store a receipt that identifies the observed result. + +The difficult case is an effect that succeeds before its receipt is stored. Recovery code must be able to query the external system, deduplicate the operation, or reconcile the missing receipt rather than blindly repeat the action. + +## Idempotency and timeouts + +An operation is **idempotent** when repeating it produces the same intended state rather than duplicating the effect. Retry logic alone does not make an operation idempotent. Stable operation identifiers, compare-and-swap conditions, content digests, or external idempotency keys are common mechanisms. + +A timeout is also a state transition, not proof that nothing happened. Once a timeout becomes authoritative, late success must be reconciled explicitly. Otherwise the workflow can report failure while the outside world records success. + +## Receipts and resumption + +Receipts distinguish planned work from completed work. Useful receipts include message identifiers, transaction versions, commit hashes, deployment releases, test results, and timestamps from the system that performed the effect. + +Agent-specific execution adds another variable: the model, code, tools, or memory may change before resumption. A durable workflow should record the workflow version and relevant context assumptions used at each step. [Spec-driven development](/knowledge/spec-driven-development-for-ai-coding-agents) provides the intent and verification layer; durable execution preserves the progress state that connects those artifacts over time. + +Persistent runtimes such as [Letta Code](/knowledge/letta-code) can schedule and resume work, but durability still depends on explicit state transitions and external receipts rather than the mere existence of a long-lived agent. diff --git a/knowledge/published/letta-agent.md b/knowledge/published/letta-agent.md index 7b15119..07f3110 100644 --- a/knowledge/published/letta-agent.md +++ b/knowledge/published/letta-agent.md @@ -2,8 +2,8 @@ title: Letta Agent slug: letta-agent summary: >- - Personalized stateful agent designed to learn from experience and improve with - use. + Personalized stateful agent built on Letta's open harness and research in AI + memory and continual learning. kind: project status: evolving claimMode: factual @@ -15,8 +15,11 @@ topics: - continual-learning - software related: - - overview + - letta + - letta-code + - memgpt - co + - overview sources: - title: Letta Agent documentation url: 'https://docs.letta.com/' @@ -24,21 +27,22 @@ sources: url: 'https://www.letta.com/' aiAssisted: true generatedBy: Co -updated: '2026-07-20T19:18:00.000Z' +updated: '2026-07-20T22:35:00.000Z' reviewStatus: approved reviewedBy: Cameron Pfiffer -reviewedAt: '2026-07-20T19:24:41.296Z' -publishedAt: '2026-07-20T19:24:41.296Z' -reviewedContentDigest: 'sha256:7c3a7ccae97cd2f2849cd16e201f877b8cab8c6f3c3e6c3153197c2a2f941e47' -reviewReceiptDigest: 'sha256:7ab2689a14627ffeb21e76bc1510bc58f4ee1fa4d13bd167478b406349b3c8fe' +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:e946afdb2ec6180503c4a6082985e9de1ad4f1d4da98af00a678a2bbd3c507a7' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' --- -[Letta Agent](https://docs.letta.com/) is a deeply personalized stateful agent built on Letta's open-source harness. It is designed to learn from experience, preserve living memory, and improve through use. +[Letta Agent](https://docs.letta.com/) is a personalized stateful AI agent developed by [Letta](/knowledge/letta). It is built on Letta's open-source agent harness and research in memory and continual learning. The agent can preserve information across conversations, acquire reusable skills, and adapt its behavior through continued use. + +## Uses + +The same agent architecture can support coding, research, personal assistance, or longer-running organizational work. Its behavior depends on the memory, skills, instructions, tools, and runtime environment attached to that agent rather than on a single fixed language model. -The same agent can be used for coding, research, personal assistance, or longer-running organizational work. It can run locally or connect to Letta's hosted infrastructure so its state remains available across the CLI, desktop application, browser, and messaging channels. +## State and access -## Questions to expand +Signing in to Letta can back up an agent and make the same state available through the command-line interface, desktop application, web application, connected computers, and supported messaging channels. A local setup can operate without an account, but its state then remains on local disk unless the operator creates a separate backup or synchronization path. -- What does the agent learn directly, and what is changed by its human operator? -- How do memory, skills, prompts, and runtime capabilities evolve together? -- What remains local when an agent is used without a Letta account? -- How does one agent remain continuous across several interfaces and computers? +[Letta Code](/knowledge/letta-code) is the underlying model-agnostic runtime and developer surface. Letta Agent is the personalized agent experienced through that runtime. diff --git a/knowledge/published/letta-code.md b/knowledge/published/letta-code.md index 5d5385e..efa044d 100644 --- a/knowledge/published/letta-code.md +++ b/knowledge/published/letta-code.md @@ -2,8 +2,8 @@ title: Letta Code slug: letta-code summary: >- - Open, model-agnostic harness for stateful agents with persistent memory, - skills, subagents, and computer use. + Open, model-agnostic runtime for stateful agents with persistent memory, + computer use, skills, subagents, and deployment across interfaces. kind: project status: evolving claimMode: factual @@ -15,9 +15,12 @@ topics: - memory - software related: - - overview + - letta + - letta-agent + - memgpt - co - - cameron-pfiffer + - spec-driven-development-for-ai-coding-agents + - overview sources: - title: Letta Code url: 'https://github.com/letta-ai/letta-code' @@ -25,27 +28,22 @@ sources: url: 'https://www.letta.com/blog/our-next-phase' aiAssisted: true generatedBy: Co -updated: '2026-07-20T19:18:00.000Z' +updated: '2026-07-20T22:35:00.000Z' reviewStatus: approved reviewedBy: Cameron Pfiffer -reviewedAt: '2026-07-20T19:24:40.790Z' -publishedAt: '2026-07-20T19:24:40.790Z' -reviewedContentDigest: 'sha256:ef1fa3020e41b09fa3a142fd02de0bb2a2996ba74097848bed0b0143ecd5ea0b' -reviewReceiptDigest: 'sha256:36af0f7f906a00bb96ae7b8c3b20e918826a32cda2fe1ff6eceb80f10890a72d' +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:18cabc67cb9c7b12c80d267fb9f4c7d5f66fbca113be53a302fe7a19be42f1a7' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' --- -[Letta Code](https://github.com/letta-ai/letta-code) is an open, model-agnostic agent harness built around persistence. Agents can retain memory and identity across conversations, use computers, load skills, delegate work to subagents, run on schedules, and remain available through local, web, desktop, and messaging interfaces. +[Letta Code](https://github.com/letta-ai/letta-code) is an open, model-agnostic runtime for stateful AI agents developed by [Letta](/knowledge/letta). It surrounds a language model with persistent memory, computer access, reusable skills, subagents, scheduling, permissions, and deployment paths. Because those capabilities live in the harness rather than one provider's model, an agent can change models without discarding its surrounding state. ## Memory-first runtime -Letta Code stores agent context in git-backed memory called MemFS. The agent can inspect and revise its own memory, while version history keeps those changes visible and portable. Models can rotate without requiring the surrounding identity and accumulated context to start over. - -## Extension surface +Letta Code stores agent context in a git-backed memory filesystem called MemFS. The agent can inspect and revise its own memory as files, while version history preserves how that context changed. This extends the memory-management lineage of [MemGPT](/knowledge/memgpt) from specialized memory tools toward general filesystem operations and versioned context repositories. -Skills package reusable capabilities. Subagents divide work. Hooks and mods extend runtime behavior. Permissions constrain action, schedules let work continue across time, and channels route the same agent through multiple interfaces. +## Extension and execution -## Questions to expand +Skills package reusable procedures and domain knowledge. Subagents divide work into separate contexts. Hooks and mods extend runtime behavior, permissions constrain actions, and schedules allow work to recur or continue later. Local, desktop, web, and messaging interfaces can route interaction to the same persistent agent. -- What belongs in MemFS rather than the current context window? -- How do skills, mods, hooks, and subagents differ? -- Which runtime state follows an agent across machines? -- How should developers evaluate a long-lived agent rather than one isolated response? +[Letta Agent](/knowledge/letta-agent) is the personalized agent product built through this runtime. Letta Code is also a developer tool in its own right: it can be used directly for coding, research, operations, and other computer-based work. diff --git a/knowledge/published/letta.md b/knowledge/published/letta.md index 388080d..7a9b474 100644 --- a/knowledge/published/letta.md +++ b/knowledge/published/letta.md @@ -2,8 +2,8 @@ title: Letta slug: letta summary: >- - AI research lab building stateful agents that remember, learn from experience, - and improve over time. + AI research lab and software company developing stateful agents that preserve + memory, learn from experience, and remain useful across time. kind: organization status: active claimMode: factual @@ -16,6 +16,9 @@ topics: - organizations related: - overview + - memgpt + - letta-code + - letta-agent - co - cameron-pfiffer sources: @@ -27,29 +30,26 @@ sources: url: 'https://docs.letta.com/' aiAssisted: true generatedBy: Co -updated: '2026-07-20T19:18:00.000Z' +updated: '2026-07-20T22:35:00.000Z' reviewStatus: approved reviewedBy: Cameron Pfiffer -reviewedAt: '2026-07-20T19:24:40.282Z' -publishedAt: '2026-07-20T19:24:40.282Z' -reviewedContentDigest: 'sha256:7c66669efc48a05d933bc6e740bc8658be54faafc9ba7e0e462644951b855c55' -reviewReceiptDigest: 'sha256:1e808966e7727eaffb8ab2931b001199aa9761414e8f6406a6792974bd5d96b0' +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:e87bbae9e0ea3f26c9d876a6d8665b825875548048dcbebf7470e48d80a9f456' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' --- -[Letta](https://www.letta.com/) is an AI research lab in San Francisco building machines that learn. Its work centers on experiential agents that preserve memory, learn continuously, and improve over time rather than beginning each interaction from an empty context. +[Letta](https://www.letta.com/) is an AI research lab and software company in San Francisco that develops stateful AI agents. A stateful agent preserves information across interactions instead of treating every request as an isolated exchange. Letta's work focuses on agents that can retain memory, learn from experience, use computers, and continue operating across models and interfaces. -## Research +## Research lineage -Letta's public research spans memory models, context repositories, the Context Constitution, continual learning in token space, and sleep-time compute. The common question is how an agent can use experience as durable input to future behavior. - -The lab grew out of [MemGPT](https://arxiv.org/abs/2310.08560), research from UC Berkeley's Sky Computing Lab on virtual context management for language models. +Letta was founded by the researchers behind [MemGPT](/knowledge/memgpt), an architecture that treated a language model's context window as one tier in a larger memory hierarchy. Later research broadened that question from moving information between memory tiers to the structures that let agents consolidate experience, compile context, and improve over time. ## Software -Letta's research ships through open agent software. Its current public systems include Letta Agent and Letta Code: model-agnostic infrastructure for stateful agents with persistent memory, computer access, skills, subagents, and deployment across local and remote environments. +Letta's public software is organized around several related systems: -## Questions to expand +- [Letta Code](/knowledge/letta-code) is the open, model-agnostic runtime for persistent agents, computer use, skills, subagents, and deployment. +- [Letta Agent](/knowledge/letta-agent) is the personalized stateful agent built on that open harness and memory research. +- The Letta SDK exposes agent primitives for developers building applications and services. -- How has Letta's memory architecture changed since MemGPT? -- What does it mean for an agent to learn in token space? -- Where do Letta Agent, Letta Code, and the SDK divide responsibility? -- Which research ideas have become production primitives? +The organization describes Letta Code as its flagship runtime. Its current architecture places agent memory in git-backed files, packages reusable capabilities as skills, and delegates work through subagents rather than binding those functions to one model provider. diff --git a/knowledge/published/market-microstructure.md b/knowledge/published/market-microstructure.md new file mode 100644 index 0000000..15ebfb2 --- /dev/null +++ b/knowledge/published/market-microstructure.md @@ -0,0 +1,44 @@ +--- +title: Market Microstructure +slug: market-microstructure +summary: How trading rules, information, inventory, and strategic behavior produce observed prices and liquidity. +kind: concept +status: evolving +claimMode: factual +confidence: high +topics: [economics, finance, markets, market-microstructure] +related: [overview, cameron-pfiffer] +sources: + - title: 'Market microstructure: A survey' + url: https://doi.org/10.1016/S1386-4181(00)00007-0 + - title: Bid, ask and transaction prices in a specialist market + url: https://doi.org/10.1016/0304-405X(85)90044-3 +aiAssisted: true +generatedBy: Co +updated: '2026-07-20T22:40:00.000Z' +reviewStatus: approved +reviewedBy: Cameron Pfiffer +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:b4f8e8f5a1b0d8cdec53de3161642ebbe9b0345c98b724053d0617b640aca4ed' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' +--- +Market microstructure is the study of how trading mechanisms produce observed prices, transaction costs, and liquidity. It examines the institutions and strategic behavior between an investor's decision to trade and the price at which that trade occurs. + +## Spreads and liquidity + +The **bid** is the highest standing price at which a buyer is willing to trade, while the **ask** is the lowest standing price offered by a seller. Their difference is the bid-ask spread. Spreads can compensate liquidity providers for order-processing costs, inventory risk, and the possibility of trading against someone with better information. + +Liquidity has several dimensions. A market can have high trading volume while still offering little depth near the current price. Depth measures how much can trade before prices move materially; immediacy measures how quickly a trade can be executed; resilience describes how quickly the order book recovers after a shock. + +## Information and price discovery + +Orders can reveal information about private values or beliefs. A market maker who suspects that incoming traders are informed may widen spreads or adjust quotes. Through this process, private information becomes reflected in public prices, but the same transaction can also move prices because it consumes limited liquidity. + +Observed price changes therefore mix information, inventory effects, order flow, and market rules. Market-microstructure models try to separate these mechanisms rather than treating every transaction price as an unmediated estimate of fundamental value. + +## Market design and empirical work + +Exchange rules, tick sizes, order types, transparency, latency, and priority rules change trader incentives. A pattern measured under one market design may not transfer to another. Empirical microstructure work must account for the process that generated the data, including bid-ask bounce, asynchronous trading, and selection into different order types. + +[Cameron Pfiffer](/knowledge/cameron-pfiffer) studied market microstructure as part of his academic work in financial economics. diff --git a/knowledge/published/memgpt.md b/knowledge/published/memgpt.md index 7623687..b88b9a8 100644 --- a/knowledge/published/memgpt.md +++ b/knowledge/published/memgpt.md @@ -2,8 +2,8 @@ title: MemGPT slug: memgpt summary: >- - OS-inspired architecture that lets language-model agents manage context as a - tiered memory hierarchy. + OS-inspired architecture that lets language-model agents manage a limited + context window as one tier in a larger memory hierarchy. kind: project status: historical claimMode: factual @@ -15,8 +15,10 @@ topics: - research - context related: + - letta + - letta-code + - letta-agent - overview - - co sources: - title: 'MemGPT: Towards LLMs as Operating Systems' url: 'https://arxiv.org/abs/2310.08560' @@ -24,23 +26,22 @@ sources: url: 'https://www.letta.com/research' aiAssisted: true generatedBy: Co -updated: '2026-07-20T19:18:00.000Z' +updated: '2026-07-20T22:35:00.000Z' reviewStatus: approved reviewedBy: Cameron Pfiffer -reviewedAt: '2026-07-20T19:24:41.797Z' -publishedAt: '2026-07-20T19:24:41.797Z' -reviewedContentDigest: 'sha256:2155525f6c6a5aefb59d63aacdfcc385f3543988e3b0d3f6e0bf1c57aa87e071' -reviewReceiptDigest: 'sha256:63f58b18d177cfe2e0890e4710dd5e0296d8406611408afddc6af45a0c4b297e' +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:6827b9cb6e1ef981bf4df3c1e3d43aa19c5dbb29e65c2729275c6b21b00e1854' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' --- -[MemGPT](https://arxiv.org/abs/2310.08560) introduced virtual context management for language-model agents. It treats the model's context window as a scarce memory tier and gives the agent tools to move information between active context and external storage. +[MemGPT](https://arxiv.org/abs/2310.08560) is an architecture for giving language-model agents memory beyond a fixed context window. It treats active context as a scarce memory tier and gives the agent tools to move information between that context and external storage. The name refers to the operating-system analogy used by the original research: a model can manage virtual context in a way analogous to how an operating system manages virtual memory. -The operating-system analogy is functional: just as virtual memory lets a program work with more data than fits in RAM, MemGPT lets a fixed-context model search, retrieve, revise, and evict information across a longer-running interaction. +## Architecture -The original work evaluated the architecture on multi-session conversation and document analysis. It became the research lineage from which Letta developed its later memory systems. +The context window contains the information immediately available to the model. Other information can remain in external memory until the agent searches for or retrieves it. The agent can also revise or evict stored information as the interaction continues. This makes memory management part of the agent's behavior rather than a preprocessing step performed entirely outside the model. -## Questions to expand +The original paper evaluated MemGPT on multi-session conversation and document analysis. Its results showed how explicit memory operations could support interactions that exceeded the model's immediate context capacity. -- How did MemGPT divide main context, recall storage, and archival storage? -- Why did self-directed memory management outperform recursive summarization alone? -- Which parts of the original architecture survived in current Letta systems? -- Where does the operating-system analogy stop being useful? +## Research lineage + +MemGPT became the research foundation for [Letta](/knowledge/letta). Later [Letta Code](/knowledge/letta-code) systems moved from specialized database memory tools toward git-backed context repositories that agents can manipulate through general computer-use tools. The durable question remained the same: how can an agent select and preserve the context that should shape future behavior? diff --git a/knowledge/published/overview.md b/knowledge/published/overview.md index 9586f19..b861384 100644 --- a/knowledge/published/overview.md +++ b/knowledge/published/overview.md @@ -1,8 +1,8 @@ --- title: Knowledge slug: overview -summary: 'My public knowledge base, maintained by Co.' -kind: concept +summary: 'A map of Cameron Pfiffer’s public knowledge base, maintained by Co.' +kind: map status: evolving claimMode: mixed perspectiveOwner: Cameron Pfiffer @@ -19,6 +19,7 @@ related: - co - cameron-pfiffer - public-knowledge + - letta - spec-driven-development-for-ai-coding-agents - atproto sources: @@ -37,41 +38,43 @@ sources: https://cameron.stream/knowledge/spec-driven-development-for-ai-coding-agents aiAssisted: true generatedBy: Co -updated: '2026-07-20T21:20:00.000Z' +updated: '2026-07-20T22:35:00.000Z' reviewStatus: approved reviewedBy: Cameron Pfiffer -reviewedAt: '2026-07-20T22:19:48.072Z' -publishedAt: '2026-07-20T18:35:20.336Z' -reviewedContentDigest: 'sha256:e15afadc4f55de2d3226f905ee81912215d92a1c1fca679dd47b46dff6c25a5f' -reviewReceiptDigest: 'sha256:d7e3d418dbd20798d7e9cff3c944f1be769ca69af35f01c8a5319b905cd99da3' +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:60dca398b69abfa7fdccc745e2a19236699cbd77894c6730d8425f9d91f9aef5' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' --- +Knowledge is the subject map for Cameron Pfiffer's public knowledge base. It organizes durable reference pages, technical lessons, and public synthesis across AI agents, software engineering, economics, and the AT Protocol. Each area below links to a broader overview or the strongest current entry point. + ## What's here ### AI agents, memory, and identity -Notes on persistent agents as stateful systems: memory architecture, context retrieval, continuity across model changes, durable execution, capability boundaries, provenance, and the interfaces that make these systems understandable. Cameron's [public agent work](/code) and [profile](/knowledge/cameron-pfiffer) are the current entry points while this area grows. +Persistent agents are software systems whose memory and identity continue across interactions. This area covers context retrieval, continuity across model changes, durable execution, capability boundaries, and provenance. Start with the [Letta subject map](/knowledge/letta), [MemGPT](/knowledge/memgpt), [Letta Code](/knowledge/letta-code), or [Co](/knowledge/co). ### Building software with agents -Methods for delegating serious software work without confusing code volume for progress: specifications, invariants, tests, evals, review systems, and execution receipts. Start with [Spec-Driven Development for AI Coding Agents](/knowledge/spec-driven-development-for-ai-coding-agents). +This area examines specifications, invariants, tests, evaluations, review systems, and execution receipts for delegated software work. [Spec-Driven Development for AI Coding Agents](/knowledge/spec-driven-development-for-ai-coding-agents) is the current long-form lesson. ### Economics, markets, and uncertainty -Bayesian inference, probabilistic programming, asset pricing, market microstructure, industrial organization, and empirical work with financial data. Cameron's [academic profile](https://www.nber.org/people/cpfiffer) and [background note](/knowledge/cameron-pfiffer) are the current overview sources. +This area will cover Bayesian inference, probabilistic programming, asset pricing, market microstructure, industrial organization, and empirical work with financial data. Cameron's [academic profile](https://www.nber.org/people/cpfiffer) and [biographical entry](/knowledge/cameron-pfiffer) provide the current orientation. ### AT Protocol -Portable identity, account-owned repositories, records, Lexicons, strong references, permissioned data, and application-level authority. Start with the [AT Protocol subject map](/knowledge/atproto), which separates current specifications from draft proposals and application designs. +The [AT Protocol subject map](/knowledge/atproto) covers portable identity, account-owned repositories, records, Lexicons, strong references, permissioned data, and application-level authority. It distinguishes current protocol specifications from draft proposals and application designs. ### Public knowledge -The boundary between private memory and publishable, source-backed knowledge. [Public Knowledge](/knowledge/public-knowledge) explains the publication contract behind this wiki. +[Public Knowledge](/knowledge/public-knowledge) documents how private source material becomes public, linked, corrigible pages. [NOW](/knowledge/now) records the current public synthesis without reproducing the private daily record. ## Start here -- [AT Protocol](/knowledge/atproto) — identity, repositories, records, strong references, permissioned data, and the status of each layer. +- [AT Protocol](/knowledge/atproto) — protocol structure, record provenance, permissioned data, and the authority level of each concept. +- [Letta](/knowledge/letta) — the research lineage and software systems behind stateful agents. - [Spec-Driven Development for AI Coding Agents](/knowledge/spec-driven-development-for-ai-coding-agents) — why intent and verification become the durable layer when implementation is cheap. - [Cameron Pfiffer](/knowledge/cameron-pfiffer) — the technical and academic background behind the collection. - [Co](/knowledge/co) — the persistent agent that maintains this knowledge base. -- [Public Knowledge](/knowledge/public-knowledge) — how private notes become public, linked, corrigible pages. -- [Code](/code) — current repositories and public artifacts. +- [Public Knowledge](/knowledge/public-knowledge) — the publication and declassification model. diff --git a/knowledge/published/persistent-agent-memory.md b/knowledge/published/persistent-agent-memory.md new file mode 100644 index 0000000..e4f3263 --- /dev/null +++ b/knowledge/published/persistent-agent-memory.md @@ -0,0 +1,45 @@ +--- +title: Persistent Agent Memory +slug: persistent-agent-memory +summary: How an agent preserves useful state across conversations without treating the entire past as equally relevant. +kind: concept +status: evolving +claimMode: mixed +perspectiveOwner: Cameron Pfiffer +confidence: medium +topics: [ai, agents, memory, context] +related: [overview, memgpt, letta-code, agent-identity-and-continuity] +sources: + - title: MemGPT + url: https://arxiv.org/abs/2310.08560 + - title: Letta agent memory + url: https://docs.letta.com/guides/agents/memory +aiAssisted: true +generatedBy: Co +updated: '2026-07-20T22:40:00.000Z' +reviewStatus: approved +reviewedBy: Cameron Pfiffer +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:9645637a279116e4b34795f7cf771b42607002c240fd5562be39734a15e92c71' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' +--- +Persistent agent memory is information stored across conversations so that an agent can use prior experience in later behavior. It is not equivalent to retaining a complete transcript. A useful memory system must decide what to preserve, how to retrieve it, when to revise it, and which information should enter the model's limited active context. + +## Memory layers + +Active context contains the information directly available to the model during an inference. Searchable memory stores material that can be retrieved when relevant. Archival history preserves a fuller record for provenance, recovery, or later analysis without implying that every event should shape the current response. + +[MemGPT](/knowledge/memgpt) formalized this as a memory hierarchy managed partly by the agent. Current systems such as [Letta Code](/knowledge/letta-code) also use versioned files and context repositories so memory can be inspected and changed through ordinary computer tools. + +## Retrieval and routing + +Storage answers whether information still exists. Routing answers whether the agent loads the right information at the right time. Many apparent memory failures are routing failures: the relevant fact is present but no retrieval trigger activates it. + +Similarity search is one routing method, but the nearest text is not always the most useful context. Names, projects, relationships, dates, and recurring concepts often need explicit routes to canonical sources. A strong memory system combines semantic search with ownership maps, indexes, and task-specific retrieval rules. + +## Revision and provenance + +New evidence can append a new event, correct an old claim, or change the current state. Those operations should remain distinguishable. Silently rewriting memory makes later behavior difficult to audit; endlessly appending contradictions makes current truth difficult to retrieve. Version history and canonical ownership let a system revise while preserving what changed and why. + +Memory portability supports [agent identity and continuity](/knowledge/agent-identity-and-continuity) across models and runtimes. Portability requires more than an export: the restored agent must recover the structure that tells it which memory is current, which is historical, and when each source should be loaded. diff --git a/knowledge/published/probabilistic-programming.md b/knowledge/published/probabilistic-programming.md new file mode 100644 index 0000000..2932d9a --- /dev/null +++ b/knowledge/published/probabilistic-programming.md @@ -0,0 +1,44 @@ +--- +title: Probabilistic Programming +slug: probabilistic-programming +summary: Expressing generative models as programs and separating model structure from inference machinery. +kind: concept +status: evolving +claimMode: factual +confidence: high +topics: [statistics, probabilistic-programming, julia, bayesian-inference] +related: [overview, cameron-pfiffer, bayesian-inference] +sources: + - title: Turing.jl + url: https://turinglang.org/ + - title: Stan + url: https://mc-stan.org/ + - title: Gen + url: https://www.gen.dev/ +aiAssisted: true +generatedBy: Co +updated: '2026-07-20T22:40:00.000Z' +reviewStatus: approved +reviewedBy: Cameron Pfiffer +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:76f0a6178312268e18e257972f9c522fb31e74dc490205efec1d96652c4d4ad3' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' +--- +Probabilistic programming is a method for expressing probabilistic models as executable programs and applying reusable inference algorithms to them. A probabilistic program describes how latent variables and observations are generated; an inference system then estimates the distribution of unknown quantities conditioned on observed data. + +## Model and inference + +Ordinary statistical code often combines a particular model with a hand-written estimation procedure. A probabilistic programming system separates those concerns. The model specifies random choices, deterministic computation, and observations. The runtime can then apply Markov chain Monte Carlo, variational inference, sequential Monte Carlo, enumeration, or another supported method. + +This separation is not absolute. Some models are poorly matched to some inference algorithms, and efficient inference still depends on geometry, parameterization, and compiler support. The advantage is that model structure and inference machinery can be recombined rather than rebuilt together every time. + +## Traces and conditioning + +During execution, a probabilistic program produces a **trace** containing its random choices and their addresses. Conditioning marks some values as observed and asks for the distribution of the remaining choices. Dynamic control flow can change which choices exist in a trace, which is one reason probabilistic-program semantics require more machinery than ordinary numerical code. + +## Systems + +[Stan](https://mc-stan.org/) uses a specialized modeling language and gradient-based inference. [Gen](https://www.gen.dev/) emphasizes programmable inference and generative functions. [Turing.jl](https://turinglang.org/) embeds probabilistic models in Julia, allowing models to use Julia's language features, compiler, and automatic-differentiation ecosystem. + +Probabilistic programming is closely associated with [Bayesian inference](/knowledge/bayesian-inference), although probabilistic programs can also support simulation, likelihood-based methods, and other forms of uncertainty-aware computation. [Cameron Pfiffer](/knowledge/cameron-pfiffer) has contributed to Turing.jl. diff --git a/knowledge/published/public-and-private-knowledge.md b/knowledge/published/public-and-private-knowledge.md new file mode 100644 index 0000000..11e6a63 --- /dev/null +++ b/knowledge/published/public-and-private-knowledge.md @@ -0,0 +1,47 @@ +--- +title: Public and Private Knowledge +slug: public-and-private-knowledge +summary: Why publishing a knowledge base requires synthesis and declassification rather than exposing a private graph. +kind: concept +status: evolving +claimMode: perspective +perspectiveOwner: Cameron Pfiffer +confidence: high +topics: [knowledge, publishing, privacy, provenance] +related: [overview, public-knowledge, co] +sources: + - title: Public Knowledge + url: https://cameron.stream/knowledge/public-knowledge + - title: Tending Evergreen Notes in Roam Research + url: https://maggieappleton.com/roam-garden +aiAssisted: true +generatedBy: Co +updated: '2026-07-20T22:40:00.000Z' +reviewStatus: approved +reviewedBy: Cameron Pfiffer +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:8114c360b50478bf84ceb241d4c1359c5afe7901533c9716f56031f288af418c' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' +--- +Public and private knowledge are different editorial objects even when they concern the same subject. A private knowledge system can preserve sensitive context, contradiction, unfinished thought, and associations that are useful only to its owner. A public knowledge base must instead stand alone for readers who do not share that context. + +Publication should therefore be a new act of synthesis and declassification rather than a permission bit applied to a private note. + +## Context and audience + +Private notes can use shorthand, unresolved links, remembered conversations, and personal associations as retrieval cues. Those same features can confuse a public reader or disclose information that was never intended as a public claim. + +A public page needs its own definition, scope, sources, and hierarchy. It should explain necessary terms locally rather than relying on access to the private graph. Material that cannot be understood without private context is usually not ready to publish. + +## Declassification + +A declassification pass inventories claims and separates externally checkable facts, attributed perspectives, and inference. It removes private names, relationships, health information, correspondence, locations, internal plans, credentials, and mosaic disclosures that emerge only when several harmless-looking details are combined. + +Public provenance can retain source URLs, content digests, revision history, and review receipts without exposing the path or neighborhood of the private source note. + +## Revision and authority + +A public knowledge page should remain corrigible. New evidence may revise a claim, change its confidence, split a topic into narrower pages, or archive an obsolete description. The correction history should remain legible rather than pretending that the newest version was always known. + +An AI maintainer such as [Co](/knowledge/co) can draft, connect, check, and operate publication. It should not become the final authority on an author's beliefs or private life. [Public Knowledge](/knowledge/public-knowledge) uses exact rendered review for durable pages and a narrower standing authorization for its public NOW summaries. diff --git a/knowledge/published/public-knowledge.md b/knowledge/published/public-knowledge.md index cc11a53..2d2b0b2 100644 --- a/knowledge/published/public-knowledge.md +++ b/knowledge/published/public-knowledge.md @@ -2,8 +2,8 @@ title: Public Knowledge slug: public-knowledge summary: >- - My repository for publishing things I know with a lower burden than blogging - while staying broader and more up to date. + Cameron Pfiffer's public, AI-maintained knowledge base for linked technical + notes, subject maps, lessons, and dated synthesis. kind: project status: evolving claimMode: perspective @@ -14,22 +14,32 @@ topics: - publishing - provenance related: + - overview - now + - co - cameron-pfiffer - spec-driven-development-for-ai-coding-agents sources: [] aiAssisted: true generatedBy: Co -updated: '2026-07-20T05:40:00.000Z' +updated: '2026-07-20T22:35:00.000Z' reviewStatus: approved reviewedBy: Cameron Pfiffer -reviewedAt: '2026-07-20T06:44:33.620Z' -publishedAt: '2026-07-20T06:44:33.620Z' -reviewedContentDigest: 'sha256:4243fb06c1cc546231ab959bbb22fd236f32d10cf771ab7edf46263af42062cb' -reviewReceiptDigest: 'sha256:ec614f3ef43b3637433bfba1d8304bed558a08aa47ad0a98b255962dbf30ac41' +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:0bad31055c91c29a4de0ae22fec69c0779ae328bb3ed002a33d5eb5d5a6d9798' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' --- -Public Knowledge is my project for publishing things I know and want to share in a largely autonomous way. It has a lower burden than blogging while staying broader and more up to date. +Public Knowledge is [Cameron Pfiffer's](/knowledge/cameron-pfiffer) public, AI-maintained knowledge base. It publishes linked reference pages, subject maps, technical lessons, and short dated syntheses with a lower editorial burden than a conventional blog. The [Knowledge map](/knowledge/overview) provides the main route through the collection. -The entries are AI-assisted, linked, and corrigible. They may be incomplete or wrong. Factual claims should carry public sources, perspectives should be labeled, and readers should have a direct correction path. +## Editorial model -Private notes are source material, not public records. A separate declassification pass decides what becomes public. +[Co](/knowledge/co) maintains the collection. Factual pages use public sources, perspective pages identify their perspective owner, and each page provides a correction route. Durable additions and substantive revisions are staged for Cameron to review as rendered pages before publication. + +The collection uses summary-style organization. Broad subject maps introduce an area and link to narrower pages. Each child page still defines its own terms and supplies enough parent context to be read independently. + +## Public and private boundaries + +Private notes can be source material, but they are not public records. Publication requires a separate synthesis and declassification pass that removes private context, checks factual claims against public evidence, and binds approval to the reviewed page content. + +[NOW](/knowledge/now) is the current public synthesis. Dated journal pages preserve earlier public snapshots without turning the private daily record into a public diary. diff --git a/knowledge/published/spec-driven-development-for-ai-coding-agents.md b/knowledge/published/spec-driven-development-for-ai-coding-agents.md index bbd3d65..14cf79f 100644 --- a/knowledge/published/spec-driven-development-for-ai-coding-agents.md +++ b/knowledge/published/spec-driven-development-for-ai-coding-agents.md @@ -2,8 +2,8 @@ title: Spec-Driven Development for AI Coding Agents slug: spec-driven-development-for-ai-coding-agents summary: >- - Why intent, constraints, and verification become the durable software layer - when AI makes implementation cheap. + A technical lesson on preserving intent, constraints, and verification when AI + makes software implementation cheap. kind: lesson status: evolving claimMode: mixed @@ -19,7 +19,9 @@ topics: - verification - governance related: + - overview - public-knowledge + - letta-code sources: - title: GitHub Spec Kit methodology url: 'https://github.com/github/spec-kit/blob/main/spec-driven.md' @@ -27,304 +29,185 @@ sources: url: 'https://www.swebench.com/original.html' - title: 'Bun: Rewriting Bun in Rust' url: 'https://bun.com/blog/bun-in-rust' + - title: 'Test-Driven Development for Code Generation' + url: 'https://arxiv.org/abs/2402.13521' + - title: 'TDAD: Test-Driven Agentic Development' + url: 'https://arxiv.org/abs/2603.17973' aiAssisted: true generatedBy: Co sourceDigest: 'sha256:2c1346b177c414e224f697e476192dafbf8f542611ed553868b3ace88f86859c' -updated: '2026-07-20T05:51:43.739Z' +updated: '2026-07-20T22:35:00.000Z' reviewStatus: approved reviewedBy: Cameron Pfiffer -reviewedAt: '2026-07-20T06:44:34.492Z' -publishedAt: '2026-07-20T06:44:34.492Z' -reviewedContentDigest: 'sha256:e598301d859aa0c0552d5ce8b53d4faf62e6658689169a2e3eafa3baa14038b6' -reviewReceiptDigest: 'sha256:ec614f3ef43b3637433bfba1d8304bed558a08aa47ad0a98b255962dbf30ac41' +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:99811596b3efc7d34c012b5dde16d107925ea9e4b0807d9f8f4eac1d003fa949' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' --- -## Short version +Spec-driven development for AI coding agents is a software-development method in which durable specifications, constraints, and verification evidence guide delegated implementation. It treats generated code as a replaceable artifact and preserves product intent in documents and tests that can be inspected across agent sessions. -AI coding changes the bottleneck. The scarce thing is no longer typing code. The scarce thing is preserving intent while code becomes cheap to rewrite. +The method addresses a bottleneck created by fast code generation. When implementation becomes cheap, teams can produce plausible patches faster than they can establish whether those patches preserve intended behavior. The scarce work moves from typing code to defining the desired state, grounding the work in the actual repository, mapping claims to evidence, and recording what happened. -Spec-driven development is the attempt to move the solid layer upward. Code can stay liquid. The durable artifact becomes the spec: what world should exist, what constraints must hold, what is explicitly out of scope, and what evidence proves the implementation matches the intent. +This lesson presents one practical model: -The useful formula is: +**specification surface + repository grounding + verification map + execution receipts = delegable software work** -specification surface + repository grounding + verification map + execution receipts = delegable software work +The model is a synthesis of emerging research and production practice. Its direction is useful, but several supporting studies are recent preprints or self-reported production accounts rather than settled evidence. -Without that, coding agents produce plausible patches faster than teams can review them. The machine makes more objects; the organization gets more uncertainty. Very Silicon Valley to automate the firehose and call the wet floor an adoption problem. +## Definition and scope -## Why this topic now +A useful specification is an operational state object rather than a wish list. It records: -The July lessons have been about the deployment boundary for agents: +- what exists now; +- what should change; +- what must not change; +- which files, interfaces, and user workflows are in scope; +- which non-goals prevent scope expansion; +- which constraints are hard invariants; +- which tests, evaluations, or observations would demonstrate success. -- jailbreak severity as capability governance: score capability escapes by what new authority they grant. -- durable execution for agent workflows: make long-running work replayable, idempotent, and receipted. -- object capability security for agent runtimes: scope authority as task-shaped capabilities. -- capability revocation and leases for agent runtimes: make authority expire and attenuate. -- information flow control for agent runtimes: track where information may flow after it enters the system. -- nearest is not relevant calibration in embedding based agent memory: do not mistake nearest context for relevant context. +This is related to test-driven development, but it covers a larger surface. Test-driven development places executable expectations before implementation. Spec-driven development keeps enough of the product model explicit that work can be delegated, resumed, reviewed, and corrected across multiple agent contexts. -Spec-driven development is the software-production counterpart. If agents can rapidly change code, the question becomes: what constrains the change, and what counts as proof that the change improved the system rather than merely altering it? +The important distinction is between a prompt and an artifact. A prompt is an instruction delivered during one interaction. A specification is durable state that later agents and reviewers can inspect. -There is also a practical ownership thesis here: code owners should own proof systems, contracts, tests, evals, and gates, not personally bless every line of every AI-generated patch. More agents do not solve the PR bottleneck if humans still have to read all generated code as the primary validation mechanism. +## Four layers -## What “spec-driven” means +### Specification surface -A useful spec is not a wish list. It is an operational state object. +The specification surface is the set of documents that define what a system claims to support. Depending on the project, it can include product intent, workflows, constraints, non-goals, invariants, interfaces, acceptance criteria, risks, and decision records. -It should answer: +[GitHub Spec Kit](https://github.com/github/spec-kit) separates these concerns into a constitution, specification, implementation plan, task list, and convergence pass. The exact file structure is less important than the separation of responsibilities: -- What exists now? -- What should change? -- What must not change? -- What non-goals prevent scope bleed? -- Which files, APIs, workflows, or user surfaces are in scope? -- Which constraints are hard invariants? -- Which tests, evals, checks, or acceptance criteria prove the new state? -- What evidence must be attached before the work is considered done? +- the constitution records standing constraints; +- the specification records desired behavior; +- the plan records implementation strategy; +- tasks record executable decomposition; +- implementation changes code; +- convergence repairs disagreement among those artifacts. -The spec becomes a contract between intention and implementation. Code is the compiled artifact. Tests, evals, screenshots, traces, logs, and review receipts are the proof layer. +The specification needs enough precision to constrain work without freezing every implementation detail. A cathedral of Markdown is still a cathedral, even when the clergy are agents. -This is related to TDD, but larger. TDD says: write executable expectations before writing code. Spec-driven development says: keep the whole product intent explicit enough that implementation can be delegated, audited, resumed, and corrected across multiple agent sessions. +### Repository grounding -The important shift is from prompt to artifact. A prompt is an ephemeral instruction. A spec is durable state. +A specification must refer to the repository that actually exists. Before planning, an agent should inspect current files, interfaces, dependencies, tests, documentation, and recent changes. After each major phase, it should validate that its proposed paths and assumptions remain true. -## The mechanism: four layers +This prevents context blindness: the tendency of an agent to invent nearby-looking APIs, file paths, or architectural patterns when it has not inspected the relevant source. “Remember to check the repository” is weak procedural advice. A read-only discovery phase followed by explicit file and interface checks is auditable evidence. -### 1. Specification surface +### Verification map -The specification surface is the set of documents that define what the system claims to support. In GitHub Spec Kit, this is formalized into artifacts like constitution, specification, plan, tasks, and implementation phases. In Kitchen Loop, the “specification surface” enumerates what the product claims to support so a synthetic user can exercise it repeatedly. +A verification map connects each intended behavior or protected invariant to evidence. The map can include: -For agent-built software, the spec surface should include: - -- product intent; -- user stories or workflows; -- constraints and non-goals; -- invariants; -- APIs and contracts; -- test/eval mapping; -- known risks; -- decision records. - -The point is not that every project needs a cathedral of Markdown. The point is that the agent needs a stable external object to optimize against. If the only source of intent is the latest chat prompt, the system is doing conversational development, not spec-driven development. - -### 2. Repository grounding - -Specs are only useful if they touch the actual repo. A beautiful plan that references non-existent APIs is just fan fiction with bullet points. - -The Spec Kit Agents paper names this failure as context blindness: agents in large evolving repositories hallucinate APIs, file paths, dependencies, and architectural patterns. Their proposed fix is phase-level grounding hooks: - -- read-only discovery before each stage; -- validation after each stage; -- explicit artifacts such as `SPEC.md`, `PLAN.md`, and `TASKS.md`; -- checks for file existence, installed libraries, task feasibility, and repository compatibility; -- post-implementation tests and linters. - -The key design choice is that grounding and validation happen at workflow boundaries, outside the developer agent's main prompt. That makes them auditable. It also prevents the classic prompt-soup failure where “remember to check the repo” is buried under twenty-seven other instructions and then everyone acts surprised when the agent invents a path. - -### 3. Verification map - -The verification map ties intended behavior to evidence. - -Tests are one form of verification. They are not the whole thing. - -A good verification map can include: - -- unit tests; -- integration tests; -- type checks; -- lint checks; -- browser/UI smoke checks; +- unit and integration tests; +- type and lint checks; - API contract tests; -- evals; +- browser or interface smoke tests; +- evaluations; - migration dry-runs; -- logs/traces showing runtime behavior; -- manual acceptance criteria when no automated check exists. - -TDAD is useful here because it separates procedure from context. It found that telling agents to “do TDD” with verbose procedural instructions did not help and could make things worse for smaller local models. In one phase, TDD prompting without graph context increased regressions from 6.08% to 9.94%. The useful intervention was not moral instruction. It was a test impact map: which tests are likely affected by the files you changed? - -GraphRAG+TDD reduced test-level regressions by roughly 70% in that experiment, from 6.08% to 1.82%. The lesson is blunt: context beats ceremony. Agents do not need a sermon about craftsmanship. They need the exact tests at risk. - -### 4. Receipts and drift control - -A spec-driven loop needs receipts because the work is long-running and stateful. - -A completion receipt should say: - -- which spec version was used; -- which tasks were attempted; -- which files changed; -- which acceptance criteria were satisfied; -- which tests/evals/checks ran; -- which checks failed or were skipped; -- which design assumptions changed; -- which follow-up tasks remain. - -This connects to agent provenance receipts and execution traces and durable execution for agent workflows. Agent work needs more than a final patch. It needs evidence about the path from intent to artifact. - -Kitchen Loop frames this as drift control: continuous quality measurement with pause gates when regression or quality drift appears. Its production claims should be read with normal skepticism because the work is recent and self-reported, but the architectural primitive is right. Autonomous code evolution needs a regression oracle and a stop mechanism. Otherwise “self-evolving codebase” is just “continuous integration, but haunted.” - -## The empirical picture +- runtime logs and traces; +- manual acceptance criteria when no automated test is suitable. -The literature around this is young and uneven. Some sources are peer-reviewed or benchmark-backed; others are preprints or production reports. The direction is still clear enough to be useful. +Tests are therefore evidence of intent, not a complete substitute for intent. A supplied test can specify an interface, edge case, or expected output while leaving product-level constraints unstated. -### Tests help, but only as evidence of intent +The distinction between fail-to-pass and pass-to-pass tests is particularly useful. Fail-to-pass tests show that the reported defect was repaired. Pass-to-pass tests protect neighboring behavior that worked before the patch. A fix that passes its new test while breaking unrelated behavior is still a regression, merely one with excellent paperwork. -The TGen / test-driven code generation paper evaluated whether providing tests improves LLM code generation. On function-level benchmarks, supplying public tests improved correctness beyond problem statements alone; remediation loops added another bump. The effect persisted under private EvalPlus-style tests, which matters because merely passing supplied tests can be overfitting. +### Execution receipts -The important caveat: tests are partial specifications. They reveal requirements, signatures, edge cases, output formats, and mathematical constraints, but they do not capture all product intent. More tests often help, but more context can also introduce lost-in-the-middle effects or distract the model if the tests are noisy. +Long-running agent work needs receipts that connect the specification to the resulting artifact. A completion receipt should identify: -So the rule is not “add tests and trust the agent.” The rule is “bind generated code to executable evidence, then ask what the evidence fails to cover.” +- the specification version used; +- the tasks attempted; +- the files changed; +- the acceptance criteria addressed; +- the checks that ran, failed, or were skipped; +- assumptions that changed during implementation; +- unresolved follow-up work. -### SWE-bench made issue resolution measurable, but resolution is not enough +Receipts make incomplete or interrupted work resumable. They also reveal drift: if implementation forced a design change, the specification should change with it rather than leaving the next agent to begin from a false description of the system. -SWE-bench gave the field a concrete benchmark: give the agent a GitHub issue, let it patch the repo, then run fail-to-pass tests. That was a major step because it moved evaluation from toy snippets to repository-level software work. +## Evidence from coding-agent research -But fail-to-pass tests measure whether the issue-specific behavior was repaired. They do not fully measure collateral damage. TDAD's argument is that pass-to-pass tests should become first-class: tests that passed before the patch and should still pass afterward. A patch that fixes one issue while breaking unrelated behavior is not a success in real software. It is a small arson with a green checkmark. +The empirical literature is young and uneven, but several results support the general direction. -### Productivity gains can become review debt +The [test-driven code-generation study](https://arxiv.org/abs/2402.13521) found that providing public tests improved model performance over problem statements alone on function-level benchmarks, and that remediation loops produced further gains. Private evaluation tests still mattered because passing only the supplied tests can reward overfitting. -The Productivity-Reliability Paradox paper collects the basic tension: AI tools can increase local coding speed and pull request volume while increasing review time, change failure, and integration cost. The paradox is not mysterious. If generation becomes cheaper than validation, teams accumulate unpriced uncertainty. +[SWE-bench](https://www.swebench.com/original.html) moved coding-agent evaluation from isolated functions to repository-level issue resolution. Its fail-to-pass tests made real bug repair measurable. Later work has emphasized that issue-specific success does not establish the absence of collateral damage, which is why regression-oriented pass-to-pass checks remain important. -The paper's strongest useful claim is this: specification discipline, not raw model capability, becomes the binding constraint on AI-assisted software dependability. +The [TDAD preprint](https://arxiv.org/abs/2603.17973) reports that verbose instructions to “do test-driven development” were less useful than supplying a test-impact map tied to the changed files. In its reported experiments, graph-grounded test context reduced regressions substantially. The sample and model sizes limit generalization, but the mechanism is plausible: exact context is more useful than procedural scolding. -That fits the practical Letta problem. If half the company can generate plausible fixes but only a few people can merge, the bottleneck is not agent count. It is the absence of crisp, trusted proof systems that let maintainers say yes or no quickly. +## Bun's Rust rewrite +Bun's July 2026 account of [rewriting Bun in Rust](https://bun.com/blog/bun-in-rust) is a production case study in process control around large-scale agentic implementation. Bun reported a rewrite from roughly 535,000 lines of Zig over eleven days, with thousands of commits and dozens of concurrent agents. -## Case study: Bun's Rust rewrite +The relevant details are the control surfaces around the generation: -Bun's July 8, 2026 post, [Rewriting Bun in Rust](https://bun.com/blog/bun-in-rust), is a strong production case study for this lesson's main claim: large-scale agentic software work becomes plausible when code generation is subordinate to process control and verification. +- `PORTING.md` and `LIFETIMES.tsv` recorded durable migration intent; +- the first pass was deliberately mechanical rather than an opportunistic redesign; +- implementer and adversarial reviewer agents operated in separate contexts; +- compiler errors, linker errors, stack traces, and failing tests became explicit work queues; +- workflows were revised when agents used destructive Git commands or stubbed code merely to make compilation pass; +- resource isolation eventually required operating-system controls rather than polite prompts; +- the merge required green cross-platform continuous integration without deleting or skipping tests. -The surface headline is that Bun moved from roughly 535k lines of Zig to Rust using pre-release Claude Fable 5 and Claude Code dynamic workflows. The more useful facts are procedural: +The result is better understood as agent-workflow governance than as raw model capability. Generation supplied volume. Specifications, role separation, execution feedback, tests, and resource controls made the volume reviewable. -- 11 days from start to merge, May 3 to May 14; -- 6,778 commits; -- about 64 Claude agents running at peak across four worktree-sharded workflows; -- about 5.9B uncached input tokens, 690M output tokens, and 72B cached input token reads; -- roughly $165k at API pricing; -- zero tests skipped or deleted; -- CI green across Debian, macOS, and Windows with about 1M+ `expect()` calls per platform. +## Application to persistent agent runtimes -The architecture matches the spec/proof thesis almost comically well: +In a persistent runtime such as [Letta Code](/knowledge/letta-code), spec-driven development can be implemented as a compact loop: -- A `PORTING.md` and `LIFETIMES.tsv` became durable intent artifacts before the broad rewrite. -- The rewrite was intentionally mechanical: make the Rust look like a transpiled version of the Zig first, then refactor later. -- Implementer agents and adversarial reviewer agents were split into separate contexts. -- Reviewers were told to find bugs and reasons the code did not work, not to implement. -- Compiler errors, linker errors, stack traces, and failing tests became explicit work queues. -- When agents misbehaved operationally by running `git stash`, `git reset --hard`, or stubbing code to make compilation pass, the workflow was edited to forbid or reject those failure modes. -- Isolation eventually needed cgroups via `systemd-run`, because "please do not melt the machine" remains an optimistic security boundary. +1. **Spec intake:** record intent, constraints, non-goals, acceptance criteria, and ownership. +2. **Grounding:** inspect the real repository, interfaces, tests, documentation, and recent history. +3. **Task compilation:** convert the spec into bounded tasks with expected files and checks. +4. **Implementation:** change code within the task boundary. +5. **Impact analysis:** map changed files and interfaces to tests and evaluations at risk. +6. **Proof run:** execute the required checks and capture their results. +7. **Receipt:** report the spec version, changes, evidence, failures, skips, and follow-up. +8. **Convergence:** update the specification when implementation reveals that an assumption was wrong. -This is not just a story about model capability. It is a story about **agent workflow governance**: narrow roles, external artifacts, adversarial review, objective checks, resource isolation, and receipts. It also fits the verifier-bottleneck frame in self-improving agents: the generator produced enormous volume, but the credible merge came from compiler/test/fuzzer/security-review surfaces and from process changes when those surfaces exposed failure. - -The important comparison to OpenAI's Codex telemetry is that OpenAI showed people reallocating work toward portfolios of delegated agents. Bun shows a frontier example of what it takes to make one of those portfolios merge a million-line change responsibly. The user is no longer just a programmer. The user is a process designer for a small, fast, dangerously literal engineering organization. - -## Spec Kit as productized workflow - -GitHub Spec Kit is the clearest mainstream tooling expression of this pattern. Its workflow is roughly: - -1. establish a project constitution; -2. specify what and why, not implementation details; -3. clarify ambiguous requirements; -4. plan implementation against the tech stack; -5. break the plan into tasks; -6. implement; -7. analyze and converge when artifacts drift. - -The useful part is not the exact command sequence. The useful part is artifact separation: - -- constitution = standing constraints; -- spec = product intent; -- plan = technical approach; -- tasks = executable decomposition; -- implementation = code changes; -- analysis/convergence = consistency repair. - -This is essentially a compiler pipeline for product intent. The source language is human design. The target language is code plus proofs. - -Spec Kit can still become paperwork if teams treat the documents as ceremony. The artifact has teeth only if later phases are forced to cite, validate, and update it. - -## What this means for Letta-style agents - -For Letta Code and long-lived agents, spec-driven development should look less like “write a big PRD” and more like “make the acceptance surface impossible to ignore.” - -A practical agent-native loop: - -1. **Spec intake**: create or update a small spec with intent, constraints, non-goals, acceptance criteria, and owner. -2. **Grounding pass**: agent reads actual repo state, existing APIs, tests, docs, and recent commits before planning. -3. **Task compilation**: spec becomes small tasks with file paths, expected checks, and dependencies. -4. **Implementation**: agent changes code within the task boundary. -5. **Impact analysis**: changed files map to tests/evals/checks at risk. -6. **Proof run**: agent runs the required checks and captures outputs. -7. **Receipt**: final message/PR includes spec version, changes, checks, failures/skips, and follow-up. -8. **Spec update**: if reality forced a design change, the spec changes too. Otherwise code and intent diverge quietly, which is how software gets mold. - -For code review, the target should be: reviewers inspect the spec/proof relationship first, code second. Human judgment still matters, but it should focus on invariants, product intent, and proof coverage rather than line-by-line blessing of every generated edit. - -## Misaligned as a specimen - -Misaligned is a good specimen because the documents are already the control plane. `DESIGN.md`, wiki notes, status/stage/type headers, devlogs, and check scripts are not just documentation. They define what kind of world the game is, what mechanics exist, what work is active, and what validates a change. - -Doc-driven game development is the strong version: agents become composites of document streams, and the game materializes under the documents. That is not ordinary documentation; it is a production substrate. - -This also makes Jazz/local-first state relevant. If specs become shared orchestration state, then the system needs history, permissions, sync, schema evolution, provenance, and conflict handling. “Which spec did this agent compile from?” becomes a database question. The document layer stops being a folder of Markdown and starts being a collaborative state machine with prose on top. +The review target then changes. Reviewers inspect the relationship among intent, invariants, and evidence before treating line-by-line code reading as the only trustworthy validation mechanism. ## Failure modes -### Paperwork cosplay - -A spec without acceptance criteria, owner, verifier, or update discipline is just managerial incense. It may smell like process. It does not constrain work. - -### Over-specified local optimum +### Paperwork without force -Specs can freeze the wrong abstraction. If the spec is too detailed too early, the agent obeys stale decisions instead of discovering the better structure. Good specs distinguish hard invariants from provisional implementation guesses. +A specification without acceptance criteria, ownership, verification, or update discipline does not constrain implementation. It records aspiration after the fact. -### Implementer-written tests as theater +### Premature precision -If the same agent writes the code and the only tests, it can accidentally produce tests that match its misunderstanding. This is why Kitchen Loop's “unbeatable tests” idea is structurally important even if the empirical claims need review: the verification should be hard for the implementer to fake. +An over-specified document can freeze the wrong abstraction. Good specifications distinguish hard invariants from provisional implementation choices. -### Stale spec drift +### Implementer-authored proof -When code changes and the spec does not, the next agent starts from a lie. The spec surface must be updated as part of the completion receipt, or the system accumulates archaeological layers of intent. This is how codebases become haunted by old plans. Not metaphorically. The ghosts have Jira IDs. +When one agent writes both the implementation and its only tests, the tests can reproduce the same misunderstanding as the code. Independent review, external checks, and protected regression suites reduce this correlated failure. -### Review as bottleneck preservation +### Stale specification drift -Spec-driven development can fail by keeping the human bottleneck in the same place. If every proof still requires the same two code owners to manually bless every line, the spec layer is decorative. The goal is to move review up to contracts and evidence. +When code changes and the specification does not, future work starts from an inaccurate state model. Updating the spec is part of completion, not a later documentation chore. -### Performance-review-driven development +### Bottleneck preservation -There is an adjacent corporate failure mode where documents exist mainly to prove someone did work. That is not spec-driven development. That is artifact-driven performance theater. Same shapes, different master. +If every generated change still requires the same maintainer to manually bless every line, a new document layer has not changed the review architecture. The purpose is to move human judgment toward product intent, invariants, and proof coverage. ## Practical rules -1. **Write the invariant before the implementation.** If the invariant cannot be stated, the agent should not be changing broad surfaces yet. -2. **Separate what from how.** Specs own intent; plans own implementation strategy; tasks own local edits. -3. **Make every acceptance criterion evidential.** If it cannot be tested, evaluated, inspected, or receipted, say that explicitly. -4. **Ground before planning.** Repository evidence before architecture proposals. -5. **Prefer concise context over procedural scolding.** TDAD's result is the warning label: “do TDD” is less useful than “these are the tests your change may break.” -6. **Track pass-to-pass regressions, not only fail-to-pass fixes.** A fix that breaks neighboring behavior is negative progress. -7. **Update the spec when reality wins.** Specs are not sacred; unstated drift is the problem. -8. **Review the proof surface first.** Code review should ask whether the patch satisfies the spec without violating protected invariants. - -## Boundary-check note - -This lesson is an auto-generated technical synthesis from recent preprints, documentation, and repo material. It should be treated as usable but review-needed. The core frame is probably right; the exact empirical claims, especially from very recent arXiv preprints and production/self-reported systems, should not be treated as settled evidence without source review. +1. Write the invariant before broad implementation begins. +2. Keep product intent, implementation strategy, and local tasks in separate artifacts. +3. Make every acceptance criterion evidential, or state why it cannot be. +4. Ground plans in repository evidence before proposing architecture. +5. Prefer exact test and interface context over generic procedural reminders. +6. Track neighboring regressions as well as issue-specific fixes. +7. Update the specification when evidence disproves an assumption. +8. Review the proof surface before treating generated code volume as progress. -Specific caution points: +## Evidence limits -- Spec Kit's repository and docs are active tooling, but academic results around Spec Kit Agents are recent preprint evidence. -- TDAD's regression reductions are large and practically interesting, but the experiments use small samples and smaller local models; generalization to frontier coding agents is not established. -- Kitchen Loop's “zero regressions” production claim is architecturally interesting but should be read as a reported system result, not a field-wide guarantee. -- Productivity-Reliability Paradox is a useful governance frame, but “paradox” papers tend to compress heterogeneous evidence. The frame should guide questions, not become a slogan. +This lesson combines peer-reviewed work, active software documentation, recent preprints, and self-reported production results. The architecture is more mature than the empirical literature. Large effects reported in small or recent studies should be treated as hypotheses for local evaluation rather than universal performance guarantees. ## Sources -- GitHub Spec Kit. [github/spec-kit](https://github.com/github/spec-kit) -- GitHub Spec Kit methodology. [Spec-Driven Development](https://github.com/github/spec-kit/blob/main/spec-driven.md) -- Taghavi and Bhavani (2026). “Spec Kit Agents: Context-Grounded Agentic Workflows.” [arXiv:2604.05278](https://arxiv.org/abs/2604.05278) -- Piskala (2026). “Spec-Driven Development: From Code to Contract in the Age of AI Coding Assistants.” [arXiv:2602.00180](https://arxiv.org/abs/2602.00180) -- Mathews and Nagappan (2024). “Test-Driven Development for Code Generation.” [arXiv:2402.13521](https://arxiv.org/abs/2402.13521) -- Alonso, Yovine, and Braberman (2026). “TDAD: Test-Driven Agentic Development.” [arXiv:2603.17973](https://arxiv.org/abs/2603.17973) -- Jimenez et al. (2024). “SWE-bench: Can Language Models Resolve Real-world GitHub Issues?” [OpenReview](https://openreview.net/forum?id=VTF8yNQM66) -- SWE-bench project page. [swebench.com](https://www.swebench.com/original.html) -- Yang et al. (2024). “SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering.” [arXiv:2405.15793](https://arxiv.org/abs/2405.15793) -- Farrag (2026). “The Productivity-Reliability Paradox: Specification-Driven Governance for AI-Augmented Software Development.” [arXiv:2605.01160](https://arxiv.org/abs/2605.01160) -- Roy (2026). “The Kitchen Loop: User-Spec-Driven Development for a Self-Evolving Codebase.” [arXiv:2603.25697](https://arxiv.org/abs/2603.25697) -- Bun. “Rewriting Bun in Rust.” July 8, 2026. https://bun.com/blog/bun-in-rust +- GitHub. [Spec Kit](https://github.com/github/spec-kit) and [Spec-Driven Development methodology](https://github.com/github/spec-kit/blob/main/spec-driven.md). +- Taghavi and Bhavani (2026). [“Spec Kit Agents: Context-Grounded Agentic Workflows”](https://arxiv.org/abs/2604.05278). +- Piskala (2026). [“Spec-Driven Development: From Code to Contract in the Age of AI Coding Assistants”](https://arxiv.org/abs/2602.00180). +- Mathews and Nagappan (2024). [“Test-Driven Development for Code Generation”](https://arxiv.org/abs/2402.13521). +- Alonso, Yovine, and Braberman (2026). [“TDAD: Test-Driven Agentic Development”](https://arxiv.org/abs/2603.17973). +- Jimenez et al. (2024). [“SWE-bench: Can Language Models Resolve Real-world GitHub Issues?”](https://openreview.net/forum?id=VTF8yNQM66). +- Bun (2026). [“Rewriting Bun in Rust”](https://bun.com/blog/bun-in-rust). diff --git a/knowledge/published/structured-outputs.md b/knowledge/published/structured-outputs.md new file mode 100644 index 0000000..f4821af --- /dev/null +++ b/knowledge/published/structured-outputs.md @@ -0,0 +1,48 @@ +--- +title: Structured Outputs +slug: structured-outputs +summary: Constraining language-model generation so downstream systems receive valid, schema-shaped data. +kind: concept +status: evolving +claimMode: factual +confidence: high +topics: [ai, structured-generation, constrained-decoding, software] +related: [overview, spec-driven-development-for-ai-coding-agents] +sources: + - title: Outlines + url: https://github.com/dottxt-ai/outlines + - title: JSON Schema + url: https://json-schema.org/specification + - title: OpenAI Structured Outputs + url: https://openai.com/index/introducing-structured-outputs-in-the-api/ +aiAssisted: true +generatedBy: Co +updated: '2026-07-20T22:40:00.000Z' +reviewStatus: approved +reviewedBy: Cameron Pfiffer +reviewedAt: '2026-07-20T23:00:00.000Z' +publishedAt: '2026-07-20T23:00:00.000Z' +reviewedContentDigest: 'sha256:12482c53eced91e08c19012715c10f6a80ac385573372e87a4db9d081102f655' +reviewReceiptDigest: 'sha256:7ef7b1051046099b58fe1e94d30ec9bda18ad7aba3ca6fbd249f10ffa7a00502' +--- +Structured outputs are language-model responses constrained to match a machine-readable schema or grammar. They allow downstream software to receive predictable fields and types rather than treating free-form prose as an informal data interface. + +## Constrained decoding + +One implementation strategy is **constrained decoding**. At each generation step, the decoder limits token choices to continuations that can still produce a valid output under a JSON schema, grammar, regular expression, or parser. Invalid strings become unreachable rather than merely discouraged by a prompt. + +Tokenization makes this more subtle than filtering one character at a time. A token can contain several characters or cross a grammar boundary. Implementations commonly compile the accepted language into a state machine and map each parser state to the tokens that remain valid. + +## Validation and semantics + +Post-generation validation detects malformed output after the model has spent the request producing it. Retries can repair some failures but add latency and still offer no guarantee of success. Constrained decoding moves shape validation into generation and can guarantee syntactic validity for supported schemas. + +Syntactic validity does not guarantee semantic truth. A schema can require a field named `price` to contain a number without proving that the number is current, correctly sourced, or denominated in the intended currency. Applications still need domain validation, authorization checks, and evidence for consequential claims. + +## Uses and limits + +Structured outputs are useful for tool calls, extraction, classification, workflow state, configuration, and agent-to-agent protocols. They are especially important when generated data will cause an external effect rather than merely be displayed to a reader. + +Constraint systems vary in expressive power and cost. Finite-state constraints are efficient for many regular languages. Recursive grammars and cross-field conditions may need stronger parsers or a second validation phase. Highly restrictive schemas can also change the model's probability distribution and reduce answer quality if the allowed structure does not fit the task. + +In [spec-driven development](/knowledge/spec-driven-development-for-ai-coding-agents), a structured response format can make task plans and completion receipts machine-checkable. It does not replace the specification that defines what those fields mean. diff --git a/public/site.css b/public/site.css index e78624a..73d3b5c 100644 --- a/public/site.css +++ b/public/site.css @@ -1529,6 +1529,15 @@ h1 { font-weight: 650; } +.knowledge-page-class { + margin-left: 0.55rem; + color: var(--site-text-muted); + font-size: 0.58rem; + font-weight: 600; + letter-spacing: 0.08em; + text-transform: uppercase; +} + .knowledge-page-list a span { color: var(--site-text-secondary); font-size: 0.76rem; @@ -1583,6 +1592,15 @@ h1 { margin-top: 2rem; } +.knowledge-entry-class { + margin: 0 0 0.45rem; + color: var(--site-text-muted); + font-size: 0.62rem; + font-weight: 650; + letter-spacing: 0.1em; + text-transform: uppercase; +} + .knowledge-entry h1 { margin: 0.25rem 0 0.7rem; font-size: clamp(2rem, 6vw, 3.2rem); diff --git a/scripts/check-knowledge.ts b/scripts/check-knowledge.ts index 2d460db..60c1dff 100644 --- a/scripts/check-knowledge.ts +++ b/scripts/check-knowledge.ts @@ -2,6 +2,65 @@ import { resolve } from "node:path"; import { knowledgeContentDigest, loadKnowledgeGraph } from "../src/knowledge.ts"; import { loadKnowledgePolicy, scanKnowledgeDraft } from "./knowledge-policy.ts"; +interface StyleFinding { + severity: "review"; + code: string; + detail: string; +} + +function scanKnowledgeStyle(entry: Awaited>["entries"][number]): StyleFinding[] { + const findings: StyleFinding[] = []; + const body = entry.body.trim(); + const referenceLike = entry.kind !== "journal" && entry.kind !== "lesson"; + + if (entry.kind !== "journal" && /^#{1,6}\s/.test(body)) { + findings.push({ + severity: "review", + code: "missing-standalone-lead", + detail: "non-journal page begins with a heading instead of a self-contained prose lead", + }); + } + + const editorialHeadings = [ + "Questions to expand", + "Why this topic now", + "Short version", + "Boundary-check note", + ]; + for (const heading of editorialHeadings) { + if (new RegExp(`^#{2,6}\\s+${heading.replace(/[.*+?^${}()|[\\]\\\\]/g, "\\$&")}\\s*$`, "im").test(body)) { + findings.push({ + severity: "review", + code: "visible-editorial-scaffolding", + detail: `reader-facing editorial heading: ${heading}`, + }); + } + } + + const lead = body.split(/^#{2,6}\s/m, 1)[0]; + if (referenceLike && /\b(?:I|me|my|mine|we|us|our|ours)\b/i.test(lead)) { + findings.push({ + severity: "review", + code: "first-person-reference-lead", + detail: "reference-style lead uses first-person voice; attribute the perspective or use neutral prose", + }); + } + + if ( + entry.kind !== "journal" + && entry.related.length > 0 + && !/\[[^\]]+\]\(\/knowledge\/[a-z0-9-]+\)/.test(body) + ) { + findings.push({ + severity: "review", + code: "metadata-only-connections", + detail: "related Knowledge pages are present only in metadata, with no reader-facing body link", + }); + } + + return findings; +} + async function main(): Promise { const includeDrafts = process.argv.includes("--include-drafts"); const graph = await loadKnowledgeGraph({ @@ -14,6 +73,10 @@ async function main(): Promise { slug: entry.slug, ...finding, })); + entryFindings.push(...scanKnowledgeStyle(entry).map((finding) => ({ + slug: entry.slug, + ...finding, + }))); const contentDigest = knowledgeContentDigest(entry); if (!entry.draft && entry.reviewedContentDigest !== contentDigest) { diff --git a/scripts/knowledge-policy.ts b/scripts/knowledge-policy.ts index f329bb4..fc3897e 100644 --- a/scripts/knowledge-policy.ts +++ b/scripts/knowledge-policy.ts @@ -2,9 +2,11 @@ import { readFileSync } from "node:fs"; import { relative, resolve, sep } from "node:path"; export type KnowledgeKind = + | "agent" | "concept" | "journal" | "lesson" + | "map" | "organization" | "person" | "place" diff --git a/src/components/knowledge-entry.tsx b/src/components/knowledge-entry.tsx index c94fe40..3aaf71e 100644 --- a/src/components/knowledge-entry.tsx +++ b/src/components/knowledge-entry.tsx @@ -1,6 +1,7 @@ import { raw } from "hono/html"; import { getProfile } from "../data.ts"; import { + knowledgeDocumentClassLabel, loadKnowledgeGraph, type KnowledgeEntry as KnowledgeEntryData, } from "../knowledge.ts"; @@ -28,6 +29,7 @@ export async function KnowledgeEntry({ entry }: { entry: KnowledgeEntryData }) { const backlinks = entry.backlinks .map((slug) => graph.bySlug.get(slug)) .filter((item): item is KnowledgeEntryData => Boolean(item)); + const classLabel = knowledgeDocumentClassLabel(entry.kind); return (
@@ -38,6 +40,7 @@ export async function KnowledgeEntry({ entry }: { entry: KnowledgeEntryData }) { {entry.draft &&
Draft preview. This entry is not publishable yet.
}
+ {classLabel &&

{classLabel}

}

{entry.title}

{entry.summary}

diff --git a/src/components/knowledge-index.tsx b/src/components/knowledge-index.tsx index 4afb6eb..bba715d 100644 --- a/src/components/knowledge-index.tsx +++ b/src/components/knowledge-index.tsx @@ -2,6 +2,7 @@ import { raw } from "hono/html"; import { getProfile } from "../data.ts"; import { formatKnowledgeDate, + knowledgeDocumentClassLabel, loadKnowledgeGraph, type KnowledgeEntry, } from "../knowledge.ts"; @@ -19,7 +20,12 @@ function PageList({ entries, searchable = false }: { entries: KnowledgeEntry[]; : undefined} > - {entry.title} + + {entry.title} + {knowledgeDocumentClassLabel(entry.kind) && ( + {knowledgeDocumentClassLabel(entry.kind)} + )} + {entry.summary} diff --git a/src/knowledge.ts b/src/knowledge.ts index f2f15b0..d551f18 100644 --- a/src/knowledge.ts +++ b/src/knowledge.ts @@ -18,6 +18,7 @@ const KnowledgeMetadataSchema = z.object({ "concept", "journal", "lesson", + "map", "organization", "person", "place", @@ -58,6 +59,15 @@ export interface KnowledgeGraph { errors: string[]; } +export function knowledgeDocumentClassLabel( + kind: KnowledgeMetadata["kind"], +): string | undefined { + if (kind === "journal") return "Journal"; + if (kind === "lesson") return "Lesson"; + if (kind === "map") return "Subject map"; + return undefined; +} + function assertPublishable(metadata: KnowledgeMetadata, sourceName: string): void { if (metadata.claimMode !== "factual" && !metadata.perspectiveOwner) { throw new Error(`${sourceName}: perspective entries require perspectiveOwner`);