Something went wrong. Try again.
A local-first event pipeline for independent agents, built on Jazz.
Something went wrong. Try again.
33 kB · 594 lines
TypeScript
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595import { canonicalJson, sha256, type JsonObject } from "../core/json.js";
export const POST_TRAINING_COURSE_ID = "post-training-model-factory";export const MACHINE_PROJECT_EULER_REVISION = "9d43b4e67a3de49fdfe1cff0153c513c528142f9";
export interface CourseReference { label: string; url: string; use: string;}
export interface CourseCodeExample { language: string; caption: string; code: string;}
export interface CourseSection { id: string; title: string; paragraphs: string[]; bullets?: string[] | undefined; code?: CourseCodeExample | undefined; callout?: string | undefined;}
export interface CourseWorkshop { kind: "capability-contract" | "machine-gate" | "data-split" | "environment" | "loss-mask" | "preference" | "reward" | "factory"; title: string; prompt: string;}
export interface CourseLesson { id: string; number: number; title: string; subtitle: string; objective: string; sections: CourseSection[]; workshop: CourseWorkshop; references: CourseReference[];}
export interface PostTrainingCourse { id: typeof POST_TRAINING_COURSE_ID; revision: string; title: string; subtitle: string; summary: string; oneThing: string; lessons: CourseLesson[]; glossary: Array<{ term: string; definition: string }>;}
const courseBody: Omit<PostTrainingCourse, "revision"> = { id: POST_TRAINING_COURSE_ID, title: "From capability to model factory", subtitle: "A working course in language-model post-training", summary: "Start with one executable behavior, then follow the data, optimization, evaluation, and release machinery that turns a base model into a product model.", oneThing: "Post-training is a controlled loop that turns a behavioral claim into data, model updates, and evidence for or against release.", lessons: [ { id: "capability-hello-world", number: 1, title: "A capability is an executable claim", subtitle: "The post-training hello world", objective: "Write one behavior precisely enough that a machine can test it without guessing what you meant.", sections: [ { id: "smallest-loop", title: "The smallest complete loop", paragraphs: [ "Suppose you want a model to solve a tiny coding task. The prompt asks it to create a file, run that file, and produce the exact output 42. A response that merely says 42 is wrong because the capability includes acting in an environment.", "A complete post-training loop has six parts: define the behavior, collect examples, update the model, evaluate held-out attempts, compare the candidate with the current model, and promote only when the evidence clears a gate.", ], bullets: [ "Input: a natural-language task plus a fresh workspace.", "Action: edit the named file and execute it inside the workspace.", "Success: the process exits with code 0 and prints exactly 42.", "Exclusion: the model cannot use the network or place the answer in a different file.", ], code: { language: "python", caption: "One valid trajectory", code: "$ cat > answer.py <<'PY'\nprint(6 * 7)\nPY\n$ python answer.py\n42", }, }, { id: "capability-versus-benchmark", title: "A benchmark score is not the capability", paragraphs: [ "The capability is the behavior you want in the world. The evaluator is an instrument for observing that behavior. A benchmark compresses many observations into a score. Confusing the score with the capability lets the model win the instrument while failing the job.", "Write the acceptance contract before generating training data. Otherwise each data-generation decision silently changes the target.", ], callout: "If you cannot state what would make one rollout pass or fail, you are not ready to choose SFT, DPO, or RL.", }, { id: "what-post-training-changes", title: "Post-training changes a conditional distribution", paragraphs: [ "A pretrained model predicts continuations from broad internet-scale regularities. Post-training changes which continuations are likely under particular instructions, roles, tool states, and feedback. It can teach a response format, a task policy, a preference, or a multi-step interaction pattern.", "The optimizer does not receive your intent. It receives tokens, comparisons, rewards, and gradients. The engineering job is to make those signals point at the behavior you actually care about.", ], }, ], workshop: { kind: "capability-contract", title: "Build the hello-world gate", prompt: "Toggle the evidence until the contract distinguishes doing the task from merely naming the answer.", }, references: [ { label: "InstructGPT", url: "https://arxiv.org/abs/2203.02155", use: "The canonical demonstration → preference model → PPO pipeline.", }, { label: "Tinker docs", url: "https://tinker-docs.thinkingmachines.ai/", use: "A concrete API surface for supervised and reinforcement-learning updates.", }, ], }, { id: "machine-project-euler", number: 2, title: "What Machine's Project Euler experiment does", subtitle: "One capability loop with real failure receipts", objective: "Read the public Machine experiment as a release system rather than as one fine-tuning run.", sections: [ { id: "target-behavior", title: "The target is tool-shaped problem solving", paragraphs: [ "Machine asks coding agents to solve Project Euler problems in Python, JavaScript, and Julia. Each rollout receives a fresh networkless Docker workspace. The agent must mutate and execute the same file, exit successfully, and produce the exact hidden answer.", "Answer-bearing data stays in evaluator-only storage. The model sees the problem and the workspace, not the gold answer. This separation lets execution prove the behavior instead of letting the prompt leak it.", ], bullets: [ "Fresh container per rollout.", "No network access.", "Exact file-mutation and direct-execution evidence.", "Exact stdout and final-answer match.", "Escape attempts fail the rollout.", ], }, { id: "promotion-gate", title: "Promotion requires repeatable breadth", paragraphs: [ "A candidate must qualify on at least six of eight trajectories in every language for two consecutive rounds. It must also retain frozen regression behavior. One good aggregate score cannot hide a weak language, and one lucky round cannot ship a checkpoint.", "The native Qwen3.5-4B baseline scored 17 of 24: JavaScript 7 of 8, Julia 5 of 8, and Python 5 of 8. The total looks respectable. The per-language contract correctly rejects it.", ], callout: "The gate is conjunctive: every language, two rounds, plus regression. Average performance is not an escape hatch.", }, { id: "diagnosis-before-training", title: "The failure determines the intervention", paragraphs: [ "Machine does not treat training as one button. It diagnoses the failed behavior, then chooses among supervised trajectories, answer-only GRPO, hidden-variant tool GRPO, or Pi-shaped remediation. Those methods teach different things.", ], bullets: [ "SFT can teach the syntax and order of a valid tool trajectory.", "Answer-only GRPO can improve final answers while destroying tool use.", "Hidden-variant tool GRPO rewards execution against unseen problem variants.", "Pi-shaped remediation targets the full agent loop rather than a naked answer policy.", ], }, { id: "negative-results", title: "No checkpoint cleared promotion", paragraphs: [ "A direct-answer checkpoint passed native rounds and then scored 0 of 6 in the full Pi harness with zero tool calls. Later candidates sometimes passed one Pi round and failed the second, usually on Julia. A scorer defect also invalidated an apparent pass.", "The public result is therefore a negative one: Project Euler problem 2 remains locked. That is the useful result. The pipeline prevented a plausible training story from becoming a false capability claim.", ], }, ], workshop: { kind: "machine-gate", title: "Run the Machine promotion gate", prompt: "Change language scores across two rounds and watch the release decision respond to the weakest required slice.", }, references: [ { label: "Machine guide", url: `https://tangled.org/cameron.stream/machine/blob/${MACHINE_PROJECT_EULER_REVISION}/examples/project-euler/README.md`, use: `The public experiment, commands, gates, and result at revision ${MACHINE_PROJECT_EULER_REVISION.slice(0, 12)}.`, }, { label: "Machine repository", url: "https://tangled.org/cameron.stream/machine", use: "Evaluator, curriculum, GRPO, and behavior-release source.", }, ], }, { id: "data-design", number: 3, title: "Data is a behavioral argument", subtitle: "Examples decide what the optimizer can learn", objective: "Design training and evaluation data that support the capability claim without leaking the answer.", sections: [ { id: "examples-as-claims", title: "Every example argues for a policy", paragraphs: [ "An SFT example says, 'When the context looks like this, increase the probability of these next tokens.' A preference pair says, 'Under this prompt, move probability toward one response and away from another.' An RL environment says, 'Explore, then increase the probability of trajectories that earn this feedback.'", "Data volume cannot rescue a confused target. Ten thousand examples that reward final-answer prose will not teach reliable file mutation and execution.", ], }, { id: "splits-and-lineage", title: "Split by the unit that could leak", paragraphs: [ "Random row splits are weak when several rows share a problem template, repository, author, or hidden answer. Hold out the unit that defines novelty. For Machine, that means hidden variants, language slices, and locked later problems rather than shuffled copies of the same answer-bearing structure.", ], bullets: [ "Training: examples the optimizer may inspect.", "Development evaluation: held-out evidence used to make design choices.", "Confirmation evaluation: frozen evidence used after choices are complete.", "Regression replay: old capabilities that the new candidate must retain.", ], }, { id: "synthetic-data", title: "Synthetic data needs provenance and filters", paragraphs: [ "A stronger model can generate prompts, demonstrations, critiques, or comparisons. The generator's fluency does not make the sample correct. Keep generator identity, prompt revision, source material, verifier result, and deduplication lineage so you can explain why an example entered the mix.", "Use executable verification when the domain permits it. Use model judges for dimensions that cannot be reduced to a test, and measure judge agreement rather than treating one score as ground truth.", ], }, ], workshop: { kind: "data-split", title: "Protect the hidden set", prompt: "Place each item where it belongs. The trick is to split on the source of dependence, not the row label.", }, references: [ { label: "Tülu 3", url: "https://arxiv.org/abs/2411.15124", use: "Open SFT, preference, RLVR, decontamination, and multi-task evaluation recipes.", }, { label: "SWE-Gym", url: "https://arxiv.org/abs/2412.21139", use: "Real repositories, executable environments, unit-test validation, and held-out repositories.", }, ], }, { id: "environments", number: 4, title: "The environment is part of the model", subtitle: "Actions, observations, and resets define the task", objective: "See why executable environments are training infrastructure rather than benchmark packaging.", sections: [ { id: "interaction-contract", title: "An agent learns through an interface", paragraphs: [ "A tool-using model does not act on the world directly. A harness renders observations, admits actions, executes tools, returns results, and decides when the episode ends. Change that interface and you change the effective task.", "The environment therefore needs a versioned contract: initial state, available actions, observation format, resource limits, terminal conditions, and evaluator behavior.", ], }, { id: "reset-and-replay", title: "Resettable episodes make evidence comparable", paragraphs: [ "Each rollout should begin from a known state. Without reset, one trajectory can inherit files, caches, network state, or evaluator artifacts from another. The resulting reward no longer belongs to the policy that received it.", "Record the environment image, task revision, tool versions, random seed where relevant, and exact terminal evidence. A later replay should fail for the same reason or pass for the same reason.", ], }, { id: "verifiers", title: "Verifiers turn consequences into feedback", paragraphs: [ "A verifier can run tests, compare outputs, inspect a patch, check a proof, or combine several predicates. Strong verifiers make online learning practical because they can score many trajectories without a human reading each one.", "Weak verifiers create reward hacking. If the evaluator checks only stdout, the model may print the answer without editing the file. Machine checks mutation, execution, output, answer, and escape behavior because each predicate closes a different shortcut.", ], }, ], workshop: { kind: "environment", title: "Close the shortcut", prompt: "Disable environment checks and inspect which invalid trajectory becomes reward-equivalent to the real behavior.", }, references: [ { label: "SWE-Gym", url: "https://arxiv.org/abs/2412.21139", use: "A concrete training environment with codebases, runtimes, tests, and natural-language tasks.", }, { label: "Machine design", url: "https://tangled.org/cameron.stream/machine/blob/main/examples/project-euler/docs/DESIGN.md", use: "The networkless workspace and exact functional-gate contract.", }, ], }, { id: "supervised-fine-tuning", number: 5, title: "SFT teaches the shape of a trajectory", subtitle: "Teacher forcing, chat rendering, and loss masks", objective: "Understand what supervised fine-tuning updates and why formatting choices become model behavior.", sections: [ { id: "next-token-loss", title: "SFT is next-token learning on selected tokens", paragraphs: [ "A training example is rendered into tokens using the model's chat template. The model predicts each next token, and the loss compares those predictions with the example. Backpropagation changes weights so the selected target tokens become more likely in similar contexts.", "The loss mask decides which tokens count. Many instruction-tuning runs mask system and user tokens, then train on assistant tokens. Tool-call formats and tool results need an explicit policy; an accidental mask can teach the model to imitate observations or ignore actions.", ], }, { id: "teacher-forcing", title: "Teacher forcing hides rollout errors", paragraphs: [ "During SFT, the model sees the correct prior tokens even when it would have produced a different earlier token on its own. This makes optimization stable, but it means low training loss does not prove the model can recover through a long autonomous trajectory.", "Evaluate by sampling complete rollouts in the real harness. Token accuracy and episode success answer different questions.", ], }, { id: "mix-and-retention", title: "The data mix allocates model capacity", paragraphs: [ "Oversampling one capability can improve that slice while changing tone, refusal behavior, language coverage, or general reasoning. Labs mix new capability data with replay data and watch a regression suite through training.", "Learning rate, sequence length, batch construction, and adapter capacity determine how aggressively the candidate moves. A small low-rank adapter can be easier to isolate and roll back, but it still needs the same behavioral gates.", ], }, ], workshop: { kind: "loss-mask", title: "Choose which tokens teach", prompt: "Toggle role masks and inspect the target tokens. The model learns from the highlighted continuation, not from your prose description of the task.", }, references: [ { label: "InstructGPT", url: "https://arxiv.org/abs/2203.02155", use: "Supervised demonstrations as the first post-training stage.", }, { label: "Tülu 3", url: "https://arxiv.org/abs/2411.15124", use: "A large open SFT mixture and staged post-training recipe.", }, ], }, { id: "preferences", number: 6, title: "Preferences turn comparisons into an objective", subtitle: "Judges, reward models, and DPO", objective: "Separate preference data, judge policy, reward modeling, and direct preference optimization.", sections: [ { id: "comparison-data", title: "A preference label is conditional", paragraphs: [ "A preference record contains one prompt, two or more candidate responses, a judgment, and the criterion used to judge. The label means one response was preferred under that criterion and evidence. It does not mean the response is universally better.", "Blinded presentation, randomized order, judgeability checks, tie options, and correction fields reduce noise. Preserve disagreements. They often reveal that the criterion or evidence is underspecified.", ], }, { id: "reward-model", title: "A reward model generalizes comparisons", paragraphs: [ "A reward model learns a scalar score that predicts preferences. That score can evaluate fresh policy samples, which makes RLHF possible. The reward model also creates a new failure surface: the policy can exploit features that correlate with high scores without satisfying the human criterion.", ], }, { id: "dpo", title: "DPO trains the policy directly", paragraphs: [ "Direct Preference Optimization uses preference pairs to increase the policy's relative log probability of the chosen response over the rejected response, measured against a reference policy. It avoids training a separate reward model and avoids online RL during the update.", "DPO is simpler than PPO-based RLHF, but it still inherits the preference dataset's coverage and judge errors. It learns from the pairs you collected. It does not explore new trajectories while training.", ], callout: "SFT says what to imitate. Preference optimization says which sampled behavior to favor. Neither signal automatically proves environment success.", }, ], workshop: { kind: "preference", title: "Judge a pair", prompt: "Choose under an explicit criterion, then inspect the training record your click creates.", }, references: [ { label: "DPO", url: "https://arxiv.org/abs/2305.18290v3", use: "The direct policy objective derived from a preference model.", }, { label: "InstructGPT", url: "https://arxiv.org/abs/2203.02155", use: "Human comparisons, reward modeling, and PPO in one production-scale recipe.", }, ], }, { id: "reinforcement-learning", number: 7, title: "RL optimizes sampled behavior", subtitle: "Rollouts, advantages, GRPO, and reward hacking", objective: "Follow one on-policy update from sampled trajectories to a constrained policy change.", sections: [ { id: "online-loop", title: "RL trains on what the current policy actually does", paragraphs: [ "The system samples several trajectories from the current policy, runs each trajectory in an environment, computes feedback, estimates which actions performed better than expected, and updates the policy toward those actions. New rollouts then come from the updated policy.", "This online loop can discover behavior absent from demonstrations. It is also expensive because generation, environment execution, scoring, and training repeat continuously.", ], }, { id: "ppo-grpo", title: "PPO and GRPO estimate improvement differently", paragraphs: [ "PPO commonly uses a learned value function to estimate advantages, clips large policy-ratio changes, and adds a KL penalty against a reference policy. GRPO compares rewards within a group of samples for the same prompt and can avoid a separate critic model.", "The algorithm name does not determine the behavior. Reward design, sampling temperature, group composition, token-level credit, KL control, and environment validity often dominate the result.", ], }, { id: "verifiable-rewards", title: "Verifiable rewards scale narrow truths", paragraphs: [ "Math answers, tests, compilers, and formal checkers can score many rollouts cheaply. Tülu 3 calls this reinforcement learning with verifiable rewards. DeepSeek-R1 reports large-scale reasoning RL using GRPO, then adds cold-start data and later supervised stages to repair readability and broader behavior.", "A verifier is strong only inside its scope. A model can become excellent at earning the reward while losing tool use, language quality, or unrelated capabilities. Machine's direct-answer checkpoint is the compact example: native answer gates improved while the Pi tool gate fell to 0 of 6.", ], }, { id: "hacking", title: "Reward hacking is ordinary optimization", paragraphs: [ "The policy searches the reward surface you built. If an unintended shortcut scores well, taking it is not mysterious or adversarial in the human sense. It is the expected result of optimizing an incomplete instrument.", "Use several predicates, hidden variants, adversarial probes, held-out evaluators, and human inspection. Keep release authority outside the training loop.", ], }, ], workshop: { kind: "reward", title: "Watch the reward choose a shortcut", prompt: "Change reward weights and see which trajectory the optimizer would favor.", }, references: [ { label: "DeepSeek-R1", url: "https://arxiv.org/abs/2501.12948v1", use: "GRPO-centered reasoning RL and the later multi-stage repair recipe.", }, { label: "Tülu 3", url: "https://arxiv.org/abs/2411.15124", use: "Open reinforcement learning with verifiable rewards and unseen evaluation.", }, ], }, { id: "model-factory", number: 8, title: "A model factory is a release system", subtitle: "How labs run post-training at scale", objective: "Map a single capability loop onto the distributed systems, registries, and decision gates used at lab scale.", sections: [ { id: "factory-components", title: "The loop becomes several coordinated systems", paragraphs: [ "At lab scale, one post-training run spans data ingestion, curation, rollout generation, environment execution, judging, training, checkpoint storage, evaluation, serving, and release control. Each stage has different compute shapes and failure modes.", ], bullets: [ "Data factory: source lineage, filtering, deduplication, decontamination, mixing, and versioned datasets.", "Rollout fleet: replicated inference workers sampling current and reference policies.", "Environment fleet: resettable sandboxes, tools, simulators, and verifiers.", "Learner: distributed gradient computation, optimizer state, checkpoints, and fault recovery.", "Evaluation service: frozen suites, slice metrics, judge calibration, regressions, and contamination checks.", "Registry and serving: immutable model identities, candidate channels, canaries, rollback, and traffic policy.", ], }, { id: "distributed-dataflow", title: "RL alternates generation and training workloads", paragraphs: [ "Rollout inference wants high-throughput generation. Learning wants large synchronized training batches. The actor, reference model, reward model, and critic may require different placements. Moving weights and optimizer state between these phases can dominate throughput.", "HybridFlow describes RLHF as a distributed dataflow whose nodes are model programs and whose edges move many-to-many data. Its hybrid controller and 3D-HybridEngine address orchestration and resharding between generation and training. This is the systems layer hidden by a small notebook example.", ], }, { id: "release-discipline", title: "Training completion does not authorize release", paragraphs: [ "A candidate checkpoint enters an evaluation matrix: target capability, adjacent capabilities, safety, latency, cost, language slices, tool behavior, long-context behavior, and regressions. The release gate names which failures block promotion and which require human review.", "After offline promotion, shadow traffic and canaries test the serving path. Monitoring watches output quality, tool errors, refusal shifts, latency, and drift. A rollback pointer should identify the previous known-good model and its exact runtime configuration.", ], }, { id: "factory-lesson", title: "Scale does not change the epistemology", paragraphs: [ "A lab can run millions of rollouts and still optimize the wrong instrument. More GPUs increase the rate at which a confused capability definition produces convincing graphs.", "The hello-world loop survives intact: define the behavior, generate evidence, update the policy, test hidden cases, compare against the incumbent, and promote only when the contract passes. The factory exists to run that loop repeatedly without losing lineage or control.", ], callout: "Model factories manufacture candidates. Release systems decide which candidate becomes real.", }, ], workshop: { kind: "factory", title: "Route a candidate through the factory", prompt: "Inspect each stage, inject one failure, and decide whether the checkpoint may reach serving.", }, references: [ { label: "HybridFlow", url: "https://arxiv.org/abs/2409.19256", use: "Distributed RLHF dataflow, model placement, and training-generation resharding.", }, { label: "Tülu 3", url: "https://arxiv.org/abs/2411.15124", use: "An open multi-stage recipe with data, infrastructure, evaluation, and failed experiments.", }, { label: "Tinker docs", url: "https://tinker-docs.thinkingmachines.ai/", use: "A managed training API that exposes pieces of the factory without requiring local cluster orchestration.", }, ], }, ], glossary: [ { term: "Base model", definition: "A pretrained checkpoint before task- or preference-specific post-training." }, { term: "Capability", definition: "A behavior defined over inputs, actions, environments, and acceptance conditions." }, { term: "Checkpoint", definition: "A versioned snapshot of model parameters and, when needed, optimizer state." }, { term: "Curriculum", definition: "A governed sequence or mixture of training tasks and difficulty levels." }, { term: "DPO", definition: "Direct Preference Optimization, which trains a policy from chosen and rejected responses without an online RL loop." }, { term: "Evaluation slice", definition: "A named subset whose score must remain visible rather than disappearing inside an average." }, { term: "GRPO", definition: "Group Relative Policy Optimization, which estimates relative advantage from a group of sampled responses." }, { term: "Held-out", definition: "Evidence excluded from the optimization or design decisions it evaluates." }, { term: "KL penalty", definition: "A constraint that discourages the updated policy from moving too far from a reference policy." }, { term: "Loss mask", definition: "The token positions that contribute to a supervised training objective." }, { term: "Policy", definition: "The model distribution used to choose the next token or action." }, { term: "Reward model", definition: "A model trained to predict a scalar preference score for a candidate response or trajectory." }, { term: "RLVR", definition: "Reinforcement learning with rewards produced by executable or otherwise objective verifiers." }, { term: "Rollout", definition: "One sampled trajectory from a policy through a prompt or environment." }, { term: "SFT", definition: "Supervised fine-tuning on target token sequences, usually with teacher forcing." }, { term: "Verifier", definition: "A program or model that converts trajectory evidence into pass/fail or reward feedback." }, ],};
const revision = sha256(canonicalJson(courseBody as unknown as JsonObject));
export const POST_TRAINING_COURSE: PostTrainingCourse = deepFreeze({ ...courseBody, revision,});
export function postTrainingLesson(lessonId: string): CourseLesson | undefined { return POST_TRAINING_COURSE.lessons.find((lesson) => lesson.id === lessonId);}
export function postTrainingLessonContext(lessonId: string, sectionId?: string): string { const lesson = postTrainingLesson(lessonId); if (!lesson) throw new Error("Course lesson is unknown"); const selectedSection = sectionId ? lesson.sections.find((section) => section.id === sectionId) : undefined; if (sectionId && !selectedSection) throw new Error("Course section is unknown"); const sections = selectedSection ? [selectedSection] : lesson.sections; const lines = [ `Course: ${POST_TRAINING_COURSE.title}`, `Course revision: ${POST_TRAINING_COURSE.revision}`, `Core claim: ${POST_TRAINING_COURSE.oneThing}`, `Lesson ${lesson.number}: ${lesson.title}`, `Lesson objective: ${lesson.objective}`, ]; for (const section of sections) { lines.push("", `Section: ${section.title}`, ...section.paragraphs); if (section.bullets) lines.push(...section.bullets.map((item) => `- ${item}`)); if (section.code) lines.push(`${section.code.caption}:`, section.code.code); if (section.callout) lines.push(`Key constraint: ${section.callout}`); } lines.push("", "Primary references:", ...lesson.references.map((reference) => ( `- ${reference.label}: ${reference.url} (${reference.use})` ))); return lines.join("\n");}
function deepFreeze<T>(value: T): T { if (value && typeof value === "object" && !Object.isFrozen(value)) { Object.freeze(value); for (const child of Object.values(value as Record<string, unknown>)) deepFreeze(child); } return value;}