Something went wrong. Try again.
My website. cameron.stream
Something went wrong. Try again.
21 kB · 496 lines
TSX
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497function FlowArrow() { return <span class="tinker-flow-arrow" aria-hidden="true">→</span>;}
const sftCode = `service = tinker.ServiceClient()trainer = service.create_lora_training_client( base_model="Qwen/Qwen3-8B", rank=32,)
for batch in dataset: future = await trainer.forward_backward_async( data=batch, loss_fn="cross_entropy", ) metrics = await future.result_async()
step = await trainer.optim_step_async( types.AdamParams(learning_rate=1e-4) ) await step.result_async()`;
const datumCode = `tokens = [prompt tokens..., answer tokens...]targets = tokens[1:]weights = [0, 0, 0, ..., 1, 1, 1, ...]
datum = types.Datum( model_input=types.ModelInput.from_ints(tokens[:-1]), loss_fn_inputs={ "target_tokens": targets, "weights": weights, },)`;
const rlCode = `while training: policy = trainer.save_weights_and_get_sampling_client() result = await policy.sample_async( prompt=problem, num_samples=8, sampling_params=params, )
rollouts = result.sequences rewards = [environment.score(x) for x in rollouts] baseline = sum(rewards) / len(rewards) advantages = [reward - baseline for reward in rewards]
await trainer.forward_backward_async( data=make_rl_data(rollouts, advantages), loss_fn="importance_sampling", ) await trainer.optim_step_async(optimizer)`;
export function TinkerGuide() { return ( <main class="tinker-page"> <nav class="tinker-nav" aria-label="Tinker guide navigation"> <a href="/artifacts">← ARTIFACTS</a> <div> <span>FIELD GUIDE / 001</span> <a href="#first-run">FIRST RUN</a> </div> </nav>
<header class="tinker-hero"> <div class="tinker-hero-copy"> <p class="artifact-kicker">THINKING MACHINES LAB</p> <h1>TINKER</h1> <p class="tinker-deck"> You write the learning experiment. Tinker makes the large model actually move. </p> </div> <div class="tinker-orbit" aria-hidden="true"> <span class="tinker-orbit-ring tinker-orbit-ring-outer"></span> <span class="tinker-orbit-ring tinker-orbit-ring-inner"></span> <span class="tinker-orbit-node tinker-orbit-data">DATA</span> <span class="tinker-orbit-node tinker-orbit-loss">LOSS</span> <span class="tinker-orbit-node tinker-orbit-sample">SAMPLE</span> <span class="tinker-orbit-core">∇</span> </div> <div class="tinker-hero-meta"> <span>PYTHON OUTSIDE</span> <span>ACCELERATORS INSIDE</span> <span>JUDGMENT THROUGHOUT</span> </div> </header>
<section class="tinker-section tinker-opening" aria-labelledby="opening-title"> <div class="tinker-section-label"> <span>00</span> <span id="opening-title">THE COMPRESSION</span> </div> <div class="tinker-opening-grid"> <h2>Tinker is an API boundary around distributed post-training.</h2> <div> <p> It is lower-level than “upload a dataset and receive a model,” and higher-level than assembling a GPU cluster. Your Python code owns the data, loss, rewards, rollout environment, and experimental logic. Thinking Machines owns placement, parallelism, scheduling, failure recovery, and moving giant tensors through hardware. </p> <p> That boundary is the product. It keeps the part where research judgment lives and removes the part where NCCL develops opinions about your evening. </p> </div> </div> </section>
<section class="tinker-section" aria-labelledby="boundary-title"> <div class="tinker-section-label"> <span>01</span> <span id="boundary-title">WHERE THE SYSTEM CUTS</span> </div> <div class="tinker-boundary"> <div class="tinker-boundary-side tinker-boundary-yours"> <p class="tinker-mini-label">YOUR PROCESS</p> <h2>The experiment</h2> <ul> <li>Examples and renderers</li> <li>Loss functions</li> <li>Reward and grading logic</li> <li>Rollout environments</li> <li>Evaluation and interpretation</li> </ul> </div> <div class="tinker-api-spine"> <span>SAMPLE</span> <span>FORWARD_BACKWARD</span> <strong>API</strong> <span>OPTIM_STEP</span> <span>SAVE_WEIGHTS</span> </div> <div class="tinker-boundary-side tinker-boundary-theirs"> <p class="tinker-mini-label">TINKER SERVICE</p> <h2>The machinery</h2> <ul> <li>Model sharding</li> <li>GPU allocation</li> <li>Distributed execution</li> <li>Checkpoint storage</li> <li>Failure recovery</li> </ul> </div> </div> <p class="tinker-caption"> Switching from an 8B dense model to a much larger mixture-of-experts model can be one string change because the hardware layout stays behind the boundary. </p> </section>
<section class="tinker-section" aria-labelledby="verbs-title"> <div class="tinker-section-label"> <span>02</span> <span id="verbs-title">THE FIVE VERBS</span> </div> <div class="tinker-verbs"> <article> <span>01</span> <h3>sample</h3> <p>Generate candidate tokens from a base model or your current adapter.</p> </article> <article> <span>02</span> <h3>compute_logprobs</h3> <p>Ask how probable tokens were under a particular policy.</p> </article> <article> <span>03</span> <h3>forward_backward</h3> <p>Evaluate a loss on data and accumulate gradients.</p> </article> <article> <span>04</span> <h3>optim_step</h3> <p>Use those gradients to update the trainable adapter.</p> </article> <article> <span>05</span> <h3>save_weights</h3> <p>Freeze a named checkpoint or turn it into a sampling client.</p> </article> </div> <p class="tinker-large-note"> Every method is a composition of these verbs. </p> </section>
<section class="tinker-section" aria-labelledby="sft-title"> <div class="tinker-section-label"> <span>03</span> <span id="sft-title">SUPERVISED FINE-TUNING</span> </div> <div class="tinker-explainer-head"> <h2>Show it the behavior you want.</h2> <p> SFT is next-token prediction over examples you chose. The model sees a prompt and completion; cross-entropy pushes probability toward the completion tokens. The intellectual work is mostly upstream: what counts as a good example, what context is included, and which tokens are allowed to teach. </p> </div> <div class="tinker-flow" aria-label="Supervised fine-tuning flow"> <span>EXAMPLE</span><FlowArrow /> <span>TOKENS + MASK</span><FlowArrow /> <span>CROSS-ENTROPY</span><FlowArrow /> <span>GRADIENT</span><FlowArrow /> <span>UPDATE</span> </div>
<div class="tinker-code-grid"> <article> <div class="tinker-code-header"> <span>THE LOOP</span> <span>PYTHON</span> </div> <pre><code>{sftCode}</code></pre> </article> <article> <div class="tinker-code-header"> <span>THE DATUM</span> <span>TOKENS</span> </div> <pre><code>{datumCode}</code></pre> </article> </div>
<aside class="tinker-concept-card"> <p class="tinker-mini-label">THE IMPORTANT LITTLE ARRAY</p> <h3>The loss mask decides what the example means.</h3> <p> A zero-weight prompt token supplies context but creates no direct training pressure. A one-weight answer token contributes to loss. The same text with a different mask is a different learning signal. Dataset design is therefore not clerical work. It is an executable claim about where expertise lives. </p> </aside> </section>
<section class="tinker-section tinker-lora-section" aria-labelledby="lora-title"> <div class="tinker-section-label tinker-section-label-dark"> <span>04</span> <span id="lora-title">LORA, WITHOUT THE FOG</span> </div> <div class="tinker-lora-grid"> <div> <p class="tinker-equation">W′ = W + γBA</p> <p class="tinker-equation-key"> ORIGINAL WEIGHTS + SMALL LEARNED UPDATE </p> </div> <div class="tinker-lora-copy"> <h2>Do not rewrite the whole model. Learn a compact change.</h2> <p> A model layer contains a large weight matrix <em>W</em>. LoRA keeps it fixed and learns two skinny matrices, <em>B</em> and <em>A</em>. Their product is low-rank: it can express a structured change using far fewer trainable parameters. </p> <p> Rank controls the dimensionality of that change, not the model’s intelligence. Too little rank can become a capacity bottleneck. More rank costs storage and training memory. Thinking Machines’ experiments found a broad low-regret regime for ordinary post-training, especially when LoRA is applied across all weight matrices rather than attention alone. </p> </div> </div> <div class="tinker-lora-facts"> <div><strong>CHEAPER</strong><span>Train and store a small adapter.</span></div> <div><strong>SWAPPABLE</strong><span>Many adapters can share one base.</span></div> <div><strong>ENOUGH</strong><span>Usually, when the dataset fits its capacity.</span></div> </div> </section>
<section class="tinker-section" aria-labelledby="rl-title"> <div class="tinker-section-label"> <span>05</span> <span id="rl-title">REINFORCEMENT LEARNING</span> </div> <div class="tinker-explainer-head"> <h2>Let the model act, then teach from consequences.</h2> <p> RL closes a loop. The current policy samples several attempts. An environment or grader scores them. Advantages express which attempts were better than their local baseline. Training increases the probability of better trajectories and decreases the probability of worse ones. </p> </div> <div class="tinker-rl-loop" aria-label="Reinforcement learning loop"> <div><span>1</span><strong>POLICY</strong><small>current adapter</small></div> <FlowArrow /> <div><span>2</span><strong>ROLLOUTS</strong><small>multiple attempts</small></div> <FlowArrow /> <div><span>3</span><strong>REWARD</strong><small>environment judgment</small></div> <FlowArrow /> <div><span>4</span><strong>ADVANTAGE</strong><small>relative signal</small></div> <FlowArrow /> <div><span>5</span><strong>UPDATE</strong><small>new policy</small></div> </div> <div class="tinker-rl-detail"> <article class="tinker-code-card-wide"> <div class="tinker-code-header"> <span>THE LOOP</span> <span>SCHEMATIC</span> </div> <pre><code>{rlCode}</code></pre> </article> <aside> <p class="tinker-mini-label">WHY LOGPROBS APPEAR</p> <p class="tinker-ratio">π<sub>new</sub>(token) / π<sub>rollout</sub>(token)</p> <p> Rollouts were produced by a snapshot of the policy. The learner may have moved since then. Importance sampling uses the probability ratio to correct for that distance. PPO and CISPO add different forms of restraint so a few surprising tokens do not throw the update across the room. </p> </aside> </div> </section>
<section class="tinker-section tinker-agents" aria-labelledby="agents-title"> <div class="tinker-section-label"> <span>06</span> <span id="agents-title">WHY AGENTS FIT</span> </div> <div class="tinker-agents-grid"> <h2>An agent already emits the raw material for post-training.</h2> <div class="tinker-trace"> <span>STATE</span><FlowArrow /> <span>THOUGHT</span><FlowArrow /> <span>TOOL</span><FlowArrow /> <span>OBSERVATION</span><FlowArrow /> <span>ANSWER</span> </div> <div class="tinker-agent-notes"> <article> <span>A</span> <h3>Trajectory as data</h3> <p>A whole attempt can become an example, a comparison, or a rollout.</p> </article> <article> <span>B</span> <h3>Human correction as signal</h3> <p>Edits, approvals, reversals, and “that is the wrong frame” can become labels.</p> </article> <article> <span>C</span> <h3>Environment as grader</h3> <p>Tests, receipts, task state, and user judgment can supply rewards.</p> </article> </div> </div> <p class="tinker-large-note"> The scarce thing is not text. It is trustworthy judgment attached to text. </p> </section>
<section class="tinker-section" aria-labelledby="eval-title"> <div class="tinker-section-label"> <span>07</span> <span id="eval-title">EVALUATION IS THE EXPERIMENT</span> </div> <div class="tinker-eval-intro"> <h2>Training loss tells you that learning happened. It does not tell you what was learned.</h2> <p> A run without held-out evaluation is an expensive anecdote. Decide what improvement means before the first optimizer step, then preserve examples the model never trains on. </p> </div> <div class="tinker-eval-grid"> <article><span>01</span><h3>Loss</h3><p>Did optimization move in the expected direction?</p></article> <article><span>02</span><h3>Task metric</h3><p>Did held-out accuracy, reward, or preference rate improve?</p></article> <article><span>03</span><h3>Behavior</h3><p>Did the model acquire the intended habit rather than a shortcut?</p></article> <article><span>04</span><h3>Regression</h3><p>What unrelated ability or style got worse?</p></article> </div> </section>
<section class="tinker-section tinker-first-run" id="first-run" aria-labelledby="first-title"> <div class="tinker-section-label tinker-section-label-dark"> <span>08</span> <span id="first-title">A FIRST EXPERIMENT WORTH RUNNING</span> </div> <div class="tinker-first-run-head"> <p class="tinker-mini-label">TRAJECTORY JUDGMENT / SFT FIRST</p> <h2>Can a small model learn to recognize the better agent run?</h2> <p> This is narrow enough to finish and close enough to real agent work to be diagnostic. It tests the entire Tinker surface without requiring a synthetic math environment or pretending that reward design is solved. </p> </div> <ol class="tinker-protocol"> <li> <span>01</span> <div><strong>Collect pairs</strong><p>Two attempts at the same task, plus a human choice and one-sentence reason.</p></div> </li> <li> <span>02</span> <div><strong>Render carefully</strong><p>Include the task, compact traces, outputs, and receipts. Remove irrelevant noise.</p></div> </li> <li> <span>03</span> <div><strong>Train a judge</strong><p>Start with SFT: predict A or B, then explain the decisive evidence.</p></div> </li> <li> <span>04</span> <div><strong>Hold out hard cases</strong><p>Especially cases where polish conflicts with correctness or claimed success lacks a receipt.</p></div> </li> <li> <span>05</span> <div><strong>Interrogate errors</strong><p>The useful output is not one score. It is a taxonomy of judgment the model failed to acquire.</p></div> </li> </ol> <div class="tinker-hypothesis"> <span>HYPOTHESIS</span> <p> Expert judgment will transfer where it is visible in concrete labels and reasons. It will fail where the “expertise” only exists as tacit context withheld from the training example. </p> </div> </section>
<section class="tinker-section" aria-labelledby="map-title"> <div class="tinker-section-label"> <span>09</span> <span id="map-title">THE LEARNING MAP</span> </div> <div class="tinker-map"> <article> <span>NOW</span> <h3>Build one SFT run</h3> <p>Tokenization, masks, batches, loss curves, checkpoints, held-out samples.</p> </article> <article> <span>NEXT</span> <h3>Compare objectives</h3> <p>SFT versus preference training on the same underlying judgments.</p> </article> <article> <span>THEN</span> <h3>Close the RL loop</h3> <p>Generate fresh trajectories, grade them, and train on-policy.</p> </article> <article> <span>FINALLY</span> <h3>Change the question</h3> <p>Use observed failures to decide what data and reward should exist next.</p> </article> </div> </section>
<section class="tinker-section tinker-glossary" aria-labelledby="glossary-title"> <div class="tinker-section-label"> <span>10</span> <span id="glossary-title">WORDS THAT STOP BEING MYSTERIOUS</span> </div> <dl> <div><dt>Base model</dt><dd>The frozen pretrained or instruction-tuned network an adapter modifies.</dd></div> <div><dt>Renderer</dt><dd>Code that turns structured examples or messages into the exact tokens a model sees.</dd></div> <div><dt>Logprob</dt><dd>The log of a model’s assigned probability to a token. Conveniently additive across a sequence.</dd></div> <div><dt>Gradient</dt><dd>The local direction in parameter space that changes the loss.</dd></div> <div><dt>Checkpoint</dt><dd>A named saved state: adapter weights, and sometimes optimizer state for resuming.</dd></div> <div><dt>On-policy</dt><dd>Training from trajectories generated by the current or very recent policy.</dd></div> <div><dt>Advantage</dt><dd>How much better an action or trajectory was than an expected baseline.</dd></div> <div><dt>Distillation</dt><dd>Training one model to reproduce information carried by another model’s outputs or probabilities.</dd></div> </dl> </section>
<footer class="tinker-footer"> <div> <span>PRIMARY SOURCES</span> <a href="https://tinker-docs.thinkingmachines.ai/tinker/quickstart/" target="_blank" rel="noopener">QUICK START ↗</a> <a href="https://github.com/thinking-machines-lab/tinker-cookbook" target="_blank" rel="noopener">COOKBOOK ↗</a> <a href="https://thinkingmachines.ai/blog/lora/" target="_blank" rel="noopener">LORA WITHOUT REGRET ↗</a> <a href="https://thinkingmachines.ai/news/learning-to-replicate-expert-judgment-in-financial-tasks/" target="_blank" rel="noopener">EXPERT JUDGMENT ↗</a> </div> <div class="tinker-footer-mark"> <strong>∇</strong> <span>LEARN THE CHANGE</span> </div> </footer> </main> );}