From 65b7e9f9fff52876057915104f70d5b85fd59b0d Mon Sep 17 00:00:00 2001 From: "prompt.ac/@jeffrey" Date: Wed, 16 Sep 2026 15:00:49 -0700 Subject: [PATCH] aesel: a meter for what a turn costs to run MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit the bridges already knew and threw it away — the usage block was on the wire. it is read per round now (a turn that called a tool paid for two responses), turned into watt-hours from the model's active parameters, and parked on the footer's gauge row beside the viewer count, where it is the first thing a narrow window gives up. `/energy` prints the working. it is an estimate and says so everywhere: nobody publishes per-token energy, so the slope is anchored to the only figures anyone has released (google's 0.24 Wh median prompt, epoch's ~0.3 Wh) and the ratios — which rest on announced active-parameter counts — are the half that can be defended. `/energy` leads with them: the same conversation priced across every model, cheapest first, with the closed ones saying their size is a guess. Co-Authored-By: Claude Opus 5 (1M context) --- easel/README.md | 19 ++- easel/src/about.mjs | 2 +- easel/src/ac-server.mjs | 18 ++ easel/src/claude-server.mjs | 12 ++ easel/src/energy.mjs | 264 ++++++++++++++++++++++++++++++ easel/src/render.mjs | 13 +- easel/src/tui.mjs | 17 +- easel/test/ac-server.test.mjs | 42 +++++ easel/test/claude-server.test.mjs | 26 +++ easel/test/energy.test.mjs | 118 +++++++++++++ easel/test/fake-claude-cli.mjs | 14 +- easel/test/render.test.mjs | 23 +++ 12 files changed, 561 insertions(+), 7 deletions(-) create mode 100644 easel/src/energy.mjs create mode 100644 easel/test/energy.test.mjs diff --git a/easel/README.md b/easel/README.md index 5f09aef06d..5c4fe5caf9 100644 --- a/easel/README.md +++ b/easel/README.md @@ -39,7 +39,7 @@ the piece currently being worked on in the header. Inside the TUI: `/login`, `/logout`, `/whoami`, `/publish [file] [slug]`, `/autopublish [on|off]`, `/piece [name]`, `/runtime [mjs|lisp|processing]`, `/backend [claude|codex]`, -`/model [name]`, `/qr`, `/live`, `/new`, `/clear`, `/help`, `/quit`. Press +`/model [name]`, `/energy`, `/qr`, `/live`, `/new`, `/clear`, `/help`, `/quit`. Press `ctrl-c` to interrupt a running turn or exit while idle. ## Engine bridges @@ -85,6 +85,23 @@ Ctrl-C cancels it. Browser rendering, rasterization and display latency are excluded. Unsupported APIs/imports report an error. It requires Node permission support (Node 24 or newer recommended). +`/energy` estimates what the session cost in electricity. Every bridge reports +the tokens it spent — per round on AC hosted, per turn from the Claude CLI's own +`modelUsage` — and `src/energy.mjs` turns those counts into watt-hours: a fixed +cost per generated token plus a part that scales with the model's *active* +parameters, with prompt tokens at a tenth of a generated one and cached tokens +at a hundredth. The running total shares the footer's gauge row with the viewer +count, wearing a `~`. + +It is an estimate and cannot be anything else — no provider publishes per-token +energy. The slope is anchored so a frontier-class answer lands near the only +published figures (Google's 0.24 Wh median text prompt; Epoch AI's ~0.3 Wh for a +GPT-4o query), and the open-weight hosted models carry their announced active +parameter counts, so the *relative* half of the readout — the same conversation +priced across every model, cheapest first — rests on published numbers rather +than on guessed hardware. Rows for closed models say that their size is a guess. +Serving only: no training, no water, and not your own machine. + The Claude bridge runs `claude --print --input-format stream-json --output-format stream-json`, the same headless protocol the Claude Agent SDK speaks, driven directly over a pipe. That is why Easel still has no diff --git a/easel/src/about.mjs b/easel/src/about.mjs index 56fc76e9a5..fa7db3a4af 100644 --- a/easel/src/about.mjs +++ b/easel/src/about.mjs @@ -10,7 +10,7 @@ export function aboutMap() { " └─ Your Codex account · /backend codex", "", "MAKE /medium picture|sound|paper|gameboy|piece", - "MEASURE /performance · headless logic and drawing-call counts", + "MEASURE /performance · headless logic · /energy · estimated electricity", "WORK /artifacts · /select UUID · /artifact", "OUTPUT /open · /export FILE · Pieces: /publish · /qr", "ACCOUNT /login · /profile · /logout", diff --git a/easel/src/ac-server.mjs b/easel/src/ac-server.mjs index 06d0ae067f..6f1de69f2b 100644 --- a/easel/src/ac-server.mjs +++ b/easel/src/ac-server.mjs @@ -300,6 +300,11 @@ export class AcServer extends EventEmitter { // Tool arguments arrive as a JSON string in fragments, so they are gathered // per block index and parsed only once the block closes. const partials = new Map(); + // What this round cost. The counts arrive split across two events — + // `message_start` knows the prompt, `message_delta` knows the answer — and + // each is cumulative for its own field, so later values replace rather than + // add. The interface turns this into watt-hours; see energy.mjs. + const usage = {}; const reader = response.body.getReader(); const decoder = new TextDecoder(); @@ -328,6 +333,9 @@ export class AcServer extends EventEmitter { continue; } + const counts = event.usage || event.message?.usage; + if (counts) Object.assign(usage, counts); + if (event.type === "content_block_start" || event.type === "content_block_delta") { this.emit("notification", { method: "turn/progress", params: { phase: event.delta?.type === "input_json_delta" || event.content_block?.type === "tool_use" ? "composing" : "generating", @@ -379,6 +387,16 @@ export class AcServer extends EventEmitter { reader.releaseLock?.(); } + // Reported per round rather than per turn: a turn that called a tool paid + // for two responses, and a readout that showed one of them would understate + // the expensive kind of turn. + if (Object.keys(usage).length) { + this.emit("notification", { + method: "turn/usage", + params: { model: this.model, usage }, + }); + } + if (text) { this.emit("notification", { method: "item/completed", diff --git a/easel/src/claude-server.mjs b/easel/src/claude-server.mjs index 1d48dc6446..5a57b84451 100644 --- a/easel/src/claude-server.mjs +++ b/easel/src/claude-server.mjs @@ -566,6 +566,18 @@ export class ClaudeServer extends EventEmitter { const id = this.turnId || `turn-${this.turns}`; this.turnId = null; this.textItems.clear(); + // The CLI closes a turn with what it spent. `modelUsage` is keyed by the + // model that actually ran — which is not always the one asked for, and a + // fallback is exactly when the energy readout should not lie about which + // model it is describing. + const perModel = Object.entries(message.modelUsage || {}); + if (perModel.length) { + for (const [model, usage] of perModel) { + this.emit("notification", { method: "turn/usage", params: { model, usage } }); + } + } else if (message.usage) { + this.emit("notification", { method: "turn/usage", params: { model: this.model, usage: message.usage } }); + } const aborted = String(message.terminal_reason || "").startsWith("aborted"); const status = aborted ? "interrupted" : message.is_error ? "failed" : "completed"; const turn = { id, status, items: [] }; diff --git a/easel/src/energy.mjs b/easel/src/energy.mjs new file mode 100644 index 0000000000..a42425f3fc --- /dev/null +++ b/easel/src/energy.mjs @@ -0,0 +1,264 @@ +// energy.mjs — roughly what a turn cost in electricity. +// +// A session here is someone making a picture by talking to a machine in a +// datacenter, and nothing in the interface has ever said what that costs to +// run. Tokens are the wrong unit for the question: they are the provider's +// billing unit, they are not comparable between models, and nobody has an +// intuition for twelve thousand of them. Watt-hours are a unit people already +// own — a lightbulb, a kettle, a phone charge. +// +// What this is honest about: it is an estimate, and it cannot be anything else. +// No hosted provider publishes per-token energy, and the two frontier labs that +// have published anything at all published a per-prompt median, not a model +// card. So the number here is derived, and the derivation is written down so it +// can be argued with: +// +// 1. Generating one token runs the model's *active* parameters once. For the +// open-weight models Aesel hosts, that count is published (a mixture of +// experts announces both numbers: GLM-4.6 is 355B total, 32B active), so +// the models differ by a factor this formula can actually see. +// 2. Energy per output token is taken as a fixed part plus a part that scales +// with active parameters. The fixed part is everything that does not care +// how big the model is — host CPU, memory, networking, cooling, and the +// share of an idle-but-provisioned accelerator — which published +// full-stack figures put at a large fraction of the total. +// 3. The slope is anchored so a frontier-class model lands near the only +// measurements anyone has released: Google's median text prompt at 0.24 Wh +// (Aug 2025, full-stack, including idle and overhead), Epoch AI's estimate +// of ~0.3 Wh for a GPT-4o query, and OpenAI's own ~0.34 Wh average. At a +// few hundred output tokens per answer those all land at 1–3 J per token. +// 4. Reading the prompt is cheap per token compared to writing the answer. +// Prefill runs dense and batched; decode is memory-bound and runs one +// token at a time. An input token is counted at a tenth of an output +// token, a cached one at a hundredth — cheap, but not free, because the +// cache still has to be read out of memory and attended to. +// +// What it excludes: training, the water, the embodied cost of the hardware, and +// your own laptop. It is the marginal electricity of serving the turn. +// +// Two consequences for how this gets shown. Absolute watt-hours carry a +// precision they have not earned, so every number that reaches a person wears a +// `~`. And the comparison that *is* defensible is the relative one — the same +// conversation on a 32B-active model and on a frontier model differ by a factor +// the formula derives from published parameter counts rather than from guessed +// hardware — so `/energy` leads with the ratio and treats the watt-hours as the +// supporting detail. + +// Joules per output token: a floor that every model pays, plus a slope on +// billions of active parameters. Anchored at 200B active ≈ 2.4 J/token, which +// puts a 300-token answer at 0.2 Wh — inside the published per-prompt range. +const FIXED_J = 0.6; +const PER_BILLION_J = 0.009; + +// What a token of each other kind costs, as a share of one output token. +const SHARE = { + input: 0.1, + cacheWrite: 0.125, // prefill, plus writing the block out. + cacheRead: 0.01, +}; + +// The basis line every readout carries, so the numbers are never mistaken for +// measurements. +export const BASIS = + "Estimated from active parameters and published per-prompt figures (Google 0.24 Wh median; Epoch ~0.3 Wh). Serving only — no training, water or your own machine."; + +// Active parameters in billions. `known` marks the difference between a number +// the lab published and one this file guessed, because that difference is the +// whole reason to trust or distrust a row. +// +// The hosted models are open-weight mixtures of experts and announce both +// counts. The closed ones announce nothing, so they are placed by class — which +// is a guess, and says so wherever it is printed. +const HOSTED = { + "z-ai/glm-4.6": { label: "glm", active: 32, known: true }, + "qwen/qwen3-coder": { label: "qwen", active: 35, known: true }, + "deepseek/deepseek-chat-v3.1": { label: "deepseek", active: 37, known: true }, + "anthropic/claude-sonnet-4.6": { label: "sonnet", active: 200, known: false }, + "openai/gpt-5.4": { label: "gpt", active: 300, known: false }, +}; + +// Vendor-CLI models, matched by family. A session on `/backend claude` can name +// any model its subscription allows, so the fallback has to be a family rather +// than a list — and an unrecognized name is placed at the frontier class rather +// than at the cheap end, so an unknown model is never flattered. +const FAMILIES = [ + [/opus/i, { active: 500, known: false }], + [/fable/i, { active: 250, known: false }], + [/sonnet/i, { active: 200, known: false }], + [/haiku/i, { active: 40, known: false }], + [/gpt-5|o[34]|codex/i, { active: 300, known: false }], + [/mini|flash|small|lite/i, { active: 40, known: false }], +]; + +const UNKNOWN = { active: 200, known: false }; + +export function profileFor(model) { + const id = String(model || "").trim(); + if (Object.hasOwn(HOSTED, id)) return { id, ...HOSTED[id] }; + // `/model glm` names a model by the short name the hosted endpoint allowlists + // under. The bridge resolves it before reporting, but a session that never + // heard back from one still knows what it asked for. + for (const [hosted, profile] of Object.entries(HOSTED)) { + if (profile.label === id.toLowerCase()) return { id: hosted, ...profile }; + } + for (const [pattern, profile] of FAMILIES) { + if (pattern.test(id)) return { id, label: id, ...profile }; + } + return { id, label: id || "unknown", ...UNKNOWN }; +} + +function perOutputToken(active) { + return FIXED_J + PER_BILLION_J * active; +} + +// Providers name these fields differently — Anthropic's cache pair, Codex's +// `cached_input_tokens` — so every caller can hand over whatever it was given. +export function readUsage(usage = {}) { + const number = (value) => (Number.isFinite(value) && value > 0 ? Math.round(value) : 0); + return { + input: number(usage.input_tokens ?? usage.inputTokens), + output: number(usage.output_tokens ?? usage.outputTokens), + cacheRead: number( + usage.cache_read_input_tokens ?? + usage.cacheReadInputTokens ?? + usage.cached_input_tokens ?? + usage.cachedInputTokens, + ), + cacheWrite: number(usage.cache_creation_input_tokens ?? usage.cacheCreationInputTokens), + }; +} + +export function joulesFor(tokens, model) { + const { active } = profileFor(model); + const perToken = perOutputToken(active); + return ( + tokens.output * perToken + + tokens.input * perToken * SHARE.input + + tokens.cacheWrite * perToken * SHARE.cacheWrite + + tokens.cacheRead * perToken * SHARE.cacheRead + ); +} + +export function formatJoules(joules) { + if (!(joules > 0)) return "0 Wh"; + const wh = joules / 3600; + if (wh < 0.001) return `${joules.toFixed(0)} J`; + if (wh < 0.1) return `${wh.toFixed(3)} Wh`; + if (wh < 10) return `${wh.toFixed(2)} Wh`; + return `${wh.toFixed(1)} Wh`; +} + +// One everyday equivalent, picked so the number in front of it is small enough +// to picture. Watt-hours are a unit people own once something is plugged into +// them. +const APPLIANCES = [ + { watts: 10, name: "an LED bulb" }, + { watts: 30, name: "a laptop" }, + { watts: 2000, name: "an electric kettle" }, +]; + +export function everyday(joules) { + if (!(joules > 0)) return ""; + // The bulb answers almost everything a session can spend, which is the point: + // one appliance across a whole range keeps consecutive readouts comparable. + // Bigger draws are borrowed only once the bulb's own number stops being + // pictureable. + for (const { watts, name } of APPLIANCES) { + const seconds = joules / watts; + if (seconds <= 120) return `${name} for ${seconds < 10 ? seconds.toFixed(1) : seconds.toFixed(0)} s`; + const minutes = seconds / 60; + if (minutes <= 90) return `${name} for ${minutes.toFixed(0)} min`; + } + const hours = joules / 2000 / 3600; + return `an electric kettle for ${hours.toFixed(1)} h`; +} + +// A phone charge is the other intuition people have, and it is the one that +// makes a whole session legible rather than a single turn. +export function phoneCharges(joules, wattHours = 15) { + return joules / 3600 / wattHours; +} + +// The tally a session keeps. Per-model, because switching models mid-session is +// one command and the point of the whole readout is that the choice matters. +export class Energy { + constructor() { + this.joules = 0; + this.turns = 0; + this.tokens = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }; + this.byModel = new Map(); + } + + get counted() { + return this.tokens.input + this.tokens.output + this.tokens.cacheRead + this.tokens.cacheWrite; + } + + // `usage` is whatever the bridge was handed; `model` is what ran it. + add(model, usage) { + const tokens = readUsage(usage); + if (!(tokens.input + tokens.output + tokens.cacheRead + tokens.cacheWrite)) return 0; + const joules = joulesFor(tokens, model); + this.joules += joules; + this.turns += 1; + for (const key of Object.keys(this.tokens)) this.tokens[key] += tokens[key]; + const id = String(model || "unknown"); + const seen = this.byModel.get(id) || { joules: 0, tokens: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } }; + seen.joules += joules; + for (const key of Object.keys(seen.tokens)) seen.tokens[key] += tokens[key]; + this.byModel.set(id, seen); + return joules; + } +} + +// What this session's tokens would have cost on each hosted model, cheapest +// first and expressed as a multiple of the cheapest. This is the defensible +// half of the estimate: the ratios come from published active-parameter counts, +// so they hold even if the absolute watt-hours are off by a factor. +export function relativeModels(tokens, current = "") { + const rows = Object.keys(HOSTED).map((id) => { + const profile = profileFor(id); + return { ...profile, joules: joulesFor(tokens, id), current: id === current }; + }); + rows.sort((a, b) => a.joules - b.joules); + const floor = rows[0]?.joules || 0; + for (const row of rows) row.ratio = floor > 0 ? row.joules / floor : 1; + return rows; +} + +// The `/energy` readout, as lines. Built here rather than in the interface so +// the wording and the caveat travel with the arithmetic. +export function energyReport(energy, model = "") { + if (!energy || !energy.counted) { + return [ + "No metered turns yet — energy is counted from the usage the engine reports.", + BASIS, + ]; + } + const { tokens } = energy; + const lines = [ + `~${formatJoules(energy.joules)} this session · ${energy.turns} metered turn${energy.turns === 1 ? "" : "s"} · ${everyday(energy.joules)}`, + `${tokens.output.toLocaleString()} written · ${tokens.input.toLocaleString()} read · ${tokens.cacheRead.toLocaleString()} cached`, + ]; + const charges = phoneCharges(energy.joules); + if (charges >= 0.01) lines.push(`About ${charges < 1 ? `${(charges * 100).toFixed(0)}% of` : `${charges.toFixed(1)}×`} a phone charge.`); + + if (energy.byModel.size > 1) { + lines.push(""); + for (const [id, seen] of energy.byModel) { + lines.push(` ${(profileFor(id).label || id).padEnd(9)} ~${formatJoules(seen.joules)}`); + } + } + + lines.push(""); + lines.push("Same conversation, other models:"); + for (const row of relativeModels(tokens, model)) { + const bar = "█".repeat(Math.max(1, Math.min(24, Math.round(row.ratio * 3)))); + lines.push( + ` ${(row.label || row.id).padEnd(9)} ${bar} ${row.ratio.toFixed(1)}× · ~${formatJoules(row.joules)}` + + `${row.known ? "" : " (size undisclosed; estimated)"}${row.current ? " ← running" : ""}`, + ); + } + lines.push(""); + lines.push(BASIS); + return lines; +} diff --git a/easel/src/render.mjs b/easel/src/render.mjs index 4968440f3e..f807f02483 100644 --- a/easel/src/render.mjs +++ b/easel/src/render.mjs @@ -10,6 +10,7 @@ import { join } from "node:path"; import { MASCOT_HEIGHT, MASCOT_ROW_WIDTH, mascotAt, mascotRow } from "./mascot.mjs"; import { handleCharacterColors } from "./handle-colors.mjs"; import { aboutMap } from "./about.mjs"; +import { formatJoules } from "./energy.mjs"; const ESCAPE = /\x1b(?:\[[0-?]*[ -/]*[@-~]|\][^\x07]*(?:\x07|\x1b\\))/g; const CONTROLS = /[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]/g; @@ -353,8 +354,9 @@ export function renderBoot(elapsed = 0, columns = 80, rows = 24, useColor = true .join("\n"); } -// The readout for everything happening on the far side of the QR code: how many -// people are at the piece, and what their browsers are painting. Parts fall off +// The gauge row: everything happening on the far side of the QR code — how many +// people are at the piece, and what their browsers are painting — and, last, the +// running electricity estimate for the session. Parts fall off // the right as the window narrows, worst news first — a blank frame outranks a // viewer count, because it is the one thing here that means something is wrong. // @@ -385,6 +387,11 @@ export function audienceReadout(state, room = 80, useColor = true) { parts.push({ text: `${frame.colors} colors`, tone: "muted" }); if (Number.isFinite(state?.online)) parts.push({ text: `${state.online} on AC`, tone: "muted" }); + // Last, so it is the first thing the row gives up when the window narrows: a + // running estimate is the least urgent number here. The tilde is load-bearing + // — see energy.mjs on why this is an estimate and can only be one. + if (state?.energy > 0) + parts.push({ text: `~${formatJoules(state.energy)}`, tone: "muted" }); if (parts.length === 0) return { plain: "", painted: "" }; @@ -473,7 +480,7 @@ export function renderFrame(state, columns = 80, rows = 24, useColor = true) { // eye skips after the first second, and the count is the one number in the // interface that changes because of somebody else. const audience = audienceReadout( - { ...state.audience, frame: state.health?.frame }, + { ...state.audience, frame: state.health?.frame, energy: state.energy?.joules }, Math.max(0, width - 4 - textWidth(state.workspace || "workspace")), useColor, ); diff --git a/easel/src/tui.mjs b/easel/src/tui.mjs index 2027057d8b..0c8f3e8cb5 100755 --- a/easel/src/tui.mjs +++ b/easel/src/tui.mjs @@ -24,6 +24,7 @@ import { RuntimeFeedback, readRuntimeFeedback, runtimeFeedbackContext } from "./ import { createHash } from "node:crypto"; import { Diagnostics } from "./diagnostics.mjs"; import { EASEL_HEIGHT, easelFrame, easelNextFrame, easelWidth } from "./easel.mjs"; +import { Energy, energyReport } from "./energy.mjs"; import {codexModels,drawerKey,drawerIndex} from "./provider-picker.mjs"; import { backendFor, backendMenu, DEFAULT_BACKEND } from "./backends.mjs"; import { LivePiece } from "./live.mjs"; @@ -167,6 +168,10 @@ const state = { // What those people's browsers are actually showing — a blank frame, an // uncaught error. Null until the relay lets this session listen. health: null, + // What the session has spent in electricity, as far as the token counts the + // engine reports can say. `/energy` prints the working; energy.mjs holds the + // arithmetic and the caveat. + energy: new Energy(), qr: null, // The prompt rock in the menu bar draws this session's code at real pixel // resolution, so the transcript does not spend seventeen rows on a worse @@ -834,6 +839,9 @@ function handleNotification({ method, params = {} }) { engine.turnId = params.turn?.id || engine.turnId; slabSession.working(); break; + case "turn/usage": + state.energy.add(params.model || state.model || model, params.usage); + break; case "turn/progress": state.status = params.phase || "working"; state.progressBytes = params.bytes || state.progressBytes || 0; @@ -879,6 +887,9 @@ function handleNotification({ method, params = {} }) { break; } case "turn/completed": { + // Codex reports what it spent on the turn that closes rather than in a + // message of its own, so the meter reads it from here when it is there. + if (params.turn?.usage) state.energy.add(params.turn.model || state.model || model, params.turn.usage); state.busy = false; state.status = params.turn?.status === "failed" ? "failed" : "ready"; engine.turnId = null; @@ -1460,6 +1471,10 @@ async function submitInput() { return artifactOperation(()=>artifacts.run('generate',{provider:'openai',model:'gpt-image-2',prompt:rest,...(command==='/edit-image'?{reference:'composite.png'}:{})},{paid:true})); } if (command === "/performance" || command === "/perf") return commandPerformance(rest); + if (command === "/energy" || command === "/power") { + addEntry("notice", energyReport(state.energy, state.model || model).join("\n")); + return redraw(); + } if (command === "/latest") { state.scrollOffset = 0; return redraw(); } if (command === "/clear") { archivedConversation.push(...state.entries.filter(({ kind }) => kind === "user" || kind === "assistant")); @@ -1546,7 +1561,7 @@ async function submitInput() { if (command === "/help") { addEntry( "notice", - "/about · /medium · /artifacts · /select UUID · /artifact · /export FILE · /sharing · /transcript · /profile · /mouse [on|off] · /performance [frames] · /latest · /login · /logout · /whoami · /publish [file] · /autopublish [on|off] · /ask [on|off] · /piece [name] · /versions · /rollback vN · /runtime [id] · /frame [ocr] · /settings · /backend [id] · /model [name] · /effort · /handle [name] · /update · /open · /qr · /new [thread] · /clear · /quit ctrl-c interrupts a running turn", + "/about · /medium · /artifacts · /select UUID · /artifact · /export FILE · /sharing · /transcript · /profile · /mouse [on|off] · /performance [frames] · /energy · /latest · /login · /logout · /whoami · /publish [file] · /autopublish [on|off] · /ask [on|off] · /piece [name] · /versions · /rollback vN · /runtime [id] · /frame [ocr] · /settings · /backend [id] · /model [name] · /effort · /handle [name] · /update · /open · /qr · /new [thread] · /clear · /quit ctrl-c interrupts a running turn", ); return redraw(); } diff --git a/easel/test/ac-server.test.mjs b/easel/test/ac-server.test.mjs index 59deba3107..db317a4113 100644 --- a/easel/test/ac-server.test.mjs +++ b/easel/test/ac-server.test.mjs @@ -218,3 +218,45 @@ test("interrupting a checkpoint during validation cannot write or start another assert.equal(calls, 1); assert.equal(completed.status, "interrupted"); }); +// A turn that called a tool paid for two responses. The meter has to see both, +// or the readout understates exactly the turns that cost the most. +test("each round reports what it spent, per round rather than per turn", async () => { + const dir = await mkdtemp(join(tmpdir(), "ac-energy-")); + const file = join(dir, "vopuzi.mjs"); + await writeFile(file, "// blank\n"); + + const metered = (events, output) => [ + { type: "message_start", message: { usage: { input_tokens: 6000, cache_read_input_tokens: 24000 } } }, + ...events, + { type: "message_delta", delta: { stop_reason: events === none ? "end_turn" : "tool_use" }, usage: { output_tokens: output } }, + ]; + const none = []; + + const engine = new AcServer({ + fetch: serving( + metered([ + { type: "content_block_start", index: 0, content_block: { type: "tool_use", id: "t1", name: "write_piece" } }, + { type: "content_block_delta", index: 0, delta: { type: "input_json_delta", partial_json: JSON.stringify({ source: "function paint({ wipe }) { wipe(0); }" }) } }, + { type: "content_block_stop", index: 0 }, + ], 700), + metered(none, 40), + ), + token: async () => "tok", + piece: { file }, + model: "glm", + }); + + const spent = []; + engine.on("notification", ({ method, params }) => { + if (method === "turn/usage") spent.push(params); + }); + await engine.connect(); + await engine.startTurn("paint it black"); + + assert.equal(spent.length, 2, "one report per round"); + assert.equal(spent[0].model, "z-ai/glm-4.6", "reported under the id that ran, not the alias"); + assert.equal(spent[0].usage.output_tokens, 700); + assert.equal(spent[0].usage.cache_read_input_tokens, 24000, "prompt counts from message_start survive the round"); + assert.equal(spent[1].usage.output_tokens, 40); + await rm(dir, { recursive: true, force: true }); +}); diff --git a/easel/test/claude-server.test.mjs b/easel/test/claude-server.test.mjs index 41d430b1e3..26b8e1d8dc 100644 --- a/easel/test/claude-server.test.mjs +++ b/easel/test/claude-server.test.mjs @@ -236,3 +236,29 @@ test('missing unsaved Claude session recovers only with explicit Easel handoff', const c=await engine.connect();assert.notEqual(c.thread.id,'missing');assert.equal(fatals,0); const args=launches(argvFile).argvs;assert.equal(args.length,2);assert.match(flagIn(args[1],'--append-system-prompt'),/Retained Easel conversation/); }); +// Energy is estimated from token counts, so the counts have to arrive — and +// under the name of the model that actually ran. The fake CLI reports Fable +// while the bridge asked for the default, which is the case a readout keyed to +// the requested model would describe wrongly. +test("a finished turn reports what it spent, keyed to the model that ran", async (t) => { + const engine = bridge(t); + const spent = []; + const completed = new Promise((resolve) => { + engine.on("notification", ({ method, params }) => { + if (method === "turn/usage") spent.push(params); + if (method === "turn/completed") resolve(params.turn.status); + }); + }); + engine.on("request", (request) => engine.respond(request.id, { decision: "accept" })); + + await engine.connect(); + await engine.startTurn("make a piece"); + assert.equal(await completed, "completed"); + + assert.equal(spent.length, 1); + assert.equal(spent[0].model, "claude-fable-5-1"); + assert.equal(spent[0].usage.outputTokens, 300); + + const { joulesFor, readUsage } = await import("../src/energy.mjs"); + assert.ok(joulesFor(readUsage(spent[0].usage), spent[0].model) > 0); +}); diff --git a/easel/test/energy.test.mjs b/easel/test/energy.test.mjs new file mode 100644 index 0000000000..cf0be7bc78 --- /dev/null +++ b/easel/test/energy.test.mjs @@ -0,0 +1,118 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { + Energy, + energyReport, + everyday, + formatJoules, + joulesFor, + phoneCharges, + profileFor, + readUsage, + relativeModels, +} from "../src/energy.mjs"; + +test("published active-parameter counts are marked apart from guessed ones", () => { + assert.equal(profileFor("z-ai/glm-4.6").known, true); + assert.equal(profileFor("anthropic/claude-sonnet-4.6").known, false); + // An unrecognized name lands at the frontier class rather than the cheap end. + assert.ok(profileFor("some-unreleased-thing").active >= 200); + assert.equal(profileFor("claude-haiku-4-5-20251001").active, 40); +}); + +test("usage arrives under several spellings and reads the same", () => { + const anthropic = readUsage({ + input_tokens: 10, + output_tokens: 20, + cache_read_input_tokens: 30, + cache_creation_input_tokens: 40, + }); + const cli = readUsage({ + inputTokens: 10, + outputTokens: 20, + cacheReadInputTokens: 30, + cacheCreationInputTokens: 40, + }); + assert.deepEqual(anthropic, cli); + assert.deepEqual(readUsage({ input_tokens: 5, cached_input_tokens: 7 }), { + input: 5, + output: 0, + cacheRead: 7, + cacheWrite: 0, + }); + assert.deepEqual(readUsage({ input_tokens: -1, output_tokens: null }).input, 0); +}); + +test("writing a token costs more than reading one, and reading a cached one costs least", () => { + const model = "z-ai/glm-4.6"; + const out = joulesFor(readUsage({ output_tokens: 1000 }), model); + const read = joulesFor(readUsage({ input_tokens: 1000 }), model); + const cached = joulesFor(readUsage({ cache_read_input_tokens: 1000 }), model); + assert.ok(out > read && read > cached); + assert.ok(cached > 0, "a cached token is cheap, not free"); +}); + +test("a bigger model costs more for identical work", () => { + const tokens = readUsage({ input_tokens: 6000, output_tokens: 800 }); + const small = joulesFor(tokens, "z-ai/glm-4.6"); + const large = joulesFor(tokens, "openai/gpt-5.4"); + assert.ok(large > small * 2); +}); + +// The anchor the whole formula hangs on: a few hundred output tokens from a +// frontier-class model should land near the published per-prompt figures +// (Google 0.24 Wh median, Epoch ~0.3 Wh), or the estimate has drifted into +// numbers nobody has evidence for. +test("a frontier-class answer lands inside the published per-prompt range", () => { + const wh = joulesFor(readUsage({ input_tokens: 400, output_tokens: 300 }), "openai/gpt-5.4") / 3600; + assert.ok(wh > 0.05 && wh < 0.6, `${wh} Wh is outside the published range`); +}); + +test("the ledger totals per session and per model", () => { + const energy = new Energy(); + energy.add("z-ai/glm-4.6", { input_tokens: 100, output_tokens: 50 }); + energy.add("openai/gpt-5.4", { input_tokens: 100, output_tokens: 50 }); + assert.equal(energy.turns, 2); + assert.equal(energy.tokens.output, 100); + assert.equal(energy.byModel.size, 2); + assert.ok(energy.byModel.get("openai/gpt-5.4").joules > energy.byModel.get("z-ai/glm-4.6").joules); + // A turn the engine reported no counts for is not a turn the meter saw. + assert.equal(energy.add("z-ai/glm-4.6", {}), 0); + assert.equal(energy.turns, 2); +}); + +test("the relative table is cheapest-first and normalized to it", () => { + const rows = relativeModels(readUsage({ output_tokens: 500 }), "z-ai/glm-4.6"); + assert.equal(rows[0].ratio, 1); + assert.ok(rows.at(-1).ratio > 1); + assert.deepEqual([...rows].sort((a, b) => a.joules - b.joules), rows); + assert.equal(rows.find((row) => row.current).label, "glm"); +}); + +test("numbers are shown in units a person owns", () => { + assert.match(formatJoules(0), /^0 Wh$/); + assert.match(formatJoules(200), /Wh$/); + assert.match(formatJoules(36000), /^10\.0 Wh$/); + assert.match(everyday(100), /LED bulb/); + assert.match(everyday(500000), /kettle/); + assert.ok(phoneCharges(54000) > 0.9 && phoneCharges(54000) < 1.1); +}); + +test("the report says it is an estimate, with or without metered turns", () => { + assert.match(energyReport(new Energy()).join("\n"), /No metered turns yet/); + const energy = new Energy(); + energy.add("z-ai/glm-4.6", { input_tokens: 6000, output_tokens: 900, cache_read_input_tokens: 24000 }); + const report = energyReport(energy, "z-ai/glm-4.6").join("\n"); + assert.match(report, /this session/); + assert.match(report, /Same conversation, other models/); + assert.match(report, /← running/); + assert.match(report, /Estimated from active parameters/); + assert.match(report, /size undisclosed/, "guessed rows say so"); +}); + +// `/model glm` names a model by its short hosted name. A session that never +// heard the resolved id back should still price the right model. +test("hosted short names resolve to the model they select", () => { + assert.equal(profileFor("glm").active, profileFor("z-ai/glm-4.6").active); + assert.equal(profileFor("deepseek").known, true); +}); diff --git a/easel/test/fake-claude-cli.mjs b/easel/test/fake-claude-cli.mjs index 1bc17a75a4..aea9aa259b 100644 --- a/easel/test/fake-claude-cli.mjs +++ b/easel/test/fake-claude-cli.mjs @@ -121,6 +121,18 @@ createInterface({ input: process.stdin }).on("line", (line) => { ], }, }); - send({ type: "result", subtype: "success", is_error: false, terminal_reason: "completed", result: "done" }); + // The real CLI closes a turn with what it spent, both totalled and keyed by + // the model that actually ran. + send({ + type: "result", + subtype: "success", + is_error: false, + terminal_reason: "completed", + result: "done", + usage: { input_tokens: 1200, output_tokens: 300, cache_read_input_tokens: 18000, cache_creation_input_tokens: 0 }, + modelUsage: { + "claude-fable-5-1": { inputTokens: 1200, outputTokens: 300, cacheReadInputTokens: 18000, cacheCreationInputTokens: 0 }, + }, + }); } }); diff --git a/easel/test/render.test.mjs b/easel/test/render.test.mjs index 5735735d78..0da9d9210f 100644 --- a/easel/test/render.test.mjs +++ b/easel/test/render.test.mjs @@ -235,3 +235,26 @@ test("a blank frame is reported, and outranks the rest of the readout", async () const nothing = audienceReadout({ here: null, frame: null }, 80, false); assert.equal(nothing.plain, "", "and an unanswered session still claims nothing"); }); +// The running electricity estimate shares the gauge row, and is the first thing +// that row gives up: an estimate is the least urgent number on it. +test("the energy estimate reaches the gauge row and drops first when squeezed", async () => { + const { audienceReadout } = await import("../src/render.mjs"); + const { Energy } = await import("../src/energy.mjs"); + + const energy = new Energy(); + energy.add("z-ai/glm-4.6", { input_tokens: 6200, output_tokens: 900, cache_read_input_tokens: 24000 }); + + const frame = renderFrame( + { + workspace: "/project", mode: "remote", status: "ready", + account: "@tester", piece: "kizide.mjs", input: "", entries: [], energy, + }, + 100, 24, false, + ); + assert.match(frame, /~[\d.]+ Wh/, "the number wears a tilde, because it is an estimate"); + + const full = audienceReadout({ here: 2, peak: 9, energy: 3600 }, 80, false); + assert.equal(full.plain, "2 here · 9 peak · ~1.00 Wh"); + assert.equal(audienceReadout({ here: 2, peak: 9, energy: 3600 }, 16, false).plain, "2 here · 9 peak"); + assert.equal(audienceReadout({ energy: 0 }, 80, false).plain, "", "an unmetered session claims nothing"); +}); -- 2.51.2