From cccbc02d2c43c5672bc6227a890e29cd047c9bfc Mon Sep 17 00:00:00 2001 From: Iury Souza Date: Mon, 27 Jul 2026 23:59:39 +0200 Subject: [PATCH] feat(pi-cache-hit-predictor): show switch-impact stacked bar Replace the raw ~tokens/~tokens percentage with a compact context-window bar: green = carried cache, red = cache lost by switching, dim = remaining window. The percentage label is the source-lane cache drop.\n\n- Track active lane from the last successful assistant message.\n- Evaluate switch impact against that lane instead of the destination alone.\n- Fall back to the old concise text when context usage is unavailable.\n- Update tests and docs; bump to 0.2.0. --- ...07-28-cache-hit-predictor-switch-impact.md | 57 ++++++ package-lock.json | 2 +- packages/pi-cache-hit-predictor/CHANGELOG.md | 25 ++- packages/pi-cache-hit-predictor/README.md | 19 +- packages/pi-cache-hit-predictor/index.ts | 78 ++++++-- packages/pi-cache-hit-predictor/package.json | 2 +- .../pi-cache-hit-predictor/src/predictor.ts | 165 +++++++++++++++++ .../tests/extension.test.ts | 174 ++++++++++++++---- .../tests/predictor.test.ts | 160 +++++++++++++--- 9 files changed, 604 insertions(+), 78 deletions(-) create mode 100644 ai-artifacts/plans/2026-07-28-cache-hit-predictor-switch-impact.md diff --git a/ai-artifacts/plans/2026-07-28-cache-hit-predictor-switch-impact.md b/ai-artifacts/plans/2026-07-28-cache-hit-predictor-switch-impact.md new file mode 100644 index 0000000..5bd1d7a --- /dev/null +++ b/ai-artifacts/plans/2026-07-28-cache-hit-predictor-switch-impact.md @@ -0,0 +1,57 @@ +# Cache Hit Predictor: switch-impact stacked bar + +## Problem +The current footer shows a raw cache percentage (`~27k/~104k 26%`). That number answers an arithmetic question, not the decision question: *how much cache do I lose by switching, and how big is that loss relative to the destination model’s context window?* + +## Decision +Use a compact stacked context-window bar: + +```text +cache gpt-5.6-sol · low ↓60% [██████░░] +``` + +- Green segment = cache carried after the switch. +- Red segment = cache lost because of the switch. +- Dim segment = remaining context-window headroom. +- `↓60%` = percentage of the source-lane cache that is discarded. + +This encodes two dimensions without inventing a combined score: +1. **Continuity loss** — the percentage label. +2. **Absolute/window impact** — the visual length of the red segment inside the full window bar. + +## Behavior + +### When to show +- Show only on a model or reasoning-level switch that has a meaningful previous lane. +- Do not show on a brand-new chat (no source lane, no history). +- If source and destination are both cold, show nothing. +- If there is no loss, show the bar with a `↓0%` label and no red. +- Clear after a successful response on the displayed lane, or on session tree/compaction/shutdown. + +### Calculation +For a switch from source lane `S` to destination lane `D`: + +```text +sourceTokens = min(history[S], currentPromptTokens) +destTokens = min(history[D], currentPromptTokens) +lostTokens = max(0, sourceTokens - destTokens) +drop% = lostTokens / sourceTokens * 100 +window% = lostTokens / contextWindow * 100 +``` + +### Fallback +If `currentPromptTokens` or `contextWindow` is unavailable, fall back to the previous concise text format (`cache · cold` or `~tokens/~tokens pct%`). + +## Implementation + +### Files changed +- `packages/pi-cache-hit-predictor/src/predictor.ts`: add `predictCacheSwitchImpact`, `renderSwitchImpact`, segment allocation helper. +- `packages/pi-cache-hit-predictor/index.ts`: track `activeLane`, compute switch impact on model/thinking changes, render bar using `ctx.ui.theme`, keep legacy fallback. +- `packages/pi-cache-hit-predictor/tests/predictor.test.ts`: add impact and bar rendering tests. +- `packages/pi-cache-hit-predictor/tests/extension.test.ts`: update harness and assertions for the new status string. +- `packages/pi-cache-hit-predictor/README.md`, `CHANGELOG.md`, `package.json`: document and bump to 0.2.0. + +### Verification +- `npm run check` in the package. +- `npm run check` at the monorepo root. +- Update lockfile if the version bump requires it. diff --git a/package-lock.json b/package-lock.json index cf2cf54..2970566 100644 --- a/package-lock.json +++ b/package-lock.json @@ -11200,7 +11200,7 @@ }, "packages/pi-cache-hit-predictor": { "name": "@iurysza/pi-cache-hit-predictor", - "version": "0.1.0", + "version": "0.2.0", "license": "MIT", "devDependencies": { "@earendil-works/pi-coding-agent": "^0.80.10", diff --git a/packages/pi-cache-hit-predictor/CHANGELOG.md b/packages/pi-cache-hit-predictor/CHANGELOG.md index 425efa9..da28d86 100644 --- a/packages/pi-cache-hit-predictor/CHANGELOG.md +++ b/packages/pi-cache-hit-predictor/CHANGELOG.md @@ -1,9 +1,30 @@ -# @howaboua/pi-cache-hit-predictor +# @iurysza/pi-cache-hit-predictor + +## 0.2.0 + +### Changes + +- Replace the raw cache percentage (`~27k/~104k 26%`) with a stacked context-window bar. + + The new indicator shows how much cache continuity is lost by switching model or reasoning lanes, and how large that loss is relative to the destination model's context window. + + ```text + cache gpt-5.6-sol · low ↓60% [██████░░] + ``` + + - Green = carried cache. + - Red = cache lost by the switch. + - Dim = remaining context-window headroom. + - `↓60%` = percentage of the source lane's cache that is discarded. + +- Track the active lane from the last successful assistant message, so switches are evaluated against the lane that actually holds the likely reusable cache. + +- Keep a concise legacy fallback when context-window usage data is unavailable. ## 0.0.1 ### Changes -- [#140](https://github.com/IgorWarzocha/howaboua-pi-stuff/pull/140) [`c95d68a`](https://github.com/IgorWarzocha/howaboua-pi-stuff/commit/c95d68a21939860e4c6dcff9c58a6bf8a50044ff) Thanks [@IgorWarzocha](https://github.com/IgorWarzocha)! - Add inline cache-hit predictions when switching Pi model or reasoning lanes. +- Add inline cache-hit predictions when switching Pi model or reasoning lanes. Warn once that automatic reasoning-level changes can cause prompt-cache misses and affect provider costs or quotas. diff --git a/packages/pi-cache-hit-predictor/README.md b/packages/pi-cache-hit-predictor/README.md index f798238..0360682 100644 --- a/packages/pi-cache-hit-predictor/README.md +++ b/packages/pi-cache-hit-predictor/README.md @@ -1,12 +1,21 @@ # @iurysza/pi-cache-hit-predictor -Shows a cache-hit prediction when you switch Pi models or reasoning levels. +Shows a cache continuity indicator when you switch Pi models or reasoning levels. ```text -cache gpt-5.6-sol · low · ~27k/~104k 26% +cache gpt-5.6-sol · low ↓60% [██████░░] ``` -The prediction is UI-only. It is not sent to the model and does not change the prompt. It remains visible through aborts and failed requests, then clears after the first successful response on the predicted provider/API/model/reasoning lane. +The bar is a compact view of the destination model's context window: + +- **Green** = prompt cache that will be carried across the switch. +- **Red** = prompt cache that will be lost because of the switch. +- **Dim** = remaining context-window headroom. +- `↓60%` = percentage of the source lane's cache that the switch discards. + +This answers the decision question directly: how much of the current conversation stops carrying across the switch, and how large is that loss relative to the model's full window? + +The indicator is UI-only. It is not sent to the model and does not change the prompt. It remains visible through aborts and failed requests, then clears after the first successful response on the predicted provider/API/model/reasoning lane. The package uses Pi's native `setStatus()` API when installed alone. If `@iurysza/pi-ext` is also installed, it advertises priority metadata so pi-ext can place the same status in its bounded auxiliary footer line. @@ -24,9 +33,9 @@ pi -e npm:@iurysza/pi-cache-hit-predictor ## How it works -The extension treats each provider, API, model, and reasoning level combination as a separate cache lane. It remembers the prompt size from the latest successful request in each lane. When you return to one, the old prompt size is the prefix that may still be reusable. +The extension treats each provider, API, model, and reasoning level combination as a separate cache lane. It remembers the prompt size from the latest successful request in each lane. When you switch lanes, it compares the cache each lane would likely reuse. -For example, if `low` last saw 25k tokens and the session grew to 100k on `high`, switching back predicts a hit of up to 25k, or 25%. The next completed request refreshes that lane with the larger prompt. +For example, if `low` last saw 25k tokens and the session grew to 100k on `high`, switching back to `low` predicts a loss of up to 25k, or 100% of the `low` cache. The red segment in the bar is small because 25k is only a fraction of the model's window, but the percentage tells you the continuity cost is total. The next completed request on `low` refreshes that lane with the larger prompt. This is an estimate, not provider preflight data. Provider expiry or eviction, compaction, branch summaries, changed tools or system instructions, and provider serialization can change the actual hit. The provider's `cacheRead` usage remains the final result. diff --git a/packages/pi-cache-hit-predictor/index.ts b/packages/pi-cache-hit-predictor/index.ts index 873e4d6..36bdbc5 100644 --- a/packages/pi-cache-hit-predictor/index.ts +++ b/packages/pi-cache-hit-predictor/index.ts @@ -5,8 +5,12 @@ import type { import { type CacheLane, type CachePrediction, + type RenderTheme, + lastUsedLane, predictCacheHit, + predictCacheSwitchImpact, recordAssistantUsage, + renderSwitchImpact, scanCacheHistory, } from "./src/predictor.js"; import { createFooterSlotRegistration } from "./src/footer-slot.js"; @@ -63,6 +67,7 @@ export default function cacheHitPredictor(pi: ExtensionAPI) { let history = scanCacheHistory([]); let pendingPredictionTimer: ReturnType | undefined; let displayedLane: CacheLane | undefined; + let activeLane: CacheLane | undefined; const rebuild = (ctx: ExtensionContext) => { history = scanCacheHistory( @@ -71,25 +76,67 @@ export default function cacheHitPredictor(pi: ExtensionAPI) { ); }; + const setCurrentLane = (model: ModelIdentity) => { + activeLane = laneFor(model, pi.getThinkingLevel()); + }; + + const setActiveLaneFromHistory = (ctx: ExtensionContext) => { + activeLane = lastUsedLane(ctx.sessionManager.getBranch()); + }; + const clearPrediction = (ctx: ExtensionContext) => { displayedLane = undefined; ctx.ui.setStatus(STATUS_KEY, undefined); }; - const appendPrediction = ( + const legacyPredictionText = ( ctx: ExtensionContext, - model: ModelIdentity, - thinkingLevel: string, - ) => { - if (ctx.mode !== "tui") return; + lane: CacheLane, + ): string => { const contextTokens = ctx.getContextUsage()?.tokens ?? null; - const prediction = predictCacheHit( + const prediction = predictCacheHit(history, lane, contextTokens); + return predictionText(prediction); + }; + + const renderImpact = (ctx: ExtensionContext, dest: CacheLane): string | undefined => { + if (ctx.mode !== "tui") return undefined; + if (!activeLane || sameLane(activeLane, dest)) return undefined; + + const contextUsage = ctx.getContextUsage(); + const currentPromptTokens = contextUsage?.tokens ?? null; + const contextWindow = contextUsage?.contextWindow + ?? ctx.model?.contextWindow + ?? null; + + const impact = predictCacheSwitchImpact( history, - laneFor(model, thinkingLevel), - contextTokens, + activeLane, + dest, + currentPromptTokens, + contextWindow, ); - displayedLane = prediction.lane; - ctx.ui.setStatus(STATUS_KEY, predictionText(prediction)); + + if (impact.sourceTokens === 0 && impact.destTokens === 0) { + return undefined; + } + + if (currentPromptTokens === null || contextWindow === null) { + return legacyPredictionText(ctx, dest); + } + + const theme = (ctx.ui as { theme?: RenderTheme }).theme; + return renderSwitchImpact(impact, theme); + }; + + const showImpact = (ctx: ExtensionContext, dest: CacheLane) => { + const text = renderImpact(ctx, dest); + if (text) { + displayedLane = dest; + ctx.ui.setStatus(STATUS_KEY, text); + } else { + clearPrediction(ctx); + } + activeLane = dest; }; const schedulePrediction = ( @@ -97,10 +144,11 @@ export default function cacheHitPredictor(pi: ExtensionAPI) { model: ModelIdentity, thinkingLevel: string, ) => { + const dest = laneFor(model, thinkingLevel); if (pendingPredictionTimer) clearTimeout(pendingPredictionTimer); pendingPredictionTimer = setTimeout(() => { pendingPredictionTimer = undefined; - appendPrediction(ctx, model, thinkingLevel); + showImpact(ctx, dest); }, 0); }; @@ -108,16 +156,20 @@ export default function cacheHitPredictor(pi: ExtensionAPI) { footerSlot.register(); clearPrediction(ctx); rebuild(ctx); + setActiveLaneFromHistory(ctx); }); pi.on("session_tree", async (_event, ctx) => { clearPrediction(ctx); rebuild(ctx); + setActiveLaneFromHistory(ctx); }); pi.on("session_compact", async (_event, ctx) => { clearPrediction(ctx); rebuild(ctx); + setActiveLaneFromHistory(ctx); }); + pi.on("message_end", async (event, ctx) => { if (event.message.role !== "assistant") return; const responseLane = laneFor({ @@ -126,6 +178,7 @@ export default function cacheHitPredictor(pi: ExtensionAPI) { id: event.message.model, }, pi.getThinkingLevel()); recordAssistantUsage(history, event.message, responseLane); + activeLane = responseLane; if ( displayedLane && event.message.stopReason !== "aborted" @@ -140,16 +193,17 @@ export default function cacheHitPredictor(pi: ExtensionAPI) { }); pi.on("model_select", async (event, ctx) => { + if (!ctx.model) return; if (event.source === "restore" || !event.previousModel) { if (pendingPredictionTimer) clearTimeout(pendingPredictionTimer); pendingPredictionTimer = undefined; clearPrediction(ctx); + setCurrentLane(event.model); return; } schedulePrediction(ctx, event.model, pi.getThinkingLevel()); }); - pi.on("session_shutdown", async (_event, ctx) => { if (pendingPredictionTimer) clearTimeout(pendingPredictionTimer); pendingPredictionTimer = undefined; diff --git a/packages/pi-cache-hit-predictor/package.json b/packages/pi-cache-hit-predictor/package.json index 7df7c54..8db99e1 100644 --- a/packages/pi-cache-hit-predictor/package.json +++ b/packages/pi-cache-hit-predictor/package.json @@ -1,6 +1,6 @@ { "name": "@iurysza/pi-cache-hit-predictor", - "version": "0.1.0", + "version": "0.2.0", "description": "Predicts reusable prompt-cache prefixes when Pi switches model or reasoning lanes.", "type": "module", "license": "MIT", diff --git a/packages/pi-cache-hit-predictor/src/predictor.ts b/packages/pi-cache-hit-predictor/src/predictor.ts index 9cc1bdd..757e86d 100644 --- a/packages/pi-cache-hit-predictor/src/predictor.ts +++ b/packages/pi-cache-hit-predictor/src/predictor.ts @@ -25,6 +25,22 @@ export interface CachePrediction { hasLaneHistory: boolean; } +export interface CacheSwitchImpact { + sourceLane: CacheLane; + destLane: CacheLane; + currentPromptTokens: number | null; + contextWindow: number | null; + sourceTokens: number; + destTokens: number; + lostTokens: number; + dropPercent: number | null; + windowImpactPercent: number | null; +} + +export interface RenderTheme { + fg(color: "success" | "error" | "dim" | "muted" | "warning", text: string): string; +} + export function cacheLaneKey(lane: CacheLane): string { return JSON.stringify([ lane.provider, @@ -135,3 +151,152 @@ export function predictCacheHit( hasLaneHistory: snapshot !== undefined, }; } + +export function lastUsedLane(entries: readonly SessionEntry[]): CacheLane | undefined { + let thinkingLevel = UNKNOWN_THINKING_LEVEL; + let lastLane: CacheLane | undefined; + + for (const entry of entries) { + if (entry.type === "thinking_level_change") { + thinkingLevel = entry.thinkingLevel; + continue; + } + + if (entry.type === "compaction" || entry.type === "branch_summary") { + lastLane = undefined; + continue; + } + + if (entry.type !== "message" || entry.message.role !== "assistant") { + continue; + } + + const message = entry.message; + if (message.stopReason === "aborted" || message.stopReason === "error") { + continue; + } + + const tokens = promptTokens(message.usage); + if (tokens <= 0) continue; + + lastLane = { + provider: message.provider, + api: message.api, + model: message.model, + thinkingLevel, + }; + } + + return lastLane; +} + +export function predictCacheSwitchImpact( + history: CacheHistory, + sourceLane: CacheLane, + destLane: CacheLane, + currentPromptTokens: number | null, + contextWindow: number | null, +): CacheSwitchImpact { + const sourceSnapshot = history.lanes.get(cacheLaneKey(sourceLane))?.promptTokens ?? 0; + const destSnapshot = history.lanes.get(cacheLaneKey(destLane))?.promptTokens ?? 0; + const currentTokens = + currentPromptTokens !== null && currentPromptTokens > 0 + ? currentPromptTokens + : null; + const sourceTokens = currentTokens !== null + ? Math.min(sourceSnapshot, currentTokens) + : sourceSnapshot; + const destTokens = currentTokens !== null + ? Math.min(destSnapshot, currentTokens) + : destSnapshot; + const lostTokens = Math.max(0, sourceTokens - destTokens); + const dropPercent = sourceTokens > 0 + ? (lostTokens / sourceTokens) * 100 + : null; + const windowImpactPercent = contextWindow && contextWindow > 0 + ? (lostTokens / contextWindow) * 100 + : null; + + return { + sourceLane, + destLane, + currentPromptTokens: currentTokens, + contextWindow: contextWindow ?? null, + sourceTokens, + destTokens, + lostTokens, + dropPercent, + windowImpactPercent, + }; +} + +const BLOCK = "█"; +const DIM = "░"; +const DEFAULT_BAR_WIDTH = 8; + +function allocateSegments(values: readonly number[], width: number): number[] { + const total = values.reduce((sum, value) => sum + value, 0); + if (total === 0) return values.map(() => 0); + + const raw = values.map((value) => (value / total) * width); + const floors = raw.map(Math.floor); + const remainders = raw + .map((value, index) => ({ remainder: value - floors[index], index })) + .sort((a, b) => b.remainder - a.remainder); + let sum = floors.reduce((a, b) => a + b, 0); + + for (const { index } of remainders) { + if (sum >= width) break; + floors[index]++; + sum++; + } + + return floors; +} + +export function renderSwitchImpact( + impact: CacheSwitchImpact, + theme?: RenderTheme, + options?: { barWidth?: number }, +): string { + const lane = `${impact.destLane.model} · ${impact.destLane.thinkingLevel}`; + + if (impact.sourceTokens === 0 && impact.destTokens === 0) { + return `cache ${lane} · cold`; + } + + const barWidth = options?.barWidth ?? DEFAULT_BAR_WIDTH; + const totalTokens = Math.max( + impact.contextWindow ?? 0, + impact.currentPromptTokens ?? 0, + impact.destTokens + impact.lostTokens, + ); + const carried = impact.destTokens; + const lost = impact.lostTokens; + const rest = Math.max(0, totalTokens - carried - lost); + const [carriedWidth, lostWidth, restWidth] = allocateSegments( + [carried, lost, rest], + barWidth, + ); + + const color = (name: "success" | "error" | "dim", text: string) => + theme ? theme.fg(name, text) : text; + const segments: string[] = []; + if (carriedWidth > 0) segments.push(color("success", BLOCK.repeat(carriedWidth))); + if (lostWidth > 0) segments.push(color("error", BLOCK.repeat(lostWidth))); + if (restWidth > 0) segments.push(color("dim", DIM.repeat(restWidth))); + const bar = segments.join(""); + + let label: string; + if (impact.lostTokens > 0 && impact.dropPercent !== null) { + label = `↓${Math.round(impact.dropPercent)}%`; + } else if (impact.sourceTokens > 0) { + label = "↓0%"; + } else if (impact.destTokens > 0) { + label = "warm"; + } else { + label = "cold"; + } + + return `cache ${lane} ${label} [${bar}]`; +} diff --git a/packages/pi-cache-hit-predictor/tests/extension.test.ts b/packages/pi-cache-hit-predictor/tests/extension.test.ts index 1d86a35..9f9da8e 100644 --- a/packages/pi-cache-hit-predictor/tests/extension.test.ts +++ b/packages/pi-cache-hit-predictor/tests/extension.test.ts @@ -12,7 +12,6 @@ const oldModel = { api: "openai-responses", id: "gpt-old", name: "Old", - baseUrl: "https://api.openai.com/v1", reasoning: true, input: ["text"], cost: { input: 1, output: 1, cacheRead: 0.1, cacheWrite: 0 }, @@ -21,51 +20,89 @@ const oldModel = { } as const; const newModel = { ...oldModel, id: "gpt-new", name: "New" }; -const branch = [ - { - type: "thinking_level_change", - id: "00000001", - parentId: null, - timestamp: new Date(1_000).toISOString(), - thinkingLevel: "high", +const testTheme = { + fg(color: string, text: string) { + return `<${color}>${text}`; }, - { +}; + +let nextId = 1; + +function baseEntry(type: string) { + const id = nextId.toString(16).padStart(8, "0"); + const parentId = nextId === 1 ? null : (nextId - 1).toString(16).padStart(8, "0"); + nextId += 1; + return { + type, + id, + parentId, + timestamp: new Date(nextId * 1_000).toISOString(), + }; +} + +function thinkingChange(level: string): SessionEntry { + return { + ...baseEntry("thinking_level_change"), + type: "thinking_level_change", + thinkingLevel: level, + }; +} + +function assistant( + model: string, + prompt: number, + cacheRead: number, +): Extract { + return { + ...baseEntry("message"), type: "message", - id: "00000002", - parentId: "00000001", - timestamp: new Date(2_000).toISOString(), message: { role: "assistant", content: [], api: "openai-responses", provider: "openai", - model: "gpt-old", + model, usage: { - input: 17_000, + input: prompt - cacheRead, output: 10, - cacheRead: 8_000, + cacheRead, cacheWrite: 0, - totalTokens: 25_010, + totalTokens: prompt + 10, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, }, stopReason: "stop", - timestamp: 2_000, + timestamp: nextId * 1_000, }, - }, -] as SessionEntry[]; + }; +} -function createHarness() { - const handlers = new Map unknown>(); +function createHarness(options?: { + branch?: SessionEntry[]; + contextUsage?: { tokens: number; contextWindow: number; percent: number } | null; +}) { + nextId = 1; + const handlers = new Map< + string, + (event: never, ctx: ExtensionContext) => unknown + >(); const busHandlers = new Map void>>(); const busEvents: Array<{ channel: string; data: unknown }> = []; const statuses: Array = []; + const branch = options?.branch ?? [ + thinkingChange("high"), + assistant("gpt-old", 25_010, 8_000), + ]; + const contextUsage = options?.contextUsage === undefined + ? { tokens: 100_000, contextWindow: 200_000, percent: 50 } + : options.contextUsage; const ctx = { mode: "tui", model: newModel, - getContextUsage: () => ({ tokens: 100_000, contextWindow: 200_000, percent: 50 }), + getContextUsage: () => contextUsage, sessionManager: { getBranch: () => branch }, modelRegistry: { find: () => undefined }, ui: { + theme: testTheme, setStatus(key: string, value: string | undefined) { assert.equal(key, "pi-cache-hit-predictor"); statuses.push(value); @@ -96,6 +133,7 @@ function createHarness() { ctx, statuses, busEvents, + branch, async fire(event: string, data: unknown = {}) { await handlers.get(event)?.(data as never, ctx); }, @@ -104,30 +142,37 @@ function createHarness() { const waitForPrediction = () => new Promise((resolve) => setTimeout(resolve, 5)); -function assistantMessage(overrides: Record = {}) { - const entry = branch[1] as Extract; +function assistantMessage( + branch: SessionEntry[], + overrides: Record = {}, +) { + const entry = branch[branch.length - 1] as Extract; return { ...entry.message, ...overrides }; } -test("coalesces a model clamp into one footer status", async () => { +test("shows a full cache drop when switching to a cold model", async () => { const harness = createHarness(); await harness.fire("session_start"); - await harness.fire("thinking_level_select", { level: "high", previousLevel: "low" }); await harness.fire("model_select", { model: newModel, previousModel: oldModel, source: "set", }); await waitForPrediction(); - assert.equal(harness.statuses.filter(Boolean).length, 1); assert.equal( harness.statuses.at(-1), - "cache gpt-new · high · cold 0%/~100k", + "cache gpt-new · high ↓100% [█░░░░░░░]", ); }); -test("renders a concise warm-lane footer status", async () => { - const harness = createHarness(); +test("shows a partial drop when switching to a smaller cached lane", async () => { + const harness = createHarness({ + branch: [ + thinkingChange("high"), + assistant("gpt-old", 25_010, 8_000), + assistant("gpt-new", 100_010, 24_000), + ], + }); await harness.fire("session_start"); await harness.fire("model_select", { model: oldModel, @@ -137,7 +182,51 @@ test("renders a concise warm-lane footer status", async () => { await waitForPrediction(); assert.equal( harness.statuses.at(-1), - "cache gpt-old · high · ~25k/~100k 25%", + "cache gpt-old · high ↓75% [████░░░░]", + ); +}); + +test("shows a warm destination with no loss", async () => { + const harness = createHarness({ + branch: [ + thinkingChange("low"), + assistant("gpt-old", 25_010, 8_000), + thinkingChange("high"), + assistant("gpt-old", 100_010, 24_000), + thinkingChange("low"), + assistant("gpt-old", 25_010, 8_000), + ], + }); + await harness.fire("session_start"); + (harness.ctx as any).model = oldModel; + await harness.fire("thinking_level_select", { + level: "high", + previousLevel: "low", + }); + await waitForPrediction(); + assert.equal( + harness.statuses.at(-1), + "cache gpt-old · high ↓0% [████░░░░]", + ); +}); + +test("coalesces paired model and thinking changes into one status", async () => { + const harness = createHarness(); + await harness.fire("session_start"); + await harness.fire("thinking_level_select", { + level: "high", + previousLevel: "low", + }); + await harness.fire("model_select", { + model: newModel, + previousModel: oldModel, + source: "set", + }); + await waitForPrediction(); + assert.equal(harness.statuses.filter(Boolean).length, 1); + assert.equal( + harness.statuses.at(-1), + "cache gpt-new · high ↓100% [█░░░░░░░]", ); }); @@ -152,29 +241,42 @@ test("clears only after a successful response on the predicted lane", async () = const prediction = harness.statuses.at(-1); await harness.fire("message_end", { - message: assistantMessage({ model: newModel.id, stopReason: "error" }), + message: assistantMessage(harness.branch, { model: newModel.id, stopReason: "error" }), }); assert.equal(harness.statuses.at(-1), prediction); await harness.fire("message_end", { - message: assistantMessage({ model: newModel.id, stopReason: "aborted" }), + message: assistantMessage(harness.branch, { model: newModel.id, stopReason: "aborted" }), }); assert.equal(harness.statuses.at(-1), prediction); await harness.fire("message_end", { - message: assistantMessage({ model: oldModel.id, stopReason: "stop" }), + message: assistantMessage(harness.branch, { model: oldModel.id, stopReason: "stop" }), }); assert.equal(harness.statuses.at(-1), prediction); await harness.fire("message_end", { - message: assistantMessage({ model: newModel.id, stopReason: "stop" }), + message: assistantMessage(harness.branch, { model: newModel.id, stopReason: "stop" }), }); assert.equal(harness.statuses.at(-1), undefined); }); +test("falls back to legacy text when context usage is unavailable", async () => { + const harness = createHarness({ contextUsage: null }); + await harness.fire("session_start"); + await harness.fire("model_select", { + model: newModel, + previousModel: oldModel, + source: "set", + }); + await waitForPrediction(); + assert.equal(harness.statuses.at(-1), "cache gpt-new · high · cold"); +}); + test("registers priority metadata and cleans up at shutdown", async () => { const harness = createHarness(); - const registrations = () => harness.busEvents.filter(({ channel }) => channel.endsWith("/register/v1")); + const registrations = () => + harness.busEvents.filter(({ channel }) => channel.endsWith("/register/v1")); assert.deepEqual(registrations().at(-1)?.data, { protocolVersion: 1, id: "pi-cache-hit-predictor", diff --git a/packages/pi-cache-hit-predictor/tests/predictor.test.ts b/packages/pi-cache-hit-predictor/tests/predictor.test.ts index 945da9b..de22112 100644 --- a/packages/pi-cache-hit-predictor/tests/predictor.test.ts +++ b/packages/pi-cache-hit-predictor/tests/predictor.test.ts @@ -3,7 +3,8 @@ import { describe, test } from "node:test"; import type { SessionEntry } from "@earendil-works/pi-coding-agent"; import { cacheLaneKey, - predictCacheHit, + predictCacheSwitchImpact, + renderSwitchImpact, scanCacheHistory, } from "../src/predictor.js"; @@ -92,35 +93,152 @@ describe("cache lane history", () => { }); }); -describe("cache hit prediction", () => { - test("bounds the old lane prefix by the current prompt", () => { +const lowLane = { + provider: "openai", + api: "openai-responses", + model: "gpt-test", + thinkingLevel: "low", +} as const; + +const highLane = { + provider: "openai", + api: "openai-responses", + model: "gpt-test", + thinkingLevel: "high", +} as const; + +const otherLane = { + provider: "openai", + api: "openai-responses", + model: "gpt-other", + thinkingLevel: "low", +} as const; + +const testTheme = { + fg(color: string, text: string) { + return `[${color}:${text}]`; + }, +}; + +describe("cache switch impact", () => { + test("reports no loss when destination carries more than source", () => { const history = scanCacheHistory([ thinking("low"), assistant("gpt-test", 25_000, 1_000, 8_000), + thinking("high"), + assistant("gpt-test", 100_000, 2_000, 24_000), ]); - const prediction = predictCacheHit(history, { - provider: "openai", - api: "openai-responses", - model: "gpt-test", - thinkingLevel: "low", - }, 100_000); - assert.equal(prediction.estimatedCacheTokens, 25_000); - assert.equal(prediction.percent, 25); + const impact = predictCacheSwitchImpact(history, lowLane, highLane, 100_000, 200_000); + assert.equal(impact.sourceTokens, 25_000); + assert.equal(impact.destTokens, 100_000); + assert.equal(impact.lostTokens, 0); + assert.equal(impact.dropPercent, 0); + assert.equal(impact.windowImpactPercent, 0); }); - test("reports an unseen destination as cold", () => { + test("computes partial loss between lanes", () => { + const history = scanCacheHistory([ + thinking("low"), + assistant("gpt-test", 25_000, 1_000, 8_000), + thinking("high"), + assistant("gpt-test", 100_000, 2_000, 24_000), + ]); + const impact = predictCacheSwitchImpact(history, highLane, lowLane, 100_000, 200_000); + assert.equal(impact.sourceTokens, 100_000); + assert.equal(impact.destTokens, 25_000); + assert.equal(impact.lostTokens, 75_000); + assert.equal(impact.dropPercent, 75); + assert.equal(impact.windowImpactPercent, 37.5); + }); + + test("reports a full drop when destination is cold", () => { const history = scanCacheHistory([ thinking("high"), assistant("gpt-test", 100_000, 1_000, 20_000), ]); - const prediction = predictCacheHit(history, { - provider: "openai", - api: "openai-responses", - model: "gpt-test", - thinkingLevel: "low", - }, 100_000); - assert.equal(prediction.hasLaneHistory, false); - assert.equal(prediction.estimatedCacheTokens, 0); - assert.equal(prediction.percent, 0); + const impact = predictCacheSwitchImpact(history, highLane, otherLane, 100_000, 200_000); + assert.equal(impact.sourceTokens, 100_000); + assert.equal(impact.destTokens, 0); + assert.equal(impact.lostTokens, 100_000); + assert.equal(impact.dropPercent, 100); + assert.equal(impact.windowImpactPercent, 50); + }); + + test("caps source tokens by the current prompt size", () => { + const history = scanCacheHistory([ + thinking("low"), + assistant("gpt-test", 25_000, 1_000, 8_000), + ]); + const impact = predictCacheSwitchImpact(history, lowLane, otherLane, 10_000, 200_000); + assert.equal(impact.sourceTokens, 10_000); + assert.equal(impact.destTokens, 0); + assert.equal(impact.lostTokens, 10_000); + assert.equal(impact.dropPercent, 100); + }); + + test("returns null drop when source lane has no history", () => { + const history = scanCacheHistory([ + thinking("high"), + assistant("gpt-test", 100_000, 1_000, 20_000), + ]); + const impact = predictCacheSwitchImpact(history, lowLane, highLane, 100_000, 200_000); + assert.equal(impact.sourceTokens, 0); + assert.equal(impact.destTokens, 100_000); + assert.equal(impact.lostTokens, 0); + assert.equal(impact.dropPercent, null); + }); +}); + +describe("switch impact rendering", () => { + test("renders a full drop with a small red segment inside the window", () => { + const history = scanCacheHistory([ + thinking("low"), + assistant("gpt-test", 20_000, 1_000, 8_000), + ]); + const impact = predictCacheSwitchImpact(history, lowLane, otherLane, 20_000, 200_000); + const text = renderSwitchImpact(impact, testTheme, { barWidth: 8 }); + assert.equal(text, "cache gpt-other · low ↓100% [[error:█][dim:░░░░░░░]]"); + }); + + test("renders a partial drop with green, red, and dim segments", () => { + const history = scanCacheHistory([ + thinking("low"), + assistant("gpt-test", 25_000, 1_000, 8_000), + thinking("high"), + assistant("gpt-test", 100_000, 2_000, 24_000), + ]); + const impact = predictCacheSwitchImpact(history, highLane, lowLane, 100_000, 200_000); + const text = renderSwitchImpact(impact, testTheme, { barWidth: 8 }); + assert.equal(text, "cache gpt-test · low ↓75% [[success:█][error:███][dim:░░░░]]"); + }); + + test("renders a warm destination with no red segment", () => { + const history = scanCacheHistory([ + thinking("high"), + assistant("gpt-test", 100_000, 1_000, 20_000), + ]); + const impact = predictCacheSwitchImpact(history, lowLane, highLane, 100_000, 200_000); + const text = renderSwitchImpact(impact, testTheme, { barWidth: 8 }); + assert.equal(text, "cache gpt-test · high warm [[success:████][dim:░░░░]]"); + }); + + test("renders no-loss switch from a warm source lane", () => { + const history = scanCacheHistory([ + thinking("low"), + assistant("gpt-test", 25_000, 1_000, 8_000), + thinking("high"), + assistant("gpt-test", 100_000, 2_000, 24_000), + ]); + const impact = predictCacheSwitchImpact(history, lowLane, highLane, 100_000, 200_000); + const text = renderSwitchImpact(impact, testTheme, { barWidth: 8 }); + assert.equal(text, "cache gpt-test · high ↓0% [[success:████][dim:░░░░]]"); + }); + + test("falls back to cold label when both lanes are cold", () => { + const history = scanCacheHistory([]); + const impact = predictCacheSwitchImpact(history, lowLane, otherLane, 100_000, 200_000); + const text = renderSwitchImpact(impact, testTheme, { barWidth: 8 }); + assert.equal(text, "cache gpt-other · low · cold"); }); }); + -- 2.51.2