diff --git a/src/providers.ts b/src/providers.ts index f4b7cef..cf37d23 100644 --- a/src/providers.ts +++ b/src/providers.ts @@ -91,6 +91,11 @@ export interface ModelInfo { supportsImages: boolean; } +/** The built-in offline mock, by ref. It streams and nothing else. */ +export function isEchoModel(modelRef: string): boolean { + return modelRef === "echo" || modelRef.startsWith("echo/"); +} + const ECHO_MODEL: ModelInfo = { ref: "echo", providerId: "echo", @@ -260,7 +265,7 @@ export class ProviderRegistry { * model, both must be enabled/known, or this throws. */ resolveModel(modelRef: string): LanguageModel { - if (modelRef === "echo" || modelRef.startsWith("echo/")) { + if (isEchoModel(modelRef)) { return createEchoModel(); } diff --git a/src/store.ts b/src/store.ts index c4738b5..41e0604 100644 --- a/src/store.ts +++ b/src/store.ts @@ -141,6 +141,11 @@ export interface ConversationSearchResult extends ConversationSummary { snippet: string | null; } +// How much of the opening exchange `titleSeed` hands the model. Enough to name +// the subject, nowhere near enough to be worth paying for. +const TITLE_OPENER_CHARS = 1_200; +const TITLE_REPLY_CHARS = 800; + // The conversation-list SELECT (owner + optional project filters are appended // as a WHERE by listConversations). `last_activity` is the newest event time, // falling back to createdAt; the LEFT JOIN carries the project name for the @@ -658,6 +663,61 @@ export class Store { } } + /** + * The opening exchange, as text to title from: the first user message plus the + * start of the first reply. + * + * The user's opener alone is often nothing to work with — a conversation that + * starts "hi" has no subject in it yet, and one that starts with only an image + * has no text at all. The reply is where the subject actually surfaces, so it + * comes along, capped: a title needs the gist, not the essay. + */ + titleSeed(id: string): string | null { + const row = this.db + .prepare( + "SELECT data FROM events WHERE conversation_id = ? AND event = 'user-message' ORDER BY seq ASC LIMIT 1", + ) + .get(id) as { data: string } | null; + if (!row) return null; + let opener = ""; + try { + const d = JSON.parse(row.data) as { + content?: string; + attachments?: Array<{ name?: string }>; + }; + opener = (d.content ?? "").trim(); + // An attachments-only opener still names its files, which is a subject. + if (!opener && d.attachments?.length) { + opener = d.attachments + .map((a) => a.name) + .filter(Boolean) + .join(", "); + } + } catch { + return null; + } + // Deltas arrive in seq order and the first run's are the first ones logged, + // so reading until the cap never reaches a later turn. + const deltas = this.db + .prepare( + "SELECT data FROM events WHERE conversation_id = ? AND event = 'text-delta' ORDER BY seq ASC LIMIT 400", + ) + .all(id) as Array<{ data: string }>; + let reply = ""; + for (const d of deltas) { + try { + reply += (JSON.parse(d.data) as { delta?: string }).delta ?? ""; + } catch { + /* a malformed delta is not worth abandoning the title over */ + } + if (reply.length >= TITLE_REPLY_CHARS) break; + } + const parts: string[] = []; + if (opener) parts.push(`User: ${opener.slice(0, TITLE_OPENER_CHARS)}`); + if (reply.trim()) parts.push(`Assistant: ${reply.slice(0, TITLE_REPLY_CHARS).trim()}`); + return parts.length ? parts.join("\n\n") : null; + } + /** Set the title only if none is set yet (auto-title never clobbers a rename). * Returns whether it actually set one. */ setTitleIfEmpty(id: string, title: string): boolean { diff --git a/src/title.ts b/src/title.ts index 806f955..a6765bf 100644 --- a/src/title.ts +++ b/src/title.ts @@ -1,5 +1,6 @@ -import { generateText } from "ai"; +import { type JSONValue, streamText } from "ai"; import { getRegistry, resolveModel } from "./inference"; +import { isEchoModel } from "./providers"; import { getConfig } from "./settings"; import type { Store } from "./store"; @@ -14,34 +15,61 @@ function enabledModels(store: Store) { /** * The model for utility work (titles): `agent.smallModel` when it's set AND * enabled, otherwise the cheapest enabled model (least in+out cost per 1M) — so - * a configured ref that no longer exists gracefully falls back. Null only when - * no model is enabled at all. + * a configured ref that no longer exists gracefully falls back. Null when no + * model is enabled at all. + * + * Two exclusions, both because "cheapest" rewards missing metadata. The echo + * mock costs nothing, so it won outright whenever it was visible. A model with + * no context window is one nothing is known about — the catalog coerces absent + * pricing to zero, so an unlisted model would win the same way. A genuinely free + * local model still wins, because discovery gives it a real window. + * + * An explicitly configured `agent.smallModel` skips both checks: naming it is a + * choice, inheriting it isn't. */ export function resolveSmallModel(store: Store): string | null { const enabled = enabledModels(store); if (!enabled.length) return null; const configured = getConfig().agent.smallModel; if (configured && enabled.some((m) => m.ref === configured)) return configured; - return enabled.reduce((a, b) => + const usable = enabled.filter((m) => !isEchoModel(m.ref) && m.contextWindow > 0); + if (!usable.length) return null; + return usable.reduce((a, b) => b.costPer1MIn + b.costPer1MOut < a.costPer1MIn + a.costPer1MOut ? b : a, ).ref; } /** - * A short conversation title from the first user message, via a small/cheap - * model (config `agent.smallModel`). Best-effort — any failure returns null and - * the caller leaves the title as the truncated first message. + * A short conversation title from the opening exchange, via a small/cheap model + * (config `agent.smallModel`). Best-effort — any failure returns null and the + * caller leaves the title as the truncated first message. */ const SYSTEM = - "You write a short, specific title for a conversation from the user's first message. " + + "You write a short, specific title for a conversation from how it opens. " + "Reply with ONLY the title: 3 to 6 words, no surrounding quotes, no trailing punctuation, " + "no preamble or explanation."; -function sanitize(raw: string): string { +/** + * Six words is about ten tokens; the rest of this budget is headroom to think. + * + * Reasoning models spend the output budget before the first answer token, and + * some endpoints reason on everything (Hyper does — see discover.ts). The old + * 24-token cap was exhausted mid-thought, so the call came back `length` with + * empty text: no title, no error, nothing in the log. + */ +const MAX_OUTPUT_TOKENS = 1_024; + +/** A title is never worth stalling on, and nothing else is waiting on it. */ +const TIMEOUT_MS = 20_000; + +/** Strip what a small model wraps around the title it was asked for. */ +export function sanitize(raw: string): string { let t = raw.trim().split("\n")[0]!.trim(); - t = t.replace(/^["'`*_]+|["'`*_]+$/g, "").trim(); // strip wrapping quotes/markdown - t = t.replace(/[.]+$/, "").trim(); // drop a trailing period + const unwrap = (s: string) => s.replace(/^["'`*_]+|["'`*_]+$/g, "").trim(); + // Small models like to label their answer, inside or outside the quotes. + t = unwrap(t).replace(/^title\s*[:–-]\s*/i, ""); + t = unwrap(t).replace(/[.]+$/, "").trim(); if (t.length > 70) t = t.slice(0, 70).trimEnd() + "…"; return t; } @@ -51,17 +79,36 @@ export async function generateTitle( conversationId: string, modelRef: string, ): Promise { - const first = store.firstUserMessage(conversationId); - if (!first || !first.trim()) return null; + const seed = store.titleSeed(conversationId); + if (!seed) return null; + // Whatever the operator tuned for this endpoint (a thinking toggle, an effort + // level) applies here too — the utility call shouldn't be the one request that + // ignores the config and runs at the model's own default effort. + const slash = modelRef.indexOf("/"); + const providerId = slash > 0 ? modelRef.slice(0, slash) : modelRef; + const providerOptions = getRegistry().getConfig(providerId)?.providerOptions; try { - const { text } = await generateText({ + // streamText, not generateText: a stream-only model (the echo mock, and any + // endpoint that implements only the streaming half) has no `doGenerate` and + // threw on every title. Everything that serves chat can stream. + const result = streamText({ model: resolveModel(modelRef), system: SYSTEM, - prompt: first.slice(0, 2000), // enough to title from without feeding a whole essay - maxOutputTokens: 24, - temperature: 0.3, + prompt: seed, + maxOutputTokens: MAX_OUTPUT_TOKENS, + abortSignal: AbortSignal.timeout(TIMEOUT_MS), + // No temperature: some endpoints reject it outright when reasoning is on, + // and a title doesn't need the knob. + ...(providerOptions + ? { providerOptions: { [providerId]: providerOptions as Record } } + : {}), }); - return sanitize(text) || null; + const title = sanitize(await result.text); + if (title) return title; + // Empty text is a real outcome (a budget spent on reasoning, a filter), and + // it used to vanish silently — the caller just saw "no title". + console.warn(`[title] ${modelRef} produced no title (finish: ${await result.finishReason})`); + return null; } catch (e) { console.warn("[title] generation failed:", (e as Error).message); return null; diff --git a/tests/smallmodel.test.ts b/tests/smallmodel.test.ts new file mode 100644 index 0000000..474b22a --- /dev/null +++ b/tests/smallmodel.test.ts @@ -0,0 +1,86 @@ +import { afterEach, expect, test } from "bun:test"; +import { Catalog } from "../src/catalog"; +import { setRegistry } from "../src/inference"; +import { ProviderRegistry } from "../src/providers"; +import { loadConfig, setConfig } from "../src/settings"; +import { Store } from "../src/store"; +import { resolveSmallModel } from "../src/title"; + +afterEach(() => setConfig(null)); + +/** Schema defaults, with `agent.smallModel` pinned. */ +function configureSmallModel(smallModel: string): void { + const base = loadConfig({ path: "/nonexistent", env: {} }); + setConfig({ ...base, agent: { ...base.agent, smallModel } }); +} + +/** Two real models at different prices, alongside the built-in echo mock. */ +function registry(): ProviderRegistry { + const catalog = Catalog.fromRaw([ + { + id: "acme", + name: "Acme", + type: "openai-compat", + api_endpoint: "https://acme.test/v1", + models: [ + { + id: "big", + name: "Big", + context_window: 8000, + cost_per_1m_in: 10, + cost_per_1m_out: 30, + }, + { + id: "small", + name: "Small", + context_window: 8000, + cost_per_1m_in: 1, + cost_per_1m_out: 2, + }, + // Nothing known about it: the catalog coerces absent pricing to zero, + // so on cost alone it beats every real model. + { id: "mystery", name: "Mystery" }, + ], + }, + ]); + return new ProviderRegistry(catalog, { + config: { providers: [{ id: "acme", apiKey: "$ACME_KEY" }] }, + }); +} + +function storeWith(...visible: string[]): Store { + const store = new Store(":memory:"); + for (const ref of visible) + store.setModelSetting({ ref, visible: true, sortOrder: 0, displayName: null }); + return store; +} + +test("the echo mock never wins the cheapest-model contest", () => { + // It costs nothing, so it beat every real model outright — and it's + // stream-only, so every title then failed on doGenerate. + setRegistry(registry()); + expect(resolveSmallModel(storeWith("echo", "acme/big", "acme/small"))).toBe("acme/small"); +}); + +test("with only the mock enabled there is no small model", () => { + setRegistry(registry()); + expect(resolveSmallModel(storeWith("echo"))).toBeNull(); +}); + +test("a model with no metadata does not win on its zero price", () => { + setRegistry(registry()); + expect(resolveSmallModel(storeWith("acme/mystery", "acme/small"))).toBe("acme/small"); + expect(resolveSmallModel(storeWith("acme/mystery"))).toBeNull(); +}); + +test("an explicitly configured small model is honored", () => { + setRegistry(registry()); + configureSmallModel("acme/big"); + expect(resolveSmallModel(storeWith("echo", "acme/big", "acme/small"))).toBe("acme/big"); +}); + +test("a configured model that is not enabled falls back to the cheapest real one", () => { + setRegistry(registry()); + configureSmallModel("acme/gone"); + expect(resolveSmallModel(storeWith("echo", "acme/big", "acme/small"))).toBe("acme/small"); +}); diff --git a/tests/title.test.ts b/tests/title.test.ts index 3ea7f92..c511e8c 100644 --- a/tests/title.test.ts +++ b/tests/title.test.ts @@ -1,17 +1,29 @@ import { expect, test } from "bun:test"; import { Store } from "../src/store"; +import { sanitize } from "../src/title"; -function seedUser(store: Store, id: string, content: string): void { - const db = ( - store as unknown as { db: { query: (s: string) => { run: (...a: unknown[]) => void } } } - ).db; +type RawDb = { db: { query: (s: string) => { run: (...a: unknown[]) => void } } }; + +function seedUser(store: Store, id: string, content: string, attachments?: unknown[]): void { + const db = (store as unknown as RawDb).db; db.query("INSERT INTO conversations (id, created_at, last_seq) VALUES (?, ?, 0)").run( id, Date.now(), ); db.query( "INSERT INTO events (id, conversation_id, seq, event, data, created_at) VALUES (?, ?, 0, 'user-message', ?, ?)", - ).run(`${id}:0`, id, JSON.stringify({ content }), Date.now()); + ).run(`${id}:0`, id, JSON.stringify({ content, attachments }), Date.now()); +} + +/** Append assistant text-deltas after the seeded opener. */ +function seedReply(store: Store, id: string, ...deltas: string[]): void { + const db = (store as unknown as RawDb).db; + deltas.forEach((delta, i) => { + const seq = i + 1; + db.query( + "INSERT INTO events (id, conversation_id, seq, event, data, created_at) VALUES (?, ?, ?, 'text-delta', ?, ?)", + ).run(`${id}:${seq}`, id, seq, JSON.stringify({ delta }), Date.now()); + }); } test("firstUserMessage returns the first user prompt", () => { @@ -21,6 +33,49 @@ test("firstUserMessage returns the first user prompt", () => { expect(store.firstUserMessage("nope")).toBeNull(); }); +test("sanitize strips the wrapping small models add", () => { + expect(sanitize("Spherical Mirror Optics")).toBe("Spherical Mirror Optics"); + expect(sanitize('"Spherical Mirror Optics"')).toBe("Spherical Mirror Optics"); + expect(sanitize("Title: Spherical Mirror Optics")).toBe("Spherical Mirror Optics"); + expect(sanitize('"Title: Spherical Mirror Optics"')).toBe("Spherical Mirror Optics"); + expect(sanitize("**Spherical Mirror Optics.**")).toBe("Spherical Mirror Optics"); + expect(sanitize("Spherical Mirror Optics\nand some rambling")).toBe("Spherical Mirror Optics"); + expect(sanitize(" ")).toBe(""); +}); + +test("sanitize truncates a title that ran long", () => { + const out = sanitize("word ".repeat(40)); + expect(out.length).toBeLessThanOrEqual(71); + expect(out.endsWith("…")).toBe(true); +}); + +test("titleSeed carries the reply, so a one-word opener still has a subject", () => { + const store = new Store(":memory:"); + seedUser(store, "c1", "hi"); + seedReply(store, "c1", "Spherical mirrors ", "focus light to a point."); + const seed = store.titleSeed("c1")!; + expect(seed).toContain("User: hi"); + expect(seed).toContain("Assistant: Spherical mirrors focus light to a point."); +}); + +test("titleSeed names the files when the opener is attachments only", () => { + const store = new Store(":memory:"); + seedUser(store, "c1", "", [{ sha256: "a", name: "budget.csv", mime: "text/csv", kind: "file" }]); + expect(store.titleSeed("c1")).toContain("budget.csv"); +}); + +test("titleSeed is null for a conversation with nothing in it", () => { + const store = new Store(":memory:"); + expect(store.titleSeed("nope")).toBeNull(); +}); + +test("titleSeed caps the reply it carries", () => { + const store = new Store(":memory:"); + seedUser(store, "c1", "go"); + seedReply(store, "c1", "x".repeat(5000)); + expect(store.titleSeed("c1")!.length).toBeLessThan(2500); +}); + test("setTitleIfEmpty sets a title, then never clobbers it", () => { const store = new Store(":memory:"); seedUser(store, "c1", "hi");