import modelPrompt from "../model/prompt.txt"; import consola from "consola"; import type { ChatCompletionMessageParam } from "openai/resources/chat/completions"; import * as c from "../core"; import * as tools from "../tools"; import { env } from "../env"; import { finalizeCitations, SourceBook } from "./citations"; const logger = consola.withTag("AI"); const MAX_TOOL_ROUNDS = 3; const MAX_CALLS_PER_ROUND = 3; // Replies are capped at 1000 graphemes, so this leaves room for a little reasoning const MAX_OUTPUT_TOKENS = 2000; // OpenRouter option, ignored by models without reasoning. Not in the OpenAI types const REASONING = { reasoning: { effort: "low" } } as object; /** Source numbers of everything citable inside a tool result. */ function sourcesOf(result: unknown, found = new Set()) { if (Array.isArray(result)) { result.forEach((item) => sourcesOf(item, found)); } else if (result && typeof result === "object") { const item = result as Record; if (typeof item.source === "number") found.add(item.source); Object.values(item).forEach((value) => sourcesOf(value, found)); } return found; } type Options = { /** Extra rules added to the system prompt, such as a shorter length limit. */ extraInstructions?: string; }; /** * Runs the model, letting it call tools until it answers. * @returns The reply with `[n]` citation markers, and the URL behind each one. */ export async function generateAIResponse( parsedContext: string, media: string[], messages: ChatCompletionMessageParam[], options: Options = {}, ) { // Static instructions come first and everything after only appends, so // providers can reuse their cached prefix across requests and rounds const conversation: ChatCompletionMessageParam[] = [ { role: "system", content: `${modelPrompt.replace("$handle", env.HANDLE)}${ options.extraInstructions ? `\n\n${options.extraInstructions}` : "" }`, }, // System messages can't carry images, so the context is a user turn { role: "user", content: [ { type: "text" as const, text: `Post context:\n${parsedContext}` }, ...(media.length > 0 ? [ { type: "text" as const, text: `Pictures from the post context, as attachments 1 to ${media.length}:`, }, ...media.map((url) => ({ type: "image_url" as const, image_url: { url }, })), ] : []), ], }, ...messages, ]; const book = new SourceBook(); // Pages read in full, which are cited even if the model forgets to const pagesRead: number[] = []; // Tool results that can be trimmed once the model has moved on const toolResults: { index: number; round: number; sources: number[] }[] = []; for (let round = 0;; round++) { const inference = await c.ai.chat.completions.create({ model: env.OPENROUTER_MODEL, messages: conversation, max_tokens: MAX_OUTPUT_TOKENS, ...REASONING, // Once the round limit is hit, stop offering tools to force an answer ...(tools.declarations.length > 0 && round < MAX_TOOL_ROUNDS ? { tools: tools.declarations } : {}), }); logger.log( `Inference round ${round} took ${inference.usage?.total_tokens} tokens`, ); const message = inference.choices[0]?.message; const calls = message?.tool_calls?.filter((call) => call.type === "function" ).slice(0, MAX_CALLS_PER_ROUND); if (!message || !calls || calls.length === 0) { if (!message?.content) return undefined; return finalizeCitations(message.content, book, pagesRead); } conversation.push({ ...message, tool_calls: calls }); // Old results are rarely needed again but are resent on every round for (const old of toolResults) { if (old.round > round - 2) continue; const note = old.sources.length > 0 ? ` Sources: ${old.sources.map((n) => `[${n}]`).join(", ")}.` : ""; (conversation[old.index] as { content: string }).content = JSON.stringify({ omitted: `Earlier tool result trimmed to save tokens.${note}`, }); } for (const call of calls) { logger.log("Function called invoked:", call.function.name); let args: unknown = {}; try { args = JSON.parse(call.function.arguments || "{}"); } catch {} const result = book.annotate( await tools.handler({ name: call.function.name, args }), ); if ( call.function.name === "fetch_url" && result && typeof result === "object" && "source" in result ) { pagesRead.push(result.source as number); } toolResults.push({ index: conversation.length, round, sources: [...sourcesOf(result)], }); conversation.push({ role: "tool", tool_call_id: call.id, content: JSON.stringify(result ?? null), }); } } }