import { decodeHTML } from "entities"; import { createFetchTools, type FetchResult, type FetchErrorCode, } from "@cloudflare/think/tools/fetch"; import type { ToolExecutionOptions } from "ai"; import type { WebErrorCode, WebResult } from "../shared/web"; import { WebDeadline } from "./browser-session"; import { webError, webFailure } from "./web-errors"; import { cleanText, clip, publicWebUrl, webSource } from "./web-source"; export const WEB_MAX_BYTES = 256_000; export const WEB_MAX_CONTENT = 16_000; export const WEB_READ_TIMEOUT = 15_000; const fetchTool = createFetchTools({ allowlist: ["https://**", "http://**"], maxBytes: WEB_MAX_BYTES, maxModelChars: WEB_MAX_BYTES, timeoutMs: WEB_READ_TIMEOUT, response: "text", spillToWorkspace: false, followRedirects: "allowlisted", modelHeaderAllowlist: [], }).fetch_url; const codes: Record = { disallowed_url: "blocked_url", disallowed_redirect: "blocked_url", timeout: "timeout", aborted: "cancelled", non_2xx: "http_error", unsupported_content_type: "unsupported_content", invalid_json: "unsupported_content", too_large: "unsupported_content", request_failed: "request_failed", }; /** Native HTML parsing without executing JS; decode text after assembling chunks. */ async function htmlText(html: string) { let title = "", content = "", truncated = false; let skip = 0, inTitle = 0; const rewriter = new HTMLRewriter() .on("script, style, noscript, template, svg, nav, header, footer, form", { element(element) { skip++; element.onEndTag(() => { skip--; }); }, }) .on("title", { element(element) { inTitle++; element.onEndTag(() => { inTitle--; }); }, text(chunk) { title += clip(chunk.text, Math.max(0, 240 - title.length)); }, }) .on("p, div, article, main, section, h1, h2, h3, li, br, tr", { element() { if (!skip && content.length < WEB_MAX_CONTENT) content += "\n"; }, }) .onDocument({ text(chunk) { if (skip || inTitle) return; const remaining = WEB_MAX_CONTENT - content.length; content += clip(chunk.text, Math.max(0, remaining)); if (chunk.text.length > remaining) truncated = true; }, }); // Drain parsing without retaining a second rewritten HTML string. await rewriter .transform(new Response(html)) .body!.pipeTo(new WritableStream({ write() {} })); return { title: cleanText(decodeHTML(title), 240), content: decodeHTML(content).trim(), truncated, }; } export async function readWebUrl( raw: string, options: ToolExecutionOptions, ): Promise { const deadline = new WebDeadline(WEB_READ_TIMEOUT, options.abortSignal); try { const requestedUrl = publicWebUrl(raw).href; const result = (await deadline.run(() => Promise.resolve( fetchTool.execute!( { url: requestedUrl }, { ...options, abortSignal: deadline.signal }, ), ), )) as FetchResult; if (!result.ok) return webFailure(codes[result.code], result.status); const finalUrl = publicWebUrl(result.finalUrl).href; if ( ![ "text/html", "application/xhtml+xml", "text/plain", "text/markdown", "text/x-markdown", ].includes(result.contentType) ) return webFailure("unsupported_content"); const body = result.body ?? ""; const extracted = result.contentType === "text/html" || result.contentType === "application/xhtml+xml" ? await deadline.run(() => htmlText(body)) : { title: "", content: clip(body, WEB_MAX_CONTENT), truncated: body.length > WEB_MAX_CONTENT, }; deadline.remaining(); if (!extracted.content.trim()) return webFailure("request_failed"); return { ok: true, sources: [ await webSource({ title: extracted.title || new URL(finalUrl).hostname, requestedUrl, finalUrl, sourceKind: "page", contentType: result.contentType, content: extracted.content, truncated: result.truncated || extracted.truncated, }), ], }; } catch (error) { return webError(error, deadline.signal); } finally { deadline.dispose(); } }