import * as Effect from "effect/Effect"; import { decodeHTML } from "entities"; import { createFetchTools, type FetchResult, type FetchErrorCode, } from "@cloudflare/think/tools/fetch"; import type { ToolExecutionOptions } from "ai"; import type { WebErrorCode, WebResult } from "../shared/web"; import { WebFailure, webValidation } from "./web-errors"; import { cleanText, clip, publicWebUrl, webSource } from "./web-source"; export const WEB_MAX_BYTES = 256_000; export const WEB_MAX_CONTENT = 16_000; export const WEB_READ_TIMEOUT = 15_000; const fetchTool = createFetchTools({ allowlist: ["https://**", "http://**"], maxBytes: WEB_MAX_BYTES, maxModelChars: WEB_MAX_BYTES, timeoutMs: WEB_READ_TIMEOUT, response: "text", spillToWorkspace: false, followRedirects: "allowlisted", modelHeaderAllowlist: [], }).fetch_url; const codes: Record = { disallowed_url: "blocked_url", disallowed_redirect: "blocked_url", timeout: "timeout", aborted: "cancelled", non_2xx: "http_error", unsupported_content_type: "unsupported_content", invalid_json: "unsupported_content", too_large: "unsupported_content", request_failed: "request_failed", }; /** Native HTML parsing without executing JS; decode text after assembling chunks. */ function htmlText(html: string) { return Effect.gen(function* () { let title = "", content = "", truncated = false; let skip = 0, inTitle = 0; const rewriter = new HTMLRewriter() .on("script, style, noscript, template, svg, nav, header, footer, form", { element(element) { skip++; element.onEndTag(() => { skip--; }); }, }) .on("title", { element(element) { inTitle++; element.onEndTag(() => { inTitle--; }); }, text(chunk) { title += clip(chunk.text, Math.max(0, 240 - title.length)); }, }) .on("p, div, article, main, section, h1, h2, h3, li, br, tr", { element() { if (!skip && content.length < WEB_MAX_CONTENT) content += "\n"; }, }) .onDocument({ text(chunk) { if (skip || inTitle) return; const remaining = WEB_MAX_CONTENT - content.length; content += clip(chunk.text, Math.max(0, remaining)); if (chunk.text.length > remaining) truncated = true; }, }); // Drain parsing without retaining a second rewritten HTML string. yield* Effect.tryPromise({ try: (signal) => rewriter .transform(new Response(html)) .body!.pipeTo(new WritableStream({ write() {} }), { signal }), catch: () => new WebFailure("request_failed"), }); return { title: cleanText(decodeHTML(title), 240), content: decodeHTML(content).trim(), truncated, }; }); } export function readWebUrl( raw: string, options: ToolExecutionOptions, ): Effect.Effect { return Effect.gen(function* () { const requestedUrl = yield* webValidation(() => publicWebUrl(raw).href); const result = yield* Effect.tryPromise({ try: (signal) => Promise.resolve( fetchTool.execute!( { url: requestedUrl }, { ...options, abortSignal: signal }, ), ) as Promise, catch: () => new WebFailure("request_failed"), }); if (!result.ok) return yield* Effect.fail( new WebFailure(codes[result.code], result.status), ); const finalUrl = yield* webValidation( () => publicWebUrl(result.finalUrl).href, ); if ( ![ "text/html", "application/xhtml+xml", "text/plain", "text/markdown", "text/x-markdown", ].includes(result.contentType) ) return yield* Effect.fail(new WebFailure("unsupported_content")); const body = result.body ?? ""; const extracted = result.contentType === "text/html" || result.contentType === "application/xhtml+xml" ? yield* htmlText(body) : { title: "", content: clip(body, WEB_MAX_CONTENT), truncated: body.length > WEB_MAX_CONTENT, }; if (!extracted.content.trim()) return yield* Effect.fail(new WebFailure("request_failed")); const source = yield* webSource({ title: extracted.title || new URL(finalUrl).hostname, requestedUrl, finalUrl, sourceKind: "page", contentType: result.contentType, content: extracted.content, truncated: result.truncated || extracted.truncated, }); return { ok: true as const, sources: [source] }; }).pipe( Effect.timeoutOrElse({ duration: WEB_READ_TIMEOUT, orElse: () => Effect.fail(new WebFailure("timeout")), }), ); }