import { decodeHTML } from "entities";
import {
createFetchTools,
type FetchResult,
type FetchErrorCode,
} from "@cloudflare/think/tools/fetch";
import type { ToolExecutionOptions } from "ai";
import type { WebErrorCode, WebResult } from "../shared/web";
import { WebDeadline } from "./browser-session";
import { webError, webFailure } from "./web-errors";
import { cleanText, clip, publicWebUrl, webSource } from "./web-source";
export const WEB_MAX_BYTES = 256_000;
export const WEB_MAX_CONTENT = 16_000;
export const WEB_READ_TIMEOUT = 15_000;
const fetchTool = createFetchTools({
allowlist: ["https://**", "http://**"],
maxBytes: WEB_MAX_BYTES,
maxModelChars: WEB_MAX_BYTES,
timeoutMs: WEB_READ_TIMEOUT,
response: "text",
spillToWorkspace: false,
followRedirects: "allowlisted",
modelHeaderAllowlist: [],
}).fetch_url;
const codes: Record = {
disallowed_url: "blocked_url",
disallowed_redirect: "blocked_url",
timeout: "timeout",
aborted: "cancelled",
non_2xx: "http_error",
unsupported_content_type: "unsupported_content",
invalid_json: "unsupported_content",
too_large: "unsupported_content",
request_failed: "request_failed",
};
/** Native HTML parsing without executing JS; decode text after assembling chunks. */
async function htmlText(html: string) {
let title = "",
content = "",
truncated = false;
let skip = 0,
inTitle = 0;
const rewriter = new HTMLRewriter()
.on("script, style, noscript, template, svg, nav, header, footer, form", {
element(element) {
skip++;
element.onEndTag(() => {
skip--;
});
},
})
.on("title", {
element(element) {
inTitle++;
element.onEndTag(() => {
inTitle--;
});
},
text(chunk) {
title += clip(chunk.text, Math.max(0, 240 - title.length));
},
})
.on("p, div, article, main, section, h1, h2, h3, li, br, tr", {
element() {
if (!skip && content.length < WEB_MAX_CONTENT) content += "\n";
},
})
.onDocument({
text(chunk) {
if (skip || inTitle) return;
const remaining = WEB_MAX_CONTENT - content.length;
content += clip(chunk.text, Math.max(0, remaining));
if (chunk.text.length > remaining) truncated = true;
},
});
// Drain parsing without retaining a second rewritten HTML string.
await rewriter
.transform(new Response(html))
.body!.pipeTo(new WritableStream({ write() {} }));
return {
title: cleanText(decodeHTML(title), 240),
content: decodeHTML(content).trim(),
truncated,
};
}
export async function readWebUrl(
raw: string,
options: ToolExecutionOptions,
): Promise {
const deadline = new WebDeadline(WEB_READ_TIMEOUT, options.abortSignal);
try {
const requestedUrl = publicWebUrl(raw).href;
const result = (await deadline.run(() =>
Promise.resolve(
fetchTool.execute!(
{ url: requestedUrl },
{ ...options, abortSignal: deadline.signal },
),
),
)) as FetchResult;
if (!result.ok) return webFailure(codes[result.code], result.status);
const finalUrl = publicWebUrl(result.finalUrl).href;
if (
![
"text/html",
"application/xhtml+xml",
"text/plain",
"text/markdown",
"text/x-markdown",
].includes(result.contentType)
)
return webFailure("unsupported_content");
const body = result.body ?? "";
const extracted =
result.contentType === "text/html" ||
result.contentType === "application/xhtml+xml"
? await deadline.run(() => htmlText(body))
: {
title: "",
content: clip(body, WEB_MAX_CONTENT),
truncated: body.length > WEB_MAX_CONTENT,
};
deadline.remaining();
if (!extracted.content.trim()) return webFailure("request_failed");
return {
ok: true,
sources: [
await webSource({
title: extracted.title || new URL(finalUrl).hostname,
requestedUrl,
finalUrl,
sourceKind: "page",
contentType: result.contentType,
content: extracted.content,
truncated: result.truncated || extracted.truncated,
}),
],
};
} catch (error) {
return webError(error, deadline.signal);
} finally {
deadline.dispose();
}
}