import type { ShardInfo, UrlEntry } from './types'; import { buildSitemapIndexXml, buildUrlsetXml, gzipCompress, gzipDecompress, measureXmlBytes, sanitizeLastmod, unescapeXml, } from './xml'; export function formatShardFilename(index: number): string { return `sitemap-${String(index).padStart(4, '0')}.xml.gz`; } export function parseUrlsetXml(xml: string): UrlEntry[] { const entries: UrlEntry[] = []; const urlRegex = /([\s\S]*?)<\/url>/g; const locRegex = /(.*?)<\/loc>/; const lastmodRegex = /(.*?)<\/lastmod>/; const changefreqRegex = /(.*?)<\/changefreq>/; const priorityRegex = /(.*?)<\/priority>/; let match: RegExpExecArray | null; while ((match = urlRegex.exec(xml)) !== null) { const block = match[1]; const locMatch = locRegex.exec(block); if (!locMatch) continue; const loc = unescapeXml(locMatch[1]); const rawLastmod = lastmodRegex.exec(block)?.[1]; const lastmod = rawLastmod ? unescapeXml(rawLastmod) : undefined; const changefreq = changefreqRegex.exec(block)?.[1] as UrlEntry['changefreq']; const priorityStr = priorityRegex.exec(block)?.[1]; const priority = priorityStr ? parseFloat(priorityStr) : undefined; entries.push({ loc, lastmod, changefreq, priority }); } return entries; } export function getStaticUrls(baseUrl: string, now = new Date()): UrlEntry[] { const normalized = baseUrl.replace(/\/+$/, ''); const today = sanitizeLastmod(undefined, now); return [ { loc: `${normalized}/`, lastmod: today, changefreq: 'daily', priority: 1.0, }, { loc: `${normalized}/about`, lastmod: today, changefreq: 'monthly', priority: 0.8, }, ]; } export async function loadShardEntries( bucket: R2Bucket, filename: string, ): Promise> { const existing = await bucket.get(`sitemaps/${filename}`); if (!existing) { return new Map(); } const arrayBuffer = await existing.arrayBuffer(); const xml = await gzipDecompress(arrayBuffer); const parsed = parseUrlsetXml(xml); const map = new Map(); for (const entry of parsed) { map.set(entry.loc, entry); } return map; } export async function flushShardToR2( bucket: R2Bucket, filename: string, entries: UrlEntry[], ): Promise { const xml = buildUrlsetXml(entries); const byteLength = measureXmlBytes(xml); const compressed = await gzipCompress(xml); await bucket.put(`sitemaps/${filename}`, compressed, { httpMetadata: { contentType: 'application/x-gzip', contentEncoding: 'gzip', cacheControl: 'public, max-age=3600, stale-while-revalidate=86400', }, }); return byteLength; } export async function updateSitemapIndex( bucket: R2Bucket, shards: ShardInfo[], baseUrl: string, ): Promise { const indexXml = buildSitemapIndexXml(shards, baseUrl); await bucket.put('sitemap.xml', indexXml, { httpMetadata: { contentType: 'application/xml', cacheControl: 'public, max-age=86400, s-maxage=86400', }, }); } /** * Splits an array of entries into a chunk that satisfies both maxUrls AND maxBytes, * and a remainder array. */ export function splitEntriesByLimits( entries: UrlEntry[], maxUrls: number, maxBytes: number, ): { chunk: UrlEntry[]; remainder: UrlEntry[] } { if (entries.length === 0) { return { chunk: [], remainder: [] }; } // Binary search or linear take up to bounds let count = Math.min(entries.length, maxUrls); let xml = buildUrlsetXml(entries.slice(0, count)); let bytes = measureXmlBytes(xml); while (bytes > maxBytes && count > 1) { // Step down proportionally const ratio = maxBytes / bytes; count = Math.max(1, Math.min(count - 1, Math.floor(count * ratio))); xml = buildUrlsetXml(entries.slice(0, count)); bytes = measureXmlBytes(xml); } return { chunk: entries.slice(0, count), remainder: entries.slice(count), }; }