// Detecting whether a page belongs to a standard.site publication. // // Per https://standard.site/docs/verification/ the authoritative check for a // publication is the /.well-known/site.standard.publication endpoint (with the // publication's base path appended for non-root publications), which returns // the publication's at:// URI. tags are // hints only. Documents are verified by their // tag. import { blobUrl, getRecord, isBlobRef, normalizeAtUri, parseAtUri, resolveDid } from './atproto' import { PROBE_OPTIONS, requestText } from './http' import type { DocInfo, DocumentRecord, PubInfo, PublicationRecord } from './types' const WK_SUFFIX = '/.well-known/site.standard.publication' // The two outcomes deserve different lifetimes. A hit means this origin *is* a // publication, so the user is on a site we already talk to and freshness is // what matters — a publisher who fixes a record should not wait a quarter of an // hour to see it. A miss means the origin has nothing to do with us, and it is // the answer nearly every site on the web gives: re-asking costs a request to a // stranger and buys almost nothing, since sites rarely become publications // mid-session. Anyone who cannot wait has the popup's Refresh button, which // bypasses this cache entirely (see the `force` argument). const CACHE_TTL_HIT = 5 * 60 * 1000 const CACHE_TTL_MISS = 60 * 60 * 1000 /** What every stored probe answer is keyed under; see `dropProbes`. */ const WK_CACHE_PREFIX = 'wk:' /** * Probes already in flight, so two tabs opening on one origin ask once. Keyed * by well-known URL and cleared as soon as the probe settles; the storage cache * below is what spans anything longer. */ const inFlight = new Map>() /** The well-known URL for a publication base: origin, plus its path if any. */ export function wellKnownUrl(base: string): string { const url = new URL(base) const path = url.pathname.replace(/\/$/, '') return url.origin + WK_SUFFIX + path } /** * Probe a well-known URL for a publication at-uri. Cached in storage.session; * `force` skips the read (not the write) so the popup's Refresh actually * re-asks the origin instead of handing back the answer it is trying to * replace. */ async function probeWellKnown(base: string, force = false): Promise { const wkUrl = wellKnownUrl(base) const cacheKey = WK_CACHE_PREFIX + wkUrl if (!force) { const cached = (await chrome.storage.session.get(cacheKey))[cacheKey] as | { uri: string | null; at: number } | undefined if (cached) { const ttl = cached.uri ? CACHE_TTL_HIT : CACHE_TTL_MISS if (Date.now() - cached.at < ttl) return cached.uri } const pending = inFlight.get(wkUrl) if (pending) return pending } const probe = (async () => { let uri: string | null = null try { const text = (await requestText(wkUrl, PROBE_OPTIONS)).trim() // May be handle-based (e.g. at://example.com/...); normalize to DID form. if (text.startsWith('at://')) uri = await normalizeAtUri(text) } catch (err) { // A miss and a failure are the same to a caller here — neither yields a // publication — but they are not the same to a reader of the logs. console.debug('[substandard] well-known probe failed', wkUrl, err) } await chrome.storage.session.set({ [cacheKey]: { uri, at: Date.now() } }) return uri })() inFlight.set(wkUrl, probe) try { return await probe } finally { inFlight.delete(wkUrl) } } /** * Forget every stored probe answer, so the next detection asks the origins * again. For the dev channel's Drop cache (src/popup/popup.ts), which drops * this alongside the caches in src/lib/cache.ts; ordinary users have Refresh, * which bypasses this cache for the one page they are looking at instead of * making every origin worth re-probing. * * Probes already in flight are left alone. `inFlight` is only a dedup for the * seconds a request is out, and cancelling one would strand its callers. * * Returns how many answers went, so the caller can say. */ export async function dropProbes(): Promise { const all = await chrome.storage.session.get(null) const keys = Object.keys(all).filter((k) => k.startsWith(WK_CACHE_PREFIX)) if (keys.length > 0) await chrome.storage.session.remove(keys) return keys.length } /** * Load and resolve a publication record from its at-uri. Null for a uri that * is not a publication; throws when the record exists but cannot be fetched, * so callers can tell "no publication" from "could not look". */ async function loadPublication(uri: string): Promise { const parsed = parseAtUri(uri) if (!parsed || parsed.collection !== 'site.standard.publication') return null const { pds, handle } = await resolveDid(parsed.did) const rec = await getRecord( pds, parsed.did, parsed.collection, parsed.rkey, ) const icon = rec.value.icon if (icon !== undefined && !isBlobRef(icon)) { console.debug('[substandard] publication icon is not a blob ref', uri, icon) } return { uri, did: parsed.did, handle, pds, record: rec.value, iconUrl: isBlobRef(icon) ? blobUrl(pds, parsed.did, icon) : undefined, verified: false, } } async function loadDocument(uri: string): Promise { const parsed = parseAtUri(uri) if (!parsed || parsed.collection !== 'site.standard.document') return null const { pds } = await resolveDid(parsed.did) const rec = await getRecord( pds, parsed.did, parsed.collection, parsed.rkey, ) return { uri, record: rec.value } } /** * Full detection for a page. Candidate publication at-uris come from the page's * link-tag hints, the document record's `site` field, and well-known probes of * the origin and the page's first path segment. A candidate only counts as * verified once the well-known endpoint at the publication record's own `url` * returns its at-uri. */ export async function detectPage( pageUrl: string, pubHint?: string, docHint?: string, force = false, ): Promise<{ pub?: PubInfo; doc?: DocInfo }> { const url = new URL(pageUrl) const candidates: string[] = [] const push = (uri: string | null | undefined) => { if (uri && parseAtUri(uri) && !candidates.includes(uri)) candidates.push(uri) } let doc: DocInfo | undefined if (docHint) { try { const normDoc = await normalizeAtUri(docHint) if (normDoc) doc = (await loadDocument(normDoc)) ?? undefined } catch (err) { // A dead document hint should not sink publication detection. console.debug('[substandard] document load failed', docHint, err) } if (doc?.record.site.startsWith('at://')) { push(await normalizeAtUri(doc.record.site)) } } if (pubHint) push(await normalizeAtUri(pubHint)) // Probe the origin, and one path level deep for non-root publications. push(await probeWellKnown(url.origin, force)) const firstSeg = url.pathname.split('/').filter(Boolean)[0] if (candidates.length === 0 && firstSeg) { push(await probeWellKnown(`${url.origin}/${firstSeg}`, force)) } // A document may point at its publication by https URL instead of at-uri. if (candidates.length === 0 && doc?.record.site.startsWith('https://')) { push(await probeWellKnown(doc.record.site, force)) } let fallback: PubInfo | undefined let loadFailures = 0 for (const uri of candidates) { let pub: PubInfo | null try { pub = await loadPublication(uri) } catch (err) { console.debug('[substandard] publication load failed', uri, err) loadFailures++ continue } if (!pub) continue // Bidirectional check: the well-known endpoint at the record's own URL // must return this at-uri, and that URL must be on the page's origin. try { if (new URL(pub.record.url).origin === url.origin) { pub.verified = (await probeWellKnown(pub.record.url, force)) === uri } } catch { // bad record.url — leave unverified } if (pub.verified) return { pub, doc } fallback ??= pub } // Candidates existed but none could even be loaded: that is a lookup // failure, not a page without a publication. if (!fallback && loadFailures > 0) { throw new Error(`all ${loadFailures} publication candidate(s) failed to load`) } return { pub: fallback, doc } }