// Detecting whether a page belongs to a standard.site publication.
//
// Per https://standard.site/docs/verification/ the authoritative check for a
// publication is the /.well-known/site.standard.publication endpoint (with the
// publication's base path appended for non-root publications), which returns
// the publication's at:// URI. tags are
// hints only. Documents are verified by their
// tag.
import { blobUrl, getRecord, isBlobRef, normalizeAtUri, parseAtUri, resolveDid } from './atproto'
import { PROBE_OPTIONS, requestText } from './http'
import type { DocInfo, DocumentRecord, PubInfo, PublicationRecord } from './types'
const WK_SUFFIX = '/.well-known/site.standard.publication'
// The two outcomes deserve different lifetimes. A hit means this origin *is* a
// publication, so the user is on a site we already talk to and freshness is
// what matters — a publisher who fixes a record should not wait a quarter of an
// hour to see it. A miss means the origin has nothing to do with us, and it is
// the answer nearly every site on the web gives: re-asking costs a request to a
// stranger and buys almost nothing, since sites rarely become publications
// mid-session. Anyone who cannot wait has the popup's Refresh button, which
// bypasses this cache entirely (see the `force` argument).
const CACHE_TTL_HIT = 5 * 60 * 1000
const CACHE_TTL_MISS = 60 * 60 * 1000
/** What every stored probe answer is keyed under; see `dropProbes`. */
const WK_CACHE_PREFIX = 'wk:'
/**
* Probes already in flight, so two tabs opening on one origin ask once. Keyed
* by well-known URL and cleared as soon as the probe settles; the storage cache
* below is what spans anything longer.
*/
const inFlight = new Map>()
/** The well-known URL for a publication base: origin, plus its path if any. */
export function wellKnownUrl(base: string): string {
const url = new URL(base)
const path = url.pathname.replace(/\/$/, '')
return url.origin + WK_SUFFIX + path
}
/**
* Probe a well-known URL for a publication at-uri. Cached in storage.session;
* `force` skips the read (not the write) so the popup's Refresh actually
* re-asks the origin instead of handing back the answer it is trying to
* replace.
*/
async function probeWellKnown(base: string, force = false): Promise {
const wkUrl = wellKnownUrl(base)
const cacheKey = WK_CACHE_PREFIX + wkUrl
if (!force) {
const cached = (await chrome.storage.session.get(cacheKey))[cacheKey] as
| { uri: string | null; at: number }
| undefined
if (cached) {
const ttl = cached.uri ? CACHE_TTL_HIT : CACHE_TTL_MISS
if (Date.now() - cached.at < ttl) return cached.uri
}
const pending = inFlight.get(wkUrl)
if (pending) return pending
}
const probe = (async () => {
let uri: string | null = null
try {
const text = (await requestText(wkUrl, PROBE_OPTIONS)).trim()
// May be handle-based (e.g. at://example.com/...); normalize to DID form.
if (text.startsWith('at://')) uri = await normalizeAtUri(text)
} catch (err) {
// A miss and a failure are the same to a caller here — neither yields a
// publication — but they are not the same to a reader of the logs.
console.debug('[substandard] well-known probe failed', wkUrl, err)
}
await chrome.storage.session.set({ [cacheKey]: { uri, at: Date.now() } })
return uri
})()
inFlight.set(wkUrl, probe)
try {
return await probe
} finally {
inFlight.delete(wkUrl)
}
}
/**
* Forget every stored probe answer, so the next detection asks the origins
* again. For the dev channel's Drop cache (src/popup/popup.ts), which drops
* this alongside the caches in src/lib/cache.ts; ordinary users have Refresh,
* which bypasses this cache for the one page they are looking at instead of
* making every origin worth re-probing.
*
* Probes already in flight are left alone. `inFlight` is only a dedup for the
* seconds a request is out, and cancelling one would strand its callers.
*
* Returns how many answers went, so the caller can say.
*/
export async function dropProbes(): Promise {
const all = await chrome.storage.session.get(null)
const keys = Object.keys(all).filter((k) => k.startsWith(WK_CACHE_PREFIX))
if (keys.length > 0) await chrome.storage.session.remove(keys)
return keys.length
}
/**
* Load and resolve a publication record from its at-uri. Null for a uri that
* is not a publication; throws when the record exists but cannot be fetched,
* so callers can tell "no publication" from "could not look".
*/
async function loadPublication(uri: string): Promise {
const parsed = parseAtUri(uri)
if (!parsed || parsed.collection !== 'site.standard.publication') return null
const { pds, handle } = await resolveDid(parsed.did)
const rec = await getRecord(
pds,
parsed.did,
parsed.collection,
parsed.rkey,
)
const icon = rec.value.icon
if (icon !== undefined && !isBlobRef(icon)) {
console.debug('[substandard] publication icon is not a blob ref', uri, icon)
}
return {
uri,
did: parsed.did,
handle,
pds,
record: rec.value,
iconUrl: isBlobRef(icon) ? blobUrl(pds, parsed.did, icon) : undefined,
verified: false,
}
}
async function loadDocument(uri: string): Promise {
const parsed = parseAtUri(uri)
if (!parsed || parsed.collection !== 'site.standard.document') return null
const { pds } = await resolveDid(parsed.did)
const rec = await getRecord(
pds,
parsed.did,
parsed.collection,
parsed.rkey,
)
return { uri, record: rec.value }
}
/**
* Full detection for a page. Candidate publication at-uris come from the page's
* link-tag hints, the document record's `site` field, and well-known probes of
* the origin and the page's first path segment. A candidate only counts as
* verified once the well-known endpoint at the publication record's own `url`
* returns its at-uri.
*/
export async function detectPage(
pageUrl: string,
pubHint?: string,
docHint?: string,
force = false,
): Promise<{ pub?: PubInfo; doc?: DocInfo }> {
const url = new URL(pageUrl)
const candidates: string[] = []
const push = (uri: string | null | undefined) => {
if (uri && parseAtUri(uri) && !candidates.includes(uri)) candidates.push(uri)
}
let doc: DocInfo | undefined
if (docHint) {
try {
const normDoc = await normalizeAtUri(docHint)
if (normDoc) doc = (await loadDocument(normDoc)) ?? undefined
} catch (err) {
// A dead document hint should not sink publication detection.
console.debug('[substandard] document load failed', docHint, err)
}
if (doc?.record.site.startsWith('at://')) {
push(await normalizeAtUri(doc.record.site))
}
}
if (pubHint) push(await normalizeAtUri(pubHint))
// Probe the origin, and one path level deep for non-root publications.
push(await probeWellKnown(url.origin, force))
const firstSeg = url.pathname.split('/').filter(Boolean)[0]
if (candidates.length === 0 && firstSeg) {
push(await probeWellKnown(`${url.origin}/${firstSeg}`, force))
}
// A document may point at its publication by https URL instead of at-uri.
if (candidates.length === 0 && doc?.record.site.startsWith('https://')) {
push(await probeWellKnown(doc.record.site, force))
}
let fallback: PubInfo | undefined
let loadFailures = 0
for (const uri of candidates) {
let pub: PubInfo | null
try {
pub = await loadPublication(uri)
} catch (err) {
console.debug('[substandard] publication load failed', uri, err)
loadFailures++
continue
}
if (!pub) continue
// Bidirectional check: the well-known endpoint at the record's own URL
// must return this at-uri, and that URL must be on the page's origin.
try {
if (new URL(pub.record.url).origin === url.origin) {
pub.verified = (await probeWellKnown(pub.record.url, force)) === uri
}
} catch {
// bad record.url — leave unverified
}
if (pub.verified) return { pub, doc }
fallback ??= pub
}
// Candidates existed but none could even be loaded: that is a lookup
// failure, not a page without a publication.
if (!fallback && loadFailures > 0) {
throw new Error(`all ${loadFailures} publication candidate(s) failed to load`)
}
return { pub: fallback, doc }
}