From 5676cf2b745498d3a416d8a7f595eec27a5f88cd Mon Sep 17 00:00:00 2001 From: dame Date: Thu, 2 Jul 2026 23:19:43 -0400 Subject: [PATCH] Route all waypoint URLs from the explore search into the explorer (#45) Close every gap so any waypoint URL round-trips to the correct explorer path: offprint/pckt flat record links, anisota.net document links, Anisota subdomains, and Aturi's own /profile and /explore links. Add an isSupportedHost() helper (www-stripping + subdomain opt-in) and use it in the extension popup and /api/resolve so those surfaces recognize Anisota subdomains too. --- extension/entrypoints/popup/App.tsx | 4 +- .../src/__tests__/reverseParsers.test.ts | 71 +++++++++++++- packages/waypoints/src/reverseParsers.ts | 94 ++++++++++++++++++- src/app/api/resolve/route.ts | 5 +- src/utils/atproto/searchRouting.ts | 56 ++++++++++- src/utils/reverseParsers.ts | 94 ++++++++++++++++++- 6 files changed, 311 insertions(+), 13 deletions(-) diff --git a/extension/entrypoints/popup/App.tsx b/extension/entrypoints/popup/App.tsx index a01060c..d49de30 100644 --- a/extension/entrypoints/popup/App.tsx +++ b/extension/entrypoints/popup/App.tsx @@ -3,7 +3,7 @@ import { browser } from '#imports'; import { MousePointer2, Telescope } from 'lucide-react'; import type { ReverseMatch } from '@aturi/reverseParsers'; import type { WaypointActivity, WaypointData, WaypointType } from '@aturi/waypoints.data'; -import { matchSupportedUrl, parseAtUri, SUPPORTED_HOSTS } from '@aturi/reverseParsers'; +import { matchSupportedUrl, parseAtUri, isSupportedHost } from '@aturi/reverseParsers'; import { waypointActivity } from '@aturi/waypoints.data'; import { matchCustomUrl } from '../../lib/template'; import { @@ -146,7 +146,7 @@ export default function App() { return; } - const isKnownHost = SUPPORTED_HOSTS.includes(url.hostname.replace(/^www\./, '')); + const isKnownHost = isSupportedHost(url.hostname); let match = matchSupportedUrl(url) ?? matchCustomUrl(url, prefs.customWaypoints); diff --git a/packages/waypoints/src/__tests__/reverseParsers.test.ts b/packages/waypoints/src/__tests__/reverseParsers.test.ts index 52f4343..8efcd10 100644 --- a/packages/waypoints/src/__tests__/reverseParsers.test.ts +++ b/packages/waypoints/src/__tests__/reverseParsers.test.ts @@ -1,5 +1,5 @@ import { describe, it, expect } from 'vitest'; -import { matchSupportedUrl, parseAtUri } from '../reverseParsers'; +import { matchSupportedUrl, parseAtUri, isSupportedHost } from '../reverseParsers'; function match(url: string) { return matchSupportedUrl(new URL(url)); @@ -27,6 +27,20 @@ describe('matchSupportedUrl - Bluesky family', () => { expect(m?.parsed.collection).toBe('app.bsky.graph.list'); }); + it('parses an anisota subdomain the same as anisota.net', () => { + const m = match('https://eclose.anisota.net/profile/alice.bsky.social/post/abc'); + expect(m?.source).toBe('anisota'); + expect(m?.parsed.type).toBe('post'); + expect(m?.parsed.rkey).toBe('abc'); + expect(m?.parsed.collection).toBe('app.bsky.feed.post'); + }); + + it('does not treat a lookalike host as an anisota subdomain', () => { + // Must be a real subdomain of anisota.net, not just a suffix match. + expect(match('https://notanisota.net/profile/alice.bsky.social')).toBeNull(); + expect(match('https://anisota.net.evil.com/profile/alice')).toBeNull(); + }); + it('parses blacksky', () => { const m = match('https://blacksky.community/profile/alice.bsky.social/post/abc'); expect(m?.source).toBe('blacksky'); @@ -105,11 +119,66 @@ describe('matchSupportedUrl - other apps', () => { expect(m?.parsed.type).toBe('profile'); }); + it('parses an anisota reader document into a site.standard.document uri', () => { + const m = match('https://anisota.net/profile/did:plc:xyz/document/rk123'); + expect(m?.source).toBe('anisota'); + expect(m?.parsed.type).toBe('record'); + expect(m?.parsed.collection).toBe('site.standard.document'); + expect(m?.parsed.rkey).toBe('rk123'); + expect(m?.parsed.did).toBe('did:plc:xyz'); + }); + + it('parses an offprint record', () => { + const m = match('https://offprint.app/did:plc:xyz/site.standard.document/rk123'); + expect(m?.source).toBe('offprint'); + expect(m?.parsed.type).toBe('record'); + expect(m?.parsed.collection).toBe('site.standard.document'); + expect(m?.parsed.rkey).toBe('rk123'); + expect(m?.parsed.did).toBe('did:plc:xyz'); + }); + + it('parses a pckt record', () => { + const m = match('https://pckt.blog/did:plc:xyz/pub.leaflet.document/rk123'); + expect(m?.source).toBe('pckt'); + expect(m?.parsed.type).toBe('record'); + expect(m?.parsed.collection).toBe('pub.leaflet.document'); + expect(m?.parsed.rkey).toBe('rk123'); + }); + + it('does not treat non-record offprint/pckt paths as records', () => { + // No NSID collection segment -> not a record link. + expect(match('https://offprint.app/settings')).toBeNull(); + expect(match('https://pckt.blog/alice.bsky.social/notacollection/rk')).toBeNull(); + }); + it('ignores unsupported hosts', () => { expect(match('https://example.com/profile/alice')).toBeNull(); }); }); +describe('isSupportedHost', () => { + it('recognizes exact hosts and strips www', () => { + expect(isSupportedHost('bsky.app')).toBe(true); + expect(isSupportedHost('www.anisota.net')).toBe(true); + expect(isSupportedHost('offprint.app')).toBe(true); + expect(isSupportedHost('ANISOTA.NET')).toBe(true); + }); + + it('recognizes anisota subdomains', () => { + expect(isSupportedHost('eclose.anisota.net')).toBe(true); + expect(isSupportedHost('sub.eclose.anisota.net')).toBe(true); + }); + + it('rejects lookalikes and non-subdomain hosts', () => { + expect(isSupportedHost('notanisota.net')).toBe(false); + expect(isSupportedHost('anisota.net.evil.com')).toBe(false); + expect(isSupportedHost('bsky.app.evil.com')).toBe(false); + expect(isSupportedHost('example.com')).toBe(false); + // Only opted-in hosts match subdomains; bsky.app does not. + expect(isSupportedHost('foo.bsky.app')).toBe(false); + }); +}); + describe('parseAtUri', () => { it('parses a record at-uri', () => { const m = parseAtUri('at://did:plc:x/app.bsky.feed.post/abc'); diff --git a/packages/waypoints/src/reverseParsers.ts b/packages/waypoints/src/reverseParsers.ts index 7a528f9..82aa1a6 100644 --- a/packages/waypoints/src/reverseParsers.ts +++ b/packages/waypoints/src/reverseParsers.ts @@ -36,6 +36,11 @@ export type ReverseMatch = { type HostConfig = { source: SourceApp; hosts: string[]; + /** + * When true, any subdomain of the listed hosts is treated as this source + * too (e.g. Anisota gives each publication its own `*.anisota.net` host). + */ + matchSubdomains?: boolean; }; /** @@ -49,15 +54,35 @@ const BLUESKY_FAMILY: HostConfig[] = [ { source: 'catsky', hosts: ['catsky.social'] }, { source: 'deer', hosts: ['deer.social'] }, { source: 'mu', hosts: ['mu.social'] }, - { source: 'anisota', hosts: ['anisota.net'] }, + { source: 'anisota', hosts: ['anisota.net'], matchSubdomains: true }, ]; +/** + * Base hosts whose subdomains are also recognized (e.g. `*.anisota.net`). + * Derived from the `matchSubdomains` opt-in so there's one source of truth. + */ +const SUBDOMAIN_HOSTS: string[] = BLUESKY_FAMILY.filter(f => f.matchSubdomains).flatMap( + f => f.hosts, +); + function normalizeHost(host: string): string { return host.toLowerCase().replace(/^www\./, ''); } +/** True when `host` is a subdomain of `base` (a real dotted-label boundary). */ +function isSubdomainOf(host: string, base: string): boolean { + return host.endsWith(`.${base}`); +} + +/** Exact host match, or a subdomain match when the config opts in. */ +function hostMatchesConfig(host: string, config: HostConfig): boolean { + if (config.hosts.includes(host)) return true; + if (!config.matchSubdomains) return false; + return config.hosts.some(h => isSubdomainOf(host, h)); +} + function matchBlueskyFamily(host: string, parts: string[]): ReverseMatch | null { - const entry = BLUESKY_FAMILY.find(f => f.hosts.includes(host)); + const entry = BLUESKY_FAMILY.find(f => hostMatchesConfig(host, f)); if (!entry) return null; if (parts[0] !== 'profile' || !parts[1]) return null; @@ -104,6 +129,24 @@ function matchBlueskyFamily(host: string, parts: string[]): ReverseMatch | null }; } + // Anisota's reader addresses Standard Site / Leaflet documents at + // `/profile/:handle/document/:rkey` without the collection NSID in the URL. + // Mirror Standard Reader's convention and reconstruct the canonical + // `site.standard.document` collection so the record still resolves. + if (parts[2] === 'document' && parts[3]) { + return { + source: entry.source, + parsed: { + type: 'record', + uri: `at://${handle}/site.standard.document/${parts[3]}`, + handle, + did, + collection: 'site.standard.document', + rkey: parts[3], + }, + }; + } + return { source: entry.source, parsed: { @@ -454,6 +497,37 @@ function matchStandardReader(host: string, parts: string[]): ReverseMatch | null return null; } +/** + * Offprint (offprint.app) and pckt (pckt.blog) address publications at the + * flat path `///` — the collection NSID sits in + * the path verbatim, so we can round-trip it straight back. Both only expose + * record-level URLs, so a profile-only path has no meaningful match. + */ +function matchFlatRecordHost( + host: string, + parts: string[], + target: { host: string; source: SourceApp }, +): ReverseMatch | null { + if (host !== target.host) return null; + const [handle, collection, rkey] = parts; + if (!handle || !collection || !rkey) return null; + // Guard against non-record pages (settings, landing, …): a real record path + // always carries an NSID collection segment. + if (!collection.includes('.')) return null; + const did = handle.startsWith('did:') ? handle : undefined; + return { + source: target.source, + parsed: { + type: inferType(collection), + uri: `at://${handle}/${collection}/${rkey}`, + handle, + did, + collection, + rkey, + }, + }; +} + /** * Taproot (atproto.at): a generic AT-URI explorer addressed at * `/uri/at://[//]`. @@ -515,6 +589,8 @@ export function matchSupportedUrl(url: URL): ReverseMatch | null { matchSifa(host, parts) || matchBlento(host, parts) || matchStandardReader(host, parts) || + matchFlatRecordHost(host, parts, { host: 'offprint.app', source: 'offprint' }) || + matchFlatRecordHost(host, parts, { host: 'pckt.blog', source: 'pckt' }) || matchTaproot(host, url.pathname) ); } @@ -575,5 +651,19 @@ export const SUPPORTED_HOSTS: string[] = [ 'sifa.id', 'blento.app', 'standard-reader.app', + 'offprint.app', + 'pckt.blog', 'atproto.at', ]; + +/** + * Whether a hostname belongs to a supported waypoint. Prefer this over a raw + * `SUPPORTED_HOSTS.includes(...)` check: it strips a leading `www.` and also + * recognizes subdomains of hosts that opt in (e.g. `eclose.anisota.net`), so + * the extension popup and resolve API treat those tabs as known. + */ +export function isSupportedHost(host: string): boolean { + const h = normalizeHost(host); + if (SUPPORTED_HOSTS.includes(h)) return true; + return SUBDOMAIN_HOSTS.some(base => isSubdomainOf(h, base)); +} diff --git a/src/app/api/resolve/route.ts b/src/app/api/resolve/route.ts index 8a60278..6d99df8 100644 --- a/src/app/api/resolve/route.ts +++ b/src/app/api/resolve/route.ts @@ -2,7 +2,7 @@ import { NextRequest, NextResponse } from 'next/server'; import { matchSupportedUrl, parseAtUri, - SUPPORTED_HOSTS, + isSupportedHost, type ReverseMatch, } from '@/utils/reverseParsers'; import { @@ -93,8 +93,7 @@ export async function GET(request: NextRequest) { return jsonError(400, 'Only http(s) URLs are supported'); } - const host = parsedUrl.hostname.replace(/^www\./, '').toLowerCase(); - isKnownHost = SUPPORTED_HOSTS.includes(host); + isKnownHost = isSupportedHost(parsedUrl.hostname); match = matchSupportedUrl(parsedUrl); if (match) { diff --git a/src/utils/atproto/searchRouting.ts b/src/utils/atproto/searchRouting.ts index 94ebaa7..672ada6 100644 --- a/src/utils/atproto/searchRouting.ts +++ b/src/utils/atproto/searchRouting.ts @@ -26,6 +26,52 @@ function explorePathFromParsed(parsed: ParsedURI): string { return `/explore/${repo}`; } +/** + * Aturi's own links (the main app's `/profile/...` shapes and the explorer's + * `/explore/...` shapes) route straight back into the explorer. Pasting an + * `aturi.to` URL should land on the same record rather than being treated as + * a PDS host, so we recognise our own domain explicitly. + */ +function explorePathFromAturiUrl(url: URL): string | null { + const host = url.hostname.toLowerCase().replace(/^www\./, ''); + if (host !== 'aturi.to') return null; + const parts = url.pathname.split('/').filter(Boolean); + + // `/explore/[/[/]]` — already an explorer path. + if (parts[0] === 'explore') { + const [, id, collection, rkey] = parts; + if (!id) return null; + const repo = encodeRepo(id); + if (collection && rkey) { + return `/explore/${repo}/${collection}/${encodeURIComponent(rkey)}`; + } + if (collection) return `/explore/${repo}/${collection}`; + return `/explore/${repo}`; + } + + // `/profile/[/(post|lists)/]` or `/profile///`. + if (parts[0] === 'profile') { + const id = parts[1]; + if (!id) return null; + const repo = encodeRepo(id); + const seg = parts[2]; + const rkey = parts[3]; + if (seg && rkey) { + if (seg === 'post') { + return `/explore/${repo}/app.bsky.feed.post/${encodeURIComponent(rkey)}`; + } + if (seg === 'lists' || seg === 'list') { + return `/explore/${repo}/app.bsky.graph.list/${encodeURIComponent(rkey)}`; + } + // Generic record route: the collection NSID sits in the path verbatim. + return `/explore/${repo}/${seg}/${encodeURIComponent(rkey)}`; + } + return `/explore/${repo}`; + } + + return null; +} + /** * Bare hostnames that begin with `pds.` are overwhelmingly atproto PDS * hosts (pds.atpota.to, pds.bsky.network, …). Anything else without a @@ -57,10 +103,14 @@ export function resolveSearchPath(rawInput: string): string | null { // 2. Explicit URL. Anything with a protocol is a URL, not a handle. if (/^https?:\/\//i.test(v)) { - // 2a. Known waypoint apps (bsky.app, pdsls.dev, …) — reverse-parse the - // URL back into repo/collection/rkey and drill into that record. + // 2a. Known waypoint apps (bsky.app, pdsls.dev, …) plus Aturi's own + // links — reverse-parse the URL back into repo/collection/rkey and + // drill into that record. try { - const match = matchSupportedUrl(new URL(v)); + const url = new URL(v); + const own = explorePathFromAturiUrl(url); + if (own) return own; + const match = matchSupportedUrl(url); if (match) return explorePathFromParsed(match.parsed); } catch { // Not a parseable URL — fall through to PDS host handling. diff --git a/src/utils/reverseParsers.ts b/src/utils/reverseParsers.ts index 7a528f9..82aa1a6 100644 --- a/src/utils/reverseParsers.ts +++ b/src/utils/reverseParsers.ts @@ -36,6 +36,11 @@ export type ReverseMatch = { type HostConfig = { source: SourceApp; hosts: string[]; + /** + * When true, any subdomain of the listed hosts is treated as this source + * too (e.g. Anisota gives each publication its own `*.anisota.net` host). + */ + matchSubdomains?: boolean; }; /** @@ -49,15 +54,35 @@ const BLUESKY_FAMILY: HostConfig[] = [ { source: 'catsky', hosts: ['catsky.social'] }, { source: 'deer', hosts: ['deer.social'] }, { source: 'mu', hosts: ['mu.social'] }, - { source: 'anisota', hosts: ['anisota.net'] }, + { source: 'anisota', hosts: ['anisota.net'], matchSubdomains: true }, ]; +/** + * Base hosts whose subdomains are also recognized (e.g. `*.anisota.net`). + * Derived from the `matchSubdomains` opt-in so there's one source of truth. + */ +const SUBDOMAIN_HOSTS: string[] = BLUESKY_FAMILY.filter(f => f.matchSubdomains).flatMap( + f => f.hosts, +); + function normalizeHost(host: string): string { return host.toLowerCase().replace(/^www\./, ''); } +/** True when `host` is a subdomain of `base` (a real dotted-label boundary). */ +function isSubdomainOf(host: string, base: string): boolean { + return host.endsWith(`.${base}`); +} + +/** Exact host match, or a subdomain match when the config opts in. */ +function hostMatchesConfig(host: string, config: HostConfig): boolean { + if (config.hosts.includes(host)) return true; + if (!config.matchSubdomains) return false; + return config.hosts.some(h => isSubdomainOf(host, h)); +} + function matchBlueskyFamily(host: string, parts: string[]): ReverseMatch | null { - const entry = BLUESKY_FAMILY.find(f => f.hosts.includes(host)); + const entry = BLUESKY_FAMILY.find(f => hostMatchesConfig(host, f)); if (!entry) return null; if (parts[0] !== 'profile' || !parts[1]) return null; @@ -104,6 +129,24 @@ function matchBlueskyFamily(host: string, parts: string[]): ReverseMatch | null }; } + // Anisota's reader addresses Standard Site / Leaflet documents at + // `/profile/:handle/document/:rkey` without the collection NSID in the URL. + // Mirror Standard Reader's convention and reconstruct the canonical + // `site.standard.document` collection so the record still resolves. + if (parts[2] === 'document' && parts[3]) { + return { + source: entry.source, + parsed: { + type: 'record', + uri: `at://${handle}/site.standard.document/${parts[3]}`, + handle, + did, + collection: 'site.standard.document', + rkey: parts[3], + }, + }; + } + return { source: entry.source, parsed: { @@ -454,6 +497,37 @@ function matchStandardReader(host: string, parts: string[]): ReverseMatch | null return null; } +/** + * Offprint (offprint.app) and pckt (pckt.blog) address publications at the + * flat path `///` — the collection NSID sits in + * the path verbatim, so we can round-trip it straight back. Both only expose + * record-level URLs, so a profile-only path has no meaningful match. + */ +function matchFlatRecordHost( + host: string, + parts: string[], + target: { host: string; source: SourceApp }, +): ReverseMatch | null { + if (host !== target.host) return null; + const [handle, collection, rkey] = parts; + if (!handle || !collection || !rkey) return null; + // Guard against non-record pages (settings, landing, …): a real record path + // always carries an NSID collection segment. + if (!collection.includes('.')) return null; + const did = handle.startsWith('did:') ? handle : undefined; + return { + source: target.source, + parsed: { + type: inferType(collection), + uri: `at://${handle}/${collection}/${rkey}`, + handle, + did, + collection, + rkey, + }, + }; +} + /** * Taproot (atproto.at): a generic AT-URI explorer addressed at * `/uri/at://[//]`. @@ -515,6 +589,8 @@ export function matchSupportedUrl(url: URL): ReverseMatch | null { matchSifa(host, parts) || matchBlento(host, parts) || matchStandardReader(host, parts) || + matchFlatRecordHost(host, parts, { host: 'offprint.app', source: 'offprint' }) || + matchFlatRecordHost(host, parts, { host: 'pckt.blog', source: 'pckt' }) || matchTaproot(host, url.pathname) ); } @@ -575,5 +651,19 @@ export const SUPPORTED_HOSTS: string[] = [ 'sifa.id', 'blento.app', 'standard-reader.app', + 'offprint.app', + 'pckt.blog', 'atproto.at', ]; + +/** + * Whether a hostname belongs to a supported waypoint. Prefer this over a raw + * `SUPPORTED_HOSTS.includes(...)` check: it strips a leading `www.` and also + * recognizes subdomains of hosts that opt in (e.g. `eclose.anisota.net`), so + * the extension popup and resolve API treat those tabs as known. + */ +export function isSupportedHost(host: string): boolean { + const h = normalizeHost(host); + if (SUPPORTED_HOSTS.includes(h)) return true; + return SUBDOMAIN_HOSTS.some(base => isSubdomainOf(h, base)); +} -- 2.51.2