// Similarity sources and rank fusion. // // Each source returns an ordered list of similar artists for a seed. Scores are // not comparable across sources (ListenBrainz emits raw session counts in the // thousands, Deezer emits nothing at all), so fusion happens over ranks only. export type Artist = { name: string; deezerId?: number; mbid?: string; fans?: number; /** Which sources voted for this artist. Drives the source-agreement knob. */ sources: string[]; /** Fused reciprocal-rank score. */ score: number; /** Source-native score, kept only for inspection; never fused directly. */ rawScore?: number; }; export type Track = { id: number; title: string; artist: string; artistId: number; album: string; duration: number; rank: number; explicit: boolean; preview: string | null; link: string; // Populated by enrichTracks(). bpm?: number; gain?: number; year?: number; /** * Position in this artist's own top-tracks list, 0 = biggest hit, 1 = deepest * cut available. This is what makes "deep cuts" mean deep *for that artist* * rather than merely globally less famous. */ depth: number; /** Which similarity sources nominated this track's artist. */ sources?: string[]; }; /** * Deezer allows roughly 50 requests per 5 seconds and signals overage two ways: * HTTP 429, or HTTP 200 carrying an error object. Both must be retried, or * enrichment silently loses half its fields. * * The sleep here is backoff against a remote rate limiter — a network-service * boundary with no local state to observe. There is no deterministic signal to * wait on: the only way to learn the quota window has reopened is to ask again. */ const json = async (url: string, attempt = 0, signal?: AbortSignal): Promise => { // An aborted signal makes fetch reject before any of the retry logic below, // so a cancelled pass stops rather than backing off and asking again. const res = await fetch(url, { signal }); const retryable = res.status === 429 || res.status >= 500; if (retryable && attempt < 4) { await new Promise((r) => setTimeout(r, 400 * 2 ** attempt)); return json(url, attempt + 1, signal); } if (!res.ok) throw new Error(`${res.status} ${url}`); const body = await res.json(); if (body?.error) { const quota = body.error.code === 4 || /quota|limit/i.test(body.error.message ?? ''); if (quota && attempt < 4) { await new Promise((r) => setTimeout(r, 600 * 2 ** attempt)); return json(url, attempt + 1, signal); } throw new Error(body.error.message ?? 'upstream error'); } return body; }; /** * Runs `work` over `items` with bounded concurrency, preserving order. * * An aborted `signal` stops the runners from taking any further item, so a * cancelled pass costs at most the requests already in flight rather than the * whole list. Items never started are left null, exactly as a failed one is. */ export async function pool( items: T[], limit: number, work: (item: T, index: number) => Promise, signal?: AbortSignal, ): Promise<(R | null)[]> { const out: (R | null)[] = new Array(items.length).fill(null); let next = 0; const runners = Array.from({ length: Math.min(limit, items.length) }, async () => { while (next < items.length && !signal?.aborted) { const i = next++; try { out[i] = await work(items[i], i); } catch { out[i] = null; } } }); await Promise.all(runners); return out; } /** * Deezer ranks search results by popularity, not by name match, so "Four Tet" * can return a novelty act ahead of the real artist. An exact name match among * the first several results beats whatever Deezer put first; failing that, the * most-followed result is the better guess. */ export async function findDeezerArtist(name: string): Promise { const d = await json(`/dz/search/artist?limit=10&q=${encodeURIComponent(name)}`); const rows: any[] = d.data ?? []; if (!rows.length) return null; const exact = rows.filter((a) => norm(a.name) === norm(name)); const best = exact.length ? exact.reduce((x, y) => ((y.nb_fan ?? 0) > (x.nb_fan ?? 0) ? y : x)) : rows[0]; return { name: best.name, deezerId: best.id, fans: best.nb_fan, sources: [], score: 0 }; } /** One row of `searchArtists()` below — enough to show and pick a match. */ export type ArtistSuggestion = { id: string; name: string; picture?: string }; /** * Search both catalogues: a Deezer-only dropdown hides artists that our * Last.fm seed fallback can resolve. Interleave source ranks, deduplicate by * normalised name (keeping Deezer artwork), then prefer exact/prefix matches * before truncating. One failed or slow source must not hide the other's hits. */ export async function searchArtists(name: string, limit: number): Promise { name = name.trim(); if (!name || limit <= 0) return []; const query = encodeURIComponent(name); const search = (url: string) => json(url, 0, AbortSignal.timeout(4000)).catch(() => null); const [dz, fm] = await Promise.all([ search(`/dz/search/artist?limit=${limit}&q=${query}`), search(`/fm?method=artist.search&artist=${query}&limit=${limit}`), ]); const deezer: ArtistSuggestion[] = (dz?.data ?? []).map((a: any) => ({ id: `deezer:${a.id}`, name: a.name, picture: `/dz/artist/${a.id}/image?size=small`, })); const lastfm: ArtistSuggestion[] = (fm?.results?.artistmatches?.artist ?? []).map((a: any) => ({ id: `lastfm:${norm(a.name)}`, name: a.name, })); const merged = new Map(); for (let i = 0; i < Math.max(deezer.length, lastfm.length); i++) { for (const artist of [deezer[i], lastfm[i]]) { if (!artist) continue; const key = norm(artist.name); const existing = merged.get(key); if (!existing) merged.set(key, artist); else if (!existing.picture && artist.picture) merged.set(key, artist); } } const key = norm(name); const relevance = (a: ArtistSuggestion) => { const candidate = norm(a.name); return candidate === key ? 0 : candidate.startsWith(key) ? 1 : candidate.includes(key) ? 2 : 3; }; return [...merged.values()].sort((a, b) => relevance(a) - relevance(b)).slice(0, limit); } // Called directly rather than through a Worker route: both answer // access-control-allow-origin: *, and going direct keeps MusicBrainz's // one-request-per-second-per-IP limit spread across users' own IPs instead of // concentrating every user behind shared Worker egress. Nothing here may add a // request header — json() sends none, so none of these requests ever // preflights, and a single custom header would double all of them. const MB = 'https://musicbrainz.org/ws/2'; const LB = 'https://labs.api.listenbrainz.org'; export async function findMbid(name: string): Promise { const d = await json(`${MB}/artist?fmt=json&limit=1&query=${encodeURIComponent(`artist:${name}`)}`); const a = d.artists?.[0]; // MusicBrainz search is fuzzy; a low score usually means no real match. return a && a.score >= 80 ? a.id : null; } /** Deezer's own related artists. Caps at 20 and ignores `limit`. */ async function deezerRelated(deezerId: number): Promise { const d = await json(`/dz/artist/${deezerId}/related`); return (d.data ?? []).map((a: any) => ({ name: a.name, deezerId: a.id, fans: a.nb_fan, sources: ['deezer'], score: 0, })); } const LB_ALGORITHM = 'session_based_days_7500_session_300_contribution_5_threshold_10_limit_100_filter_True_skip_30'; /** ListenBrainz listening-session co-occurrence. Returns ~100 scored artists. */ async function listenBrainzSimilar(mbid: string): Promise { const d = await json(`${LB}/similar-artists/json?algorithm=${LB_ALGORITHM}&artist_mbids=${mbid}`); const rows: any[] = Array.isArray(d) ? d : []; return rows .sort((a, b) => b.score - a.score) .map((a) => ({ name: a.name, mbid: a.artist_mbid, sources: ['listenbrainz'], score: 0, rawScore: a.score, })); } /** * Co-occurrence counts reward ubiquity: a universally popular artist shares * listening sessions with everything, which is how the Beastie Boys end up * adjacent to Radiohead. Dividing by audience size converts the raw count into * something closer to lift, promoting artists that co-occur *more than their * size explains*. Artists with no known audience keep their rank untouched. */ export function penalisePopularity(roster: Artist[], strength: number): Artist[] { if (strength <= 0) return roster; const fanValues = roster.map((a) => a.fans ?? 0).filter((f) => f > 0); if (!fanValues.length) return roster; const median = fanValues.sort((a, b) => a - b)[Math.floor(fanValues.length / 2)]; return [...roster] .map((a) => { if (!a.fans) return a; // ratio > 1 for artists bigger than the roster median. const ratio = a.fans / median; const damped = 1 / Math.pow(Math.max(ratio, 0.01), strength); return { ...a, score: a.score * damped }; }) .sort((a, b) => b.score - a.score); } /** Last.fm scrobble-neighbour similarity. Returns up to 100 with match scores. */ async function lastFmSimilar(name: string): Promise { const d = await json( `/fm?method=artist.getsimilar&limit=100&autocorrect=1&artist=${encodeURIComponent(name)}`, ); const rows: any[] = d?.similarartists?.artist ?? []; return rows.map((a) => ({ name: a.name, mbid: a.mbid || undefined, sources: ['lastfm'], score: 0, rawScore: Number(a.match) || 0, })); } /** Loose artist-name key used for matching across catalogues. */ export const normalizeName = (s: string) => s.toLowerCase().replace(/^the\s+/, '').replace(/[^a-z0-9]/g, ''); const norm = normalizeName; /** * Reciprocal rank fusion. The conventional k=60 comes from web search, where * result lists run to thousands; against a 13-item list it flattens everything * (rank 1 scores 1/61, rank 20 scores 1/80 — near-identical), which erases any * preference for close neighbours. k=8 keeps the head meaningfully ahead. */ export function fuse(lists: Artist[][], weights: number[], k = 8): Artist[] { const merged = new Map(); lists.forEach((list, li) => { list.forEach((artist, rank) => { const key = norm(artist.name); const existing = merged.get(key); const contribution = (weights[li] ?? 1) / (k + rank + 1); if (existing) { existing.score += contribution; // Every source in `artist.sources`, not just the first: a raw // per-source list only ever carries one, but a list fused a level up // (several seeds fusing their own already-fused rosters, see // fanoutMulti()) carries several, and dropping all but the first // would silently erase which of them actually voted. for (const src of artist.sources) if (!existing.sources.includes(src)) existing.sources.push(src); existing.deezerId ??= artist.deezerId; existing.mbid ??= artist.mbid; existing.fans ??= artist.fans; } else { merged.set(key, { ...artist, sources: [...artist.sources], score: contribution }); } }); }); return [...merged.values()].sort((a, b) => b.score - a.score); } export type FanoutResult = { seed: Artist; similar: Artist[]; notes: string[] }; /** * Second-hop expansion: ask the closest first-hop neighbours who *their* * neighbours are. One hop only ever surfaces artists directly adjacent to the * seed, which caps how much of a hand-built playlist can be reached; the * scene-mates that make a playlist feel curated usually sit two hops out. * Contributions are discounted so first-hop artists still rank ahead. */ export async function expandSecondHop( first: Artist[], hopCount: number, discount: number, /** How fast a head's contribution decays with its own rank. */ rankDecay = 0.15, ): Promise<{ lists: Artist[][]; weights: number[]; queried: number }> { const heads = first.slice(0, hopCount); const results = await pool(heads, 5, (a) => lastFmSimilar(a.name)); const lists: Artist[][] = []; const weights: number[] = []; results.forEach((list, i) => { if (!list?.length) return; // Nearer heads contribute more, decaying with their own rank. lists.push(list.map((a) => ({ ...a, sources: ['hop2'] }))); weights.push(discount / (1 + i * rankDecay)); }); return { lists, weights, queried: lists.length }; } /** * Resolves a seed Deezer has no exact match for (findDeezerArtist() either * found nothing or fell back to a fuzzy result under a different name), * through Last.fm's autocorrect first and MusicBrainz second. A misspelled * seed ("radiohed") is exactly this case too, so when Last.fm corrects the * name, Deezer is asked again under the corrected one before giving up on it: * a Deezer-covered artist keeps Deezer related and full track enrichment. * Otherwise the returned artist carries a name and possibly an mbid but no * deezerId — deezerRelated() below needs one and gets none, so a seed resolved * this way leans on Last.fm similar (and ListenBrainz, when an mbid turned up * and its weight is above its default of 0) rather than Deezer related. */ async function resolveNonDeezerSeed(seedName: string, notes: string[]): Promise { const info = await json( `/fm?method=artist.getinfo&autocorrect=1&artist=${encodeURIComponent(seedName)}`, ).catch(() => null); const a = info?.artist; if (a?.name) { if (norm(a.name) !== norm(seedName)) { const corrected = await findDeezerArtist(a.name).catch(() => null); if (corrected && norm(corrected.name) === norm(a.name)) return corrected; } notes.push(`"${seedName}" not found on Deezer — resolved through Last.fm.`); return { name: a.name, mbid: a.mbid || undefined, sources: [], score: 0 }; } const mbid = await findMbid(seedName).catch(() => null); if (!mbid) throw new Error(`No seed artist matched "${seedName}"`); notes.push(`"${seedName}" not found on Deezer — resolved through MusicBrainz.`); return { name: seedName, mbid, sources: [], score: 0 }; } /** Queries every available source for one seed artist and fuses the results. */ export async function fanout( seedName: string, weights: { deezer: number; listenbrainz: number; lastfm: number; secondHop: number }, /** How many first-hop heads to query for second-hop neighbours. */ hopCount = 15, /** Rank decay passed through to expandSecondHop. */ secondHopDecay = 0.15, ): Promise { const notes: string[] = []; const deezerSeed = await findDeezerArtist(seedName); // findDeezerArtist() falls back to its top search hit when nothing matches // exactly — fine for a candidate, which topTracks() re-checks before ever // touching it, but the seed had no such check, so a search on "Radioclit" // silently anchored the whole fan-out on an unrelated novelty act. const seed = deezerSeed && norm(deezerSeed.name) === norm(seedName) ? deezerSeed : await resolveNonDeezerSeed(seedName, notes); const wantListenBrainz = weights.listenbrainz > 0; let mbid = seed.mbid ?? null; if (wantListenBrainz && !mbid) mbid = await findMbid(seed.name).catch(() => null); if (wantListenBrainz && !mbid) notes.push('No MusicBrainz match — ListenBrainz sat this one out.'); const [dz, lb, fm] = await Promise.all([ seed.deezerId ? deezerRelated(seed.deezerId).catch((e) => { notes.push(`Deezer related failed: ${e.message}`); return [] as Artist[]; }) : Promise.resolve([] as Artist[]), mbid ? listenBrainzSimilar(mbid).catch((e) => { notes.push(`ListenBrainz failed: ${e.message}`); return [] as Artist[]; }) : Promise.resolve([] as Artist[]), lastFmSimilar(seed.name).catch((e) => { notes.push(`Last.fm failed: ${e.message}`); return [] as Artist[]; }), ]); const firstHop = fuse([dz, lb, fm], [weights.deezer, weights.listenbrainz, weights.lastfm]); let similar = firstHop; if (weights.secondHop > 0 && firstHop.length) { const { lists, weights: hopWeights, queried } = await expandSecondHop( firstHop, hopCount, weights.secondHop, secondHopDecay, ); similar = fuse( [dz, lb, fm, ...lists], [weights.deezer, weights.listenbrainz, weights.lastfm, ...hopWeights], ); notes.push( `Deezer ${dz.length}, ListenBrainz ${lb.length}, Last.fm ${fm.length}; ` + `${queried} second-hop queries grew ${firstHop.length} candidates to ${similar.length}.`, ); } else { notes.push(`Deezer ${dz.length}, ListenBrainz ${lb.length}, Last.fm ${fm.length}.`); } return { seed, similar, notes }; } export type MultiFanoutResult = { seeds: Artist[]; similar: Artist[]; notes: string[] }; /** * Several seeds fused into one candidate ranking, rather than drawn from in * proportion. This is the same question fuse() already answers for the three * similarity sources, one level up: ranks fuse, scores do not, so each seed's * own fused `similar` list counts as one more list into fuse(), weighted * equally. Fusion rather than an apportioned draw is the deliberate choice — * an artist similar to every seed should outrank one similar to only one, so * the result reads as a blend of the seeds rather than a shuffle of separate * single-seed playlists. * * Each seed still runs the full single-seed fanout() — three sources plus its * own second-hop expansion — so several seeds multiply the request count by * roughly the seed count. Seeds run two at a time rather than all at once for * the same reason expandSecondHop() bounds its own concurrency: Deezer starts * answering 403 after a few hundred requests in a session, and MusicBrainz * throttles one request per second per IP. A seed that fails to resolve is * dropped and noted rather than aborting the whole draw, matching pool()'s * tolerant-failure behaviour elsewhere in this file. * * A single seed passed through here is equivalent to calling fanout() alone: * fusing one list re-derives scores from rank only, which preserves the same * order fanout() already produced. */ export async function fanoutMulti( seedNames: string[], weights: { deezer: number; listenbrainz: number; lastfm: number; secondHop: number }, hopCount = 15, secondHopDecay = 0.15, ): Promise { const results = await pool(seedNames, 2, (name) => fanout(name, weights, hopCount, secondHopDecay)); const seeds: Artist[] = []; const lists: Artist[][] = []; const notes: string[] = []; results.forEach((r, i) => { if (!r) { notes.push(`"${seedNames[i]}" did not resolve and was dropped from the seed set.`); return; } seeds.push(r.seed); lists.push(r.similar); notes.push(...r.notes.map((n) => `${r.seed.name}: ${n}`)); }); if (!seeds.length) throw new Error(`No seed artist matched: ${seedNames.join(', ')}`); const seedKeys = new Set(seeds.map((s) => norm(s.name))); // Fused out from the candidate roster: a seed already gets its own tracks // pulled directly, so it appearing again as a "similar" candidate — which // happens whenever one seed sits in another's neighbourhood — would only // waste a slot in the pool on an artist already anchoring the playlist. const similar = fuse(lists, seeds.map(() => 1)).filter((a) => !seedKeys.has(norm(a.name))); return { seeds, similar, notes }; } /** * Deterministic negative integer from a string key, for an artist or track * Deezer carries no id for at all. A real Deezer id is always positive, so * this id space never collides with one. Deterministic hashing — rather than * a random or incrementing counter — means the same Last.fm-only artist or * track gets the same id on a later fetch, which addPlaylist()'s * duplicate-playlist check in src/playlists.ts (compares track id lists) and * a saved playlist's own stored track ids both rely on. */ function syntheticId(key: string): number { let hash = 2166136261; for (let i = 0; i < key.length; i++) { hash ^= key.charCodeAt(i); hash = Math.imul(hash, 16777619); } return -Math.abs(hash) - 1; } /** * Last.fm's own top-tracks list, for an artist Deezer has no exact match for. * Several Deezer-only fields have no equivalent here: * - `id` is synthetic (see syntheticId() above), since Last.fm carries no * numeric track id of its own. enrichTracks() skips any negative id rather * than firing a /dz/track request that would 404 for it. * - `rank` is left at 0, a sentinel selectTracks() reads as "no popularity * data" — Last.fm's own list position sits on a scale unrelated to * Deezer's `rank` field, the same reasoning fuse() already applies to * never comparing raw scores across sources directly. * - `duration` is 0 when Last.fm has none logged for the track, which * selectTracks() also reads as "no data" rather than a real zero-length * track. * - `album`, `preview`, `bpm`, `gain` and `year` stay unset and `explicit` * defaults false — none of these have a Last.fm equivalent. Playback only * needs artist and title, which this does carry. */ async function lastFmTopTracks(artist: Artist, limit: number): Promise { const d = await json( `/fm?method=artist.gettoptracks&autocorrect=1&artist=${encodeURIComponent(artist.name)}&limit=${limit}`, ).catch(() => null); const rows: any[] = d?.toptracks?.track ?? []; const artistId = syntheticId(`artist:${norm(artist.name)}`); return rows.map((t, i) => ({ id: syntheticId(`track:${norm(artist.name)}:${norm(t.name ?? '')}`), title: t.name, artist: t.artist?.name ?? artist.name, artistId, album: '', duration: Math.round((Number(t.duration) || 0) / 1000), rank: 0, explicit: false, preview: null, link: t.url ?? `https://www.last.fm/music/${encodeURIComponent(artist.name)}`, depth: rows.length > 1 ? i / (rows.length - 1) : 0, sources: artist.sources, })); } /** * Top tracks for an artist, resolving the Deezer id by name when missing, and * falling back to Last.fm's own top-tracks list when Deezer has no exact * match at all rather than dropping the artist from the pool. */ export async function topTracks(artist: Artist, limit: number): Promise { let id = artist.deezerId; if (!id) { const found = await findDeezerArtist(artist.name); if (found && norm(found.name) === norm(artist.name)) { id = found.deezerId; // Written back so the popularity correction can see the audience size of // artists that only ListenBrainz nominated. artist.deezerId = found.deezerId; artist.fans = found.fans; } } if (!id) return lastFmTopTracks(artist, limit); const d = await json(`/dz/artist/${id}/top?limit=${limit}`); const rows: any[] = d.data ?? []; return rows.map((t, i) => ({ id: t.id, title: t.title, artist: t.artist?.name ?? artist.name, artistId: id!, album: t.album?.title ?? '', duration: t.duration, rank: t.rank, explicit: !!t.explicit_lyrics, preview: t.preview || null, link: t.link ?? `https://www.deezer.com/track/${t.id}`, depth: rows.length > 1 ? i / (rows.length - 1) : 0, sources: artist.sources, })); } /** * Fills in bpm, gain, and release year, which only the single-track endpoint * carries. bpm comes back 0 for roughly half the catalog. */ export async function enrichTracks( tracks: Track[], /** Fires every 25 tracks plus a final call, carrying a full snapshot. Throttled because each call drives an expensive Map rebuild. */ onBatch?: (partial: Track[], done: number) => void, /** Fires once per track, carrying that track and the running count. Not throttled, since it only drives a status string rather than state. */ onTrack?: (track: Track, done: number) => void, /** * Cancels the pass. A superseded pass that only stops *reporting* still pays * for every request it started, which on a reload is a couple of hundred of * them for a playlist nobody will see. */ signal?: AbortSignal, ) { const out = [...tracks]; let done = 0; // Deezer tolerates roughly 10 requests a second; the retry in json() absorbs // the occasional overshoot. await pool( tracks, 10, async (t, i) => { // A Last.fm-only track (see lastFmTopTracks() above) carries a // synthetic negative id and has no /dz/track entry to fill in from. if (t.id < 0) { done += 1; onTrack?.(out[i], done); if (done % 25 === 0) onBatch?.([...out], done); return; } const d = await json(`/dz/track/${t.id}`, 0, signal); out[i] = { ...t, bpm: d.bpm > 0 ? d.bpm : undefined, gain: typeof d.gain === 'number' ? d.gain : undefined, year: d.release_date ? Number(String(d.release_date).slice(0, 4)) : undefined, }; done += 1; onTrack?.(out[i], done); // Publish partial results so filters sharpen while the rest still loads. if (done % 25 === 0) onBatch?.([...out], done); }, signal, ); onBatch?.([...out], done); return out; } export type ArtistMeta = { country?: string; type?: string; began?: number; ended?: boolean }; /** MusicBrainz artist metadata. Rate-limited to one request per second. */ export async function artistMeta(mbid: string): Promise { const a = await json(`${MB}/artist/${mbid}?fmt=json&inc=genres`); return { country: a.country ?? undefined, type: a.type ?? undefined, began: a['life-span']?.begin ? Number(String(a['life-span'].begin).slice(0, 4)) : undefined, ended: a['life-span']?.ended ?? undefined, }; }