From 889300c9d6e2e253cdd025e42c857ced468ba106 Mon Sep 17 00:00:00 2001 From: Lucas Santana Date: Sun, 7 Jun 2026 19:30:51 -0300 Subject: [PATCH 1/4] feat(bot): add last.fm seed-similarity spine to autoplay (approach a) Autoplay drifts to mainstream/unrelated music (e.g. Prince -> Drake) for the common unlinked-user case: the Last.fm collector is user-link-gated, Spotify recommendations are deprecated for this app, and the genre guard fails open when tags are absent, leaving the pool to genre-top-tracks and a genre-blind spotify-preferred boost. Approach A grounds autoplay on Last.fm track.getSimilar for the currently-playing seed track, independent of user linking: - new collectSeedSimilarCandidates() runs as a primary grounding source in the replenisher whenever a requester is known (not link-gated) - 2s timeout + ~1h per-seed cache + cap of 10 lookups bound Last.fm calls and replenish latency; empty/timeout falls through so the queue never stalls - match weighted on the correct 0..1 scale with a 0.5x floor so a thin pool stays grounded - SEED_SIMILAR kept as a distinct RecommendationSource (schema enum + migration) so per-source acceptance telemetry can measure the spine Approach B (fail-closed genre guard + genre-conditional spotify boost) follows. ADR decisions/2026-06-07-autoplay-seed-similarity-spine.md. --- ...26-06-07-autoplay-seed-similarity-spine.md | 93 ++++++++ .../music/autoplay/recommendationBasis.ts | 110 ++++----- .../autoplay/recommendationSourceMapping.ts | 42 ++-- .../src/utils/music/autoplay/replenisher.ts | 25 ++ .../autoplay/seedSimilarityCollector.spec.ts | 214 ++++++++++++++++++ .../music/autoplay/seedSimilarityCollector.ts | 173 ++++++++++++++ .../migration.sql | 11 + prisma/schema.prisma | 1 + 8 files changed, 595 insertions(+), 74 deletions(-) create mode 100644 decisions/2026-06-07-autoplay-seed-similarity-spine.md create mode 100644 packages/bot/src/utils/music/autoplay/seedSimilarityCollector.spec.ts create mode 100644 packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts create mode 100644 prisma/migrations/20260607000000_add_seed_similar_recommendation_source/migration.sql diff --git a/decisions/2026-06-07-autoplay-seed-similarity-spine.md b/decisions/2026-06-07-autoplay-seed-similarity-spine.md new file mode 100644 index 000000000..d66276b2e --- /dev/null +++ b/decisions/2026-06-07-autoplay-seed-similarity-spine.md @@ -0,0 +1,93 @@ +# ADR 2026-06-07 — Autoplay: Last.fm seed-similarity spine + fail-closed genre guard + +**Status:** Accepted +**Via:** `/research-and-decide` (critic verdict REVISE: A alone insufficient → A+B; no flip on the lead) +**Relates to:** [2026-05-21-autoplay-recommendation-roadmap](2026-05-21-autoplay-recommendation-roadmap.md) (Phase D scoring, deferred) + +## Context + +Autoplay drifts to mainstream/unrelated music — reported symptom: two Prince tracks in a row +then a jump to Drake. Root cause (verified in code): + +1. **Genre guard fails OPEN.** The scorer's cross-genre-family veto (`candidateScorer.ts:251`) + and soft genre penalty (`:423`) both require Last.fm/Spotify _tags_ for seed AND candidate. + With no Last.fm link and no/quota-exhausted Spotify token, `sessionGenreFamilies` and + `currentTrackTags` are empty → both defenses are skipped entirely. +2. **`spotify preferred +0.4`** (`:411`) is the single largest, genre-blind score term — any + Spotify track (incl. mainstream Drake) gets it and outranks more-related candidates. +3. **Spotify `/recommendations` is deprecated** for this app (the team's own `artistApi.ts:117` + comment; Discover already migrated to Last.fm) → the "primary" recommender returns little. +4. **The broad fallback pulls genre TOP-tracks** (`getTagTopTracks`) — mainstream by construction. +5. **The Last.fm collector is entirely user-link-gated** (`lastFmSeeder.ts:75` returns early with + no candidates for unlinked users). The tag-independent, no-link endpoint + `artist.getSimilar` (`getSimilarArtists`, shared `artistApi.ts:89`) exists but has **zero + callers in autoplay**. + +So for the common case (user hasn't linked Last.fm, Spotify recs dead), the candidate pool is +just Spotify-search-on-seed + genre-top-tracks fallback, with genre guards disabled → drift. + +## Decision + +**Adopt A + B now (A first), C as a fast-follow if drift persists.** A is the missing +foundation; B stops the fail-open + genre-blind boost; the critic confirmed A alone is +insufficient because B is what blocks drift when A's pool is thin. + +**A — Seed-similarity spine (foundation).** Wire Last.fm `artist.getSimilar` (no user link; +`LASTFM_API_KEY` is set) on the **currently-playing seed artist** into the candidate collector +as a backbone source, boosted as "seed-similar". Demote the deprecated Spotify-recs source to a +best-effort extra (leave as a no-op-tolerant call; do not remove yet). + +- Guardrails: wrap in a 2s timeout + per-artist cache (reuse/extend `artistTagCache`, ~1h TTL) + - 429 backoff. Cap to ~10 similar artists, one search each, to bound Last.fm calls (~5 req/s + limit) and replenish latency. + +**B — Fail genre guard CLOSED for strong-genre seeds + condition the Spotify boost.** + +- Infer the seed's family from `sessionGenreFamilies` (session history), **not just the current + track's tags**, so an untagged current track still carries the session's genre. +- When the session has a **strong** family (rap_hiphop / rock_metal / latin — extend to the + Prince case via the soul/funk/pop families as data shows) and a candidate is **untagged**, + penalize/reject as assumed cross-genre — but only outside skip-storms. +- Make `spotify preferred` **genre-conditional**: full boost only when the candidate's family + overlaps the session; **half** boost when tags are unknown (not zero — avoids starving the + pool); no boost on a known cross-family candidate. + +**C — Fast-follow (only if A+B leave residual drift).** Replace the fallback's +`getTagTopTracks` with seed-artist-similar tracks (same `artist.getSimilar`) so even the last +resort stays on-genre. + +**Non-negotiable guardrail across all:** autoplay must never stall. Keep the skip-storm +relaxation; enforce a minimum-pool floor — if A+B reject everything, fall through to the +(now seed-similar, post-C) fallback rather than returning an empty queue. + +## Alternatives considered + +- **A alone** — rejected (critic): fixes the thin-pool case but leaves the genre guard failing + open, so a mainstream Spotify track still outranks seed-similar ones. B is required. +- **Fail genre guard closed everywhere / unconditionally** — rejected: empty-queue/stall risk + (worse than occasional drift). Hence "strong family + untagged + not-skip-storm" scoping and + the half-boost-on-unknown compromise. +- **Just fix Spotify seed resolution** (make all seeds Spotify-native so recs get good input) — + rejected: the recs endpoint is deprecated for this app; better input can't fix a dead endpoint. +- **Jump straight to a Phase-D session-coherence scorer** — deferred: heavier, and A+B should + resolve the reported symptom; measure first (telemetry already ships) before that investment. + +## Consequences + +- **Positive:** autoplay grounds on seed-artist similarity regardless of user linking or Spotify + health → "continuous radio" near the seed; the largest drift vector (genre-blind Spotify boost + on a tagless session) is closed. +- **Negative:** +1–2 Last.fm calls per replenish (mitigated by cache/timeout/cap); a tuning risk + on B (over-filter → thin pool), bounded by skip-storm relaxation + the pool floor + half-boost. +- **Neutral:** the deprecated Spotify-recs source stays as tolerated dead weight until a separate + cleanup; Phase D remains deferred. + +## Revisit when + +- **~2 weeks post-deploy, read the recommendation telemetry** (per-source acceptance / skip + rate): >10% acceptance improvement → done, keep Phase D deferred; 5–10% → ship C, reassess; + **<5% or drift still reported** → escalate to the Phase-D session-coherence scorer. +- **Last.fm error rate >5% or p99 replenish latency regresses** → tighten cache/timeout or make + the seed-similar fetch fully async (pre-warm next seed's similars). +- **Over-filtering surfaces** (empty-queue warnings rise) → relax B's strong-family list / raise + the half-boost / widen skip-storm relaxation. diff --git a/packages/bot/src/utils/music/autoplay/recommendationBasis.ts b/packages/bot/src/utils/music/autoplay/recommendationBasis.ts index 0e5cbfb70..69395e169 100644 --- a/packages/bot/src/utils/music/autoplay/recommendationBasis.ts +++ b/packages/bot/src/utils/music/autoplay/recommendationBasis.ts @@ -4,44 +4,45 @@ */ export type RecommendationSource = - | 'spotify-rec' - | 'spotify-taste' - | 'lastfm-loved' - | 'lastfm-similar' - | 'lastfm-genre-fallback' - | 'artist-fallback' - | 'genre-tag' + | 'spotify-rec' + | 'spotify-taste' + | 'seed-similar' + | 'lastfm-loved' + | 'lastfm-similar' + | 'lastfm-genre-fallback' + | 'artist-fallback' + | 'genre-tag' export type RecommendationSignal = - | 'preferred artist' - | 'favourite artist' - | 'liked artist' - | 'known artist' - | 'liked track' - | 'old dislike' - | 'skipped before' - | 'completed before' - | 'album match' - | 'deep-dive artist' - | 'session novelty' - | 'source variety' - | 'similar title mood' - | 'similar energy' - | 'long track penalty' - | 'deep dive' - | 'long track match' - | 'quick hit match' - | 'restless discovery' - | 'spotify preferred' - | 'genre family drift' - | 'version variant' - | 'low quality upload' - | 'discovery boost' - | 'energy match' + | 'preferred artist' + | 'favourite artist' + | 'liked artist' + | 'known artist' + | 'liked track' + | 'old dislike' + | 'skipped before' + | 'completed before' + | 'album match' + | 'deep-dive artist' + | 'session novelty' + | 'source variety' + | 'similar title mood' + | 'similar energy' + | 'long track penalty' + | 'deep dive' + | 'long track match' + | 'quick hit match' + | 'restless discovery' + | 'spotify preferred' + | 'genre family drift' + | 'version variant' + | 'low quality upload' + | 'discovery boost' + | 'energy match' export interface RecommendationBasis { - source: RecommendationSource - signals: RecommendationSignal[] + source: RecommendationSource + signals: RecommendationSignal[] } /** @@ -55,28 +56,29 @@ export interface RecommendationBasis { * @returns Formatted string suitable for Discord display */ export function serializeBasis(basis: RecommendationBasis): string { - // Map sources to human-readable labels - const sourceLabels: Record = { - 'spotify-rec': 'spotify rec', - 'spotify-taste': 'spotify taste', - 'lastfm-loved': 'last.fm loved', - 'lastfm-similar': 'last.fm similar', - 'lastfm-genre-fallback': 'last.fm genre', - 'artist-fallback': 'artist fallback', - 'genre-tag': 'genre tag', - } + // Map sources to human-readable labels + const sourceLabels: Record = { + 'spotify-rec': 'spotify rec', + 'spotify-taste': 'spotify taste', + 'seed-similar': 'seed similar', + 'lastfm-loved': 'last.fm loved', + 'lastfm-similar': 'last.fm similar', + 'lastfm-genre-fallback': 'last.fm genre', + 'artist-fallback': 'artist fallback', + 'genre-tag': 'genre tag', + } - const sourceLabel = sourceLabels[basis.source] + const sourceLabel = sourceLabels[basis.source] - // Remove duplicate signals while preserving order - const uniqueSignals = Array.from(new Set(basis.signals)) + // Remove duplicate signals while preserving order + const uniqueSignals = Array.from(new Set(basis.signals)) - // Combine source with all unique signals - if (uniqueSignals.length === 0) { - return sourceLabel - } + // Combine source with all unique signals + if (uniqueSignals.length === 0) { + return sourceLabel + } - // Join signals with " • " separator - const signalsFormatted = uniqueSignals.join(' • ') - return `${sourceLabel} • ${signalsFormatted}` + // Join signals with " • " separator + const signalsFormatted = uniqueSignals.join(' • ') + return `${sourceLabel} • ${signalsFormatted}` } diff --git a/packages/bot/src/utils/music/autoplay/recommendationSourceMapping.ts b/packages/bot/src/utils/music/autoplay/recommendationSourceMapping.ts index b32c459a5..0b01dcbbd 100644 --- a/packages/bot/src/utils/music/autoplay/recommendationSourceMapping.ts +++ b/packages/bot/src/utils/music/autoplay/recommendationSourceMapping.ts @@ -13,25 +13,27 @@ import { RecommendationSource } from './recommendationBasis' * @throws Compile-time error if a union member is not handled (exhaustiveness check) */ export function recommendationSourceToPrisma( - src: RecommendationSource + src: RecommendationSource, ): PrismaRecommendationSource { - switch (src) { - case 'spotify-rec': - return PrismaRecommendationSource.SPOTIFY_REC - case 'spotify-taste': - return PrismaRecommendationSource.SPOTIFY_TASTE - case 'lastfm-loved': - return PrismaRecommendationSource.LASTFM_LOVED - case 'lastfm-similar': - return PrismaRecommendationSource.LASTFM_SIMILAR - case 'lastfm-genre-fallback': - return PrismaRecommendationSource.LASTFM_GENRE_FALLBACK - case 'artist-fallback': - return PrismaRecommendationSource.ARTIST_FALLBACK - case 'genre-tag': - return PrismaRecommendationSource.GENRE_TAG - default: - const _exhaustive: never = src - return _exhaustive - } + switch (src) { + case 'spotify-rec': + return PrismaRecommendationSource.SPOTIFY_REC + case 'spotify-taste': + return PrismaRecommendationSource.SPOTIFY_TASTE + case 'seed-similar': + return PrismaRecommendationSource.SEED_SIMILAR + case 'lastfm-loved': + return PrismaRecommendationSource.LASTFM_LOVED + case 'lastfm-similar': + return PrismaRecommendationSource.LASTFM_SIMILAR + case 'lastfm-genre-fallback': + return PrismaRecommendationSource.LASTFM_GENRE_FALLBACK + case 'artist-fallback': + return PrismaRecommendationSource.ARTIST_FALLBACK + case 'genre-tag': + return PrismaRecommendationSource.GENRE_TAG + default: + const _exhaustive: never = src + return _exhaustive + } } diff --git a/packages/bot/src/utils/music/autoplay/replenisher.ts b/packages/bot/src/utils/music/autoplay/replenisher.ts index 1298a6c39..2107d741f 100644 --- a/packages/bot/src/utils/music/autoplay/replenisher.ts +++ b/packages/bot/src/utils/music/autoplay/replenisher.ts @@ -33,6 +33,7 @@ import { purgeDuplicatesOfCurrentTrack, } from './diversitySelector' import { collectLastFmCandidates } from './lastFmSeeder' +import { collectSeedSimilarCandidates } from './seedSimilarityCollector' import { serializeBasis } from './recommendationBasis' import { cleanAuthor } from '../searchQueryCleaner' import type { QueueMetadata } from '../../../types/QueueMetadata' @@ -97,6 +98,7 @@ async function _replenishQueue( let candidatePoolSize = 0 const sourcesCounts = { recommendation: 0, + seedSimilar: 0, lastfm: 0, fallback: 0, genre: 0, @@ -331,6 +333,29 @@ async function _replenishQueue( data: { guildId, count: candidates.size, source: 'recommendation' }, }) + // Seed-similarity spine: grounds autoplay on the current track's Last.fm + // similars regardless of whether the user linked Last.fm. Runs before the + // user-linked Last.fm collector so the pool is anchored to the seed even + // for unlinked sessions (the common case the drift fix targets). + if (requestedBy) { + const beforeSeedSimilar = candidates.size + await collectSeedSimilarCandidates( + autoplayContext, + requestedBy, + candidates, + ) + sourcesCounts.seedSimilar = candidates.size - beforeSeedSimilar + debugLog({ + message: 'Autoplay: seed-similar candidates', + data: { + guildId, + added: candidates.size - beforeSeedSimilar, + total: candidates.size, + source: 'seed-similar', + }, + }) + } + if (requestedBy?.id) { const beforeLastFm = candidates.size await collectLastFmCandidates( diff --git a/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.spec.ts b/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.spec.ts new file mode 100644 index 000000000..b6a598507 --- /dev/null +++ b/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.spec.ts @@ -0,0 +1,214 @@ +import { beforeEach, describe, expect, it, jest } from '@jest/globals' +import type { Track, GuildQueue } from 'discord-player' +import type { AutoplayContext } from './autoplayContext' + +const getSimilarTracksMock = jest.fn() +const createArtistTagFetcherMock = jest.fn() +const cleanSearchQueryMock = jest.fn() +const cleanTitleMock = jest.fn() +const calculateRecommendationScoreMock = jest.fn() +const normalizeTrackKeyMock = jest.fn() +const shouldIncludeCandidateMock = jest.fn() +const upsertScoredCandidateMock = jest.fn() +const searchLastFmQueryMock = jest.fn() + +jest.mock('@lucky/shared/utils', () => ({ + debugLog: jest.fn(), +})) + +jest.mock('../../../lastfm', () => ({ + getSimilarTracks: (...args: unknown[]) => getSimilarTracksMock(...args), +})) + +jest.mock('./artistTagCache', () => ({ + createArtistTagFetcher: (...args: unknown[]) => + createArtistTagFetcherMock(...args), +})) + +jest.mock('../searchQueryCleaner', () => ({ + cleanSearchQuery: (...args: unknown[]) => cleanSearchQueryMock(...args), + cleanTitle: (...args: unknown[]) => cleanTitleMock(...args), +})) + +jest.mock('./candidateScorer', () => ({ + calculateRecommendationScore: (...args: unknown[]) => + calculateRecommendationScoreMock(...args), +})) + +jest.mock('./scoringUtils', () => ({ + normalizeTrackKey: (...args: unknown[]) => normalizeTrackKeyMock(...args), +})) + +jest.mock('./candidateCollector', () => ({ + shouldIncludeCandidate: (...args: unknown[]) => + shouldIncludeCandidateMock(...args), + upsertScoredCandidate: (...args: unknown[]) => + upsertScoredCandidateMock(...args), +})) + +jest.mock('./lastFmSeeder', () => ({ + searchLastFmQuery: (...args: unknown[]) => searchLastFmQueryMock(...args), +})) + +import { collectSeedSimilarCandidates } from './seedSimilarityCollector' + +function createTrack( + title = 'Track', + author = 'Artist', + url = 'https://spotify.com/t', +): Track { + return { + title, + author, + url, + durationMS: 200_000, + requestedBy: null, + } as unknown as Track +} + +function createUser(id = 'user-1') { + return { id } as never +} + +function createAutoplayContext( + overrides: Partial = {}, +): AutoplayContext { + return { + queue: { guild: { id: 'guild-1' } } as unknown as GuildQueue, + excludedUrls: new Set(), + excludedKeys: new Set(), + dislikedWeights: new Map(), + likedWeights: new Map(), + preferredArtistKeys: new Set(), + blockedArtistKeys: new Set(), + currentTrack: createTrack(), + recentArtists: new Set(), + autoplayMode: 'similar', + artistFrequency: new Map(), + implicitDislikeKeys: new Set(), + implicitLikeKeys: new Set(), + sessionMood: null, + genreContext: {}, + ...overrides, + } +} + +describe('collectSeedSimilarCandidates', () => { + beforeEach(() => { + jest.clearAllMocks() + cleanSearchQueryMock.mockImplementation( + (t: unknown, a: unknown) => `${t} ${a}`, + ) + cleanTitleMock.mockImplementation((s: unknown) => s) + normalizeTrackKeyMock.mockReturnValue('normalized-key') + shouldIncludeCandidateMock.mockReturnValue(true) + calculateRecommendationScoreMock.mockReturnValue({ + score: 0.5, + signals: [], + }) + createArtistTagFetcherMock.mockReturnValue( + jest.fn().mockResolvedValue([]), + ) + searchLastFmQueryMock.mockResolvedValue([]) + }) + + it('returns early without similars and adds nothing', async () => { + getSimilarTracksMock.mockResolvedValue([]) + const ctx = createAutoplayContext({ + currentTrack: createTrack('Empty', 'NoSimilars'), + }) + const candidates = new Map() + await collectSeedSimilarCandidates(ctx, createUser(), candidates) + expect(searchLastFmQueryMock).not.toHaveBeenCalled() + expect(upsertScoredCandidateMock).not.toHaveBeenCalled() + }) + + it('returns early when the seed track has no artist', async () => { + const ctx = createAutoplayContext({ + currentTrack: createTrack('Title', ''), + }) + const candidates = new Map() + await collectSeedSimilarCandidates(ctx, createUser(), candidates) + expect(getSimilarTracksMock).not.toHaveBeenCalled() + }) + + it('grounds on the CURRENT track (not a user seed slice) via track.getSimilar', async () => { + getSimilarTracksMock.mockResolvedValue([ + { title: 'Sim', artist: 'SimA', match: 1 }, + ]) + searchLastFmQueryMock.mockResolvedValue([createTrack('Sim', 'SimA')]) + const ctx = createAutoplayContext({ + currentTrack: createTrack('Purple Rain', 'Prince'), + }) + const candidates = new Map() + await collectSeedSimilarCandidates(ctx, createUser(), candidates) + + expect(getSimilarTracksMock).toHaveBeenCalledWith( + 'Prince', + 'Purple Rain', + expect.any(Number), + ) + const call = upsertScoredCandidateMock.mock.calls[0] + expect((call?.[2] as { source: string })?.source).toBe('seed-similar') + }) + + it('weights a perfect match (1.0) at full strength', async () => { + getSimilarTracksMock.mockResolvedValue([ + { title: 'Sim', artist: 'SimA', match: 1 }, + ]) + searchLastFmQueryMock.mockResolvedValue([createTrack('Sim', 'SimA')]) + const ctx = createAutoplayContext({ + currentTrack: createTrack('FullMatch', 'ArtistFM'), + }) + const candidates = new Map() + await collectSeedSimilarCandidates(ctx, createUser(), candidates) + const score = ( + upsertScoredCandidateMock.mock.calls[0]?.[2] as { score: number } + )?.score + // (rec 0.5 + SEED_SIMILAR_BOOST 0.25) * (0.5 + 0.5*1) = 0.75 + expect(score).toBeCloseTo(0.75, 5) + }) + + it('keeps a weak match (0.0) competitive at the 0.5x floor', async () => { + getSimilarTracksMock.mockResolvedValue([ + { title: 'Sim', artist: 'SimA', match: 0 }, + ]) + searchLastFmQueryMock.mockResolvedValue([createTrack('Sim', 'SimA')]) + const ctx = createAutoplayContext({ + currentTrack: createTrack('WeakMatch', 'ArtistWM'), + }) + const candidates = new Map() + await collectSeedSimilarCandidates(ctx, createUser(), candidates) + const score = ( + upsertScoredCandidateMock.mock.calls[0]?.[2] as { score: number } + )?.score + // (0.5 + 0.25) * (0.5 + 0.5*0) = 0.375 — never crushed to ~0 + expect(score).toBeCloseTo(0.375, 5) + }) + + it('skips disliked candidates (weight > 0.5)', async () => { + getSimilarTracksMock.mockResolvedValue([ + { title: 'Sim', artist: 'SimA', match: 0.9 }, + ]) + searchLastFmQueryMock.mockResolvedValue([createTrack('Sim', 'SimA')]) + const ctx = createAutoplayContext({ + currentTrack: createTrack('Disliked', 'ArtistD'), + dislikedWeights: new Map([['normalized-key', 0.9]]), + }) + const candidates = new Map() + await collectSeedSimilarCandidates(ctx, createUser(), candidates) + expect(upsertScoredCandidateMock).not.toHaveBeenCalled() + }) + + it('does not throw when the similar fetch rejects (never stalls)', async () => { + getSimilarTracksMock.mockRejectedValue(new Error('last.fm down')) + const ctx = createAutoplayContext({ + currentTrack: createTrack('Rejects', 'ArtistR'), + }) + const candidates = new Map() + await expect( + collectSeedSimilarCandidates(ctx, createUser(), candidates), + ).resolves.toBeUndefined() + expect(upsertScoredCandidateMock).not.toHaveBeenCalled() + }) +}) diff --git a/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts b/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts new file mode 100644 index 000000000..f2228dff0 --- /dev/null +++ b/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts @@ -0,0 +1,173 @@ +import type { User } from 'discord.js' +import { debugLog } from '@lucky/shared/utils' +import { getSimilarTracks } from '../../../lastfm' +import { createArtistTagFetcher } from './artistTagCache' +import { cleanSearchQuery, cleanTitle } from '../searchQueryCleaner' +import type { AutoplayContext } from './autoplayContext' +import { calculateRecommendationScore } from './candidateScorer' +import { normalizeTrackKey } from './scoringUtils' +import { + shouldIncludeCandidate, + upsertScoredCandidate, +} from './candidateCollector' +import type { ScoredTrack } from './diversitySelector' +import type { AutoplayAuditCollector } from './autoplayAudit' +import { searchLastFmQuery } from './lastFmSeeder' + +// Boost applied to every seed-similar candidate. Sits above the lastfm-similar +// boost (0.2) so the seed-grounded spine outranks weaker collateral sources +// when scores are otherwise close. +const SEED_SIMILAR_BOOST = 0.25 +// Cap the number of similar tracks we look up + search per pass to bound +// Last.fm calls (~5 req/s public limit) and replenish latency. +const MAX_SEED_SIMILAR = 10 +// Hard ceiling on the Last.fm similar fetch so a slow/hanging request never +// stalls the replenish pass; on timeout we fall through to the other sources. +const SEED_SIMILAR_TIMEOUT_MS = 2000 +const AUTOPLAY_BUFFER_SIZE = 8 +const SIMILAR_CACHE_TTL_MS = 60 * 60 * 1000 +const SIMILAR_CACHE_MAX = 200 + +type SimilarTrack = { artist: string; title: string; match: number } + +// Per-seed-track cache of Last.fm similar lookups. Keyed by artist::title so a +// repeated seed (deep-dive, loop) reuses the result for ~1h instead of hitting +// Last.fm every cycle. Only non-empty results are cached so a transient +// timeout/failure is retried on the next pass. +const similarCache = new Map< + string, + { tracks: SimilarTrack[]; expiresAt: number } +>() + +async function fetchSeedSimilar( + artist: string, + title: string, +): Promise { + const key = `${artist.toLowerCase()}::${title.toLowerCase()}` + const now = Date.now() + const cached = similarCache.get(key) + if (cached && cached.expiresAt > now) return cached.tracks + + let timeoutId: ReturnType | undefined + const timeout = new Promise((resolve) => { + timeoutId = setTimeout(() => resolve([]), SEED_SIMILAR_TIMEOUT_MS) + }) + const tracks = await Promise.race([ + getSimilarTracks(artist, cleanTitle(title), MAX_SEED_SIMILAR), + timeout, + ]) + .catch(() => [] as SimilarTrack[]) + .finally(() => { + if (timeoutId) clearTimeout(timeoutId) + }) + + if (tracks.length > 0) { + if (similarCache.size >= SIMILAR_CACHE_MAX) { + const oldest = similarCache.keys().next().value + if (oldest) similarCache.delete(oldest) + } + similarCache.set(key, { + tracks, + expiresAt: now + SIMILAR_CACHE_TTL_MS, + }) + } + return tracks +} + +/** + * Seed-similarity spine: grounds autoplay on Last.fm `track.getSimilar` for the + * currently-playing track, independent of whether any user has linked Last.fm. + * + * This is the tag-independent backbone source — `collectLastFmCandidates` only + * runs for users with a linked Last.fm account, and the deprecated Spotify + * recommendations endpoint returns little for this app, so without this source + * an unlinked session collapses to genre-top-tracks (mainstream) and drifts. + * + * Guardrails: the fetch is wrapped in a 2s timeout + a ~1h per-seed cache and + * capped to MAX_SEED_SIMILAR lookups; on empty/timeout it simply returns so the + * other collectors and the broad fallback still run (never stalls the queue). + */ +export async function collectSeedSimilarCandidates( + ctx: AutoplayContext, + requestedBy: User, + candidates: Map, + auditCollector?: AutoplayAuditCollector, +): Promise { + const seedArtist = ctx.currentTrack.author?.trim() + const seedTitle = ctx.currentTrack.title?.trim() + if (!seedArtist || !seedTitle) return + + const similar = await fetchSeedSimilar(seedArtist, seedTitle) + if (similar.length === 0) return + + const getArtistTags = + ctx.genreContext.getArtistTags ?? createArtistTagFetcher() + const currentTrackTags = ctx.genreContext.currentTrackTags ?? [] + const sessionGenreFamilies = + ctx.genreContext.sessionGenreFamilies ?? new Set() + + for (const s of similar.slice(0, MAX_SEED_SIMILAR)) { + if (candidates.size >= AUTOPLAY_BUFFER_SIZE) break + const query = cleanSearchQuery(s.title, s.artist) + const tracks = await searchLastFmQuery(ctx.queue, query, requestedBy) + for (const track of tracks) { + if ( + !shouldIncludeCandidate( + track, + ctx.excludedUrls, + ctx.excludedKeys, + ) + ) + continue + const normalizedKey = normalizeTrackKey(track.title, track.author) + const dislikedWeight = ctx.dislikedWeights.get(normalizedKey) + if (dislikedWeight !== undefined && dislikedWeight > 0.5) continue + const tags = await getArtistTags(track.author) + const rec = calculateRecommendationScore({ + candidate: track, + currentTrack: ctx.currentTrack, + recentArtists: ctx.recentArtists, + likedWeights: ctx.likedWeights, + preferredArtistKeys: ctx.preferredArtistKeys, + blockedArtistKeys: ctx.blockedArtistKeys, + autoplayMode: ctx.autoplayMode, + artistFrequency: ctx.artistFrequency, + implicitDislikeKeys: ctx.implicitDislikeKeys, + implicitLikeKeys: ctx.implicitLikeKeys, + dislikedWeights: ctx.dislikedWeights, + sessionMood: ctx.sessionMood, + skipNoveltyBoost: true, + genreContext: { + candidateTags: tags, + currentTrackTags, + sessionGenreFamilies, + }, + }) + // Last.fm returns `match` on a 0..1 scale. Weight by it but keep + // seed-similar competitive — never crush below 0.5× even on a weak + // match, so the spine still grounds a thin pool. + const matchWeight = 0.5 + 0.5 * Math.min(Math.max(s.match, 0), 1) + upsertScoredCandidate( + candidates, + track, + { + score: (rec.score + SEED_SIMILAR_BOOST) * matchWeight, + source: 'seed-similar', + signals: rec.signals, + }, + auditCollector, + ) + } + } + + debugLog({ + message: 'Autoplay: seed-similar candidates collected', + data: { + seedArtist, + seedTitle, + similarCount: similar.length, + total: candidates.size, + source: 'seed-similar', + }, + }) +} diff --git a/prisma/migrations/20260607000000_add_seed_similar_recommendation_source/migration.sql b/prisma/migrations/20260607000000_add_seed_similar_recommendation_source/migration.sql new file mode 100644 index 000000000..f4255260c --- /dev/null +++ b/prisma/migrations/20260607000000_add_seed_similar_recommendation_source/migration.sql @@ -0,0 +1,11 @@ +-- Add the SEED_SIMILAR value to the RecommendationSource enum. +-- +-- Backs the new Last.fm seed-similarity spine collector +-- (packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts). Kept as a +-- distinct value (not folded into LASTFM_SIMILAR) so per-source acceptance +-- telemetry can measure the spine separately — see +-- decisions/2026-06-07-autoplay-seed-similarity-spine.md ("Revisit when"). +-- +-- Positioned before LASTFM_LOVED to mirror the in-code union order in +-- packages/bot/src/utils/music/autoplay/recommendationBasis.ts. +ALTER TYPE "RecommendationSource" ADD VALUE IF NOT EXISTS 'SEED_SIMILAR' BEFORE 'LASTFM_LOVED'; diff --git a/prisma/schema.prisma b/prisma/schema.prisma index 24c7804e6..aedfe8506 100644 --- a/prisma/schema.prisma +++ b/prisma/schema.prisma @@ -461,6 +461,7 @@ model UserArtistPreference { enum RecommendationSource { SPOTIFY_REC SPOTIFY_TASTE + SEED_SIMILAR LASTFM_LOVED LASTFM_SIMILAR LASTFM_GENRE_FALLBACK From b2f8c149f2bb5d7032c59a33272e52bbaee3d679 Mon Sep 17 00:00:00 2001 From: Lucas Santana Date: Sun, 7 Jun 2026 19:35:24 -0300 Subject: [PATCH 2/4] feat(bot): genre-condition autoplay scoring (approach b) Approach B closes the two fail-open holes that let mainstream Spotify tracks (Drake on a Prince session) outrank seed-similar candidates: - spotify-preferred boost is now genre-conditional: full only when the candidate's family overlaps the session, half when the candidate genre is unknown (avoids starving the pool), none on a known cross-family candidate. This was the single largest, previously genre-blind term. - the genre guard fails closed for strong-family sessions (rap_hiphop/rock_metal/latin) when a candidate is untagged: it's demoted as assumed cross-genre via a penalty (not a hard reject) so the queue never stalls, and relaxed during skip storms. Session family is inferred from the history-derived sessionGenreFamilies, not just the current track's tags, so an untagged current track still carries the session's genre. Builds on the seed-similarity spine (approach A). ADR decisions/2026-06-07-autoplay-seed-similarity-spine.md. --- .../music/autoplay/candidateScorer.spec.ts | 108 ++++++++++++++++++ .../utils/music/autoplay/candidateScorer.ts | 63 +++++++++- 2 files changed, 167 insertions(+), 4 deletions(-) diff --git a/packages/bot/src/utils/music/autoplay/candidateScorer.spec.ts b/packages/bot/src/utils/music/autoplay/candidateScorer.spec.ts index a4b1e6844..dcc0c2749 100644 --- a/packages/bot/src/utils/music/autoplay/candidateScorer.spec.ts +++ b/packages/bot/src/utils/music/autoplay/candidateScorer.spec.ts @@ -195,6 +195,114 @@ describe('candidateScorer', () => { }) expect(result.score).toBeGreaterThan(1) }) + + describe('genre-conditional spotify-preferred boost (approach B)', () => { + const session = new Set(['rnb_soul', 'rock_metal']) + + it('gives the full boost when the candidate overlaps the session family', () => { + const result = calculateRecommendationScore({ + candidate: createTrack({ + author: 'In Genre', + source: 'spotify', + }), + currentTrack: createTrack({ author: 'Seed' }), + recentArtists: new Set(), + genreContext: { + candidateTags: ['soul'], + currentTrackTags: ['rock'], + sessionGenreFamilies: session, + }, + }) + expect(result.signals).toContain('spotify preferred') + }) + + it('drops the boost entirely for a known cross-family candidate (Drake on a Prince session)', () => { + const withBoost = calculateRecommendationScore({ + candidate: createTrack({ + author: 'Drake', + source: 'spotify', + }), + currentTrack: createTrack({ author: 'Prince' }), + recentArtists: new Set(), + genreContext: { + candidateTags: ['hip hop'], + currentTrackTags: ['soul'], + sessionGenreFamilies: session, + }, + }) + expect(withBoost.signals).not.toContain('spotify preferred') + }) + + it('halves (not drops) the boost when the candidate genre is unknown', () => { + // Non-strong session so the untagged candidate isn't also hit + // by the strong-family fail-closed guard — isolates the boost. + const softSession = new Set(['rnb_soul', 'pop']) + const onGenre = calculateRecommendationScore({ + candidate: createTrack({ + author: 'In Genre', + source: 'spotify', + }), + currentTrack: createTrack({ author: 'Seed' }), + recentArtists: new Set(), + genreContext: { + candidateTags: ['soul'], + currentTrackTags: ['soul'], + sessionGenreFamilies: softSession, + }, + }) + const unknown = calculateRecommendationScore({ + candidate: createTrack({ + author: 'Unknown Genre', + source: 'spotify', + }), + currentTrack: createTrack({ author: 'Seed' }), + recentArtists: new Set(), + genreContext: { + candidateTags: [], + currentTrackTags: ['soul'], + sessionGenreFamilies: softSession, + }, + }) + // Same author/title, so only the spotify-boost term differs: + // full (0.4) vs half (0.2) → a 0.2 gap, and the signal stays. + expect(unknown.signals).toContain('spotify preferred') + expect(onGenre.score - unknown.score).toBeCloseTo(0.2, 5) + }) + }) + + it('fails closed on an untagged candidate in a strong-family session (approach B)', () => { + const result = calculateRecommendationScore({ + candidate: createTrack({ + author: 'Untagged Mainstream', + source: 'youtube', + }), + currentTrack: createTrack({ author: 'Rapper' }), + recentArtists: new Set(), + genreContext: { + candidateTags: [], + currentTrackTags: ['hip hop'], + sessionGenreFamilies: new Set(['rap_hiphop']), + }, + }) + expect(result.signals).toContain('genre family drift') + }) + + it('does not fail closed for untagged candidates on a non-strong session', () => { + const result = calculateRecommendationScore({ + candidate: createTrack({ + author: 'Untagged', + source: 'youtube', + }), + currentTrack: createTrack({ author: 'Popstar' }), + recentArtists: new Set(), + genreContext: { + candidateTags: [], + currentTrackTags: ['pop'], + sessionGenreFamilies: new Set(['pop']), + }, + }) + expect(result.signals).not.toContain('genre family drift') + }) }) describe('calculateGenreFamilyPenalty', () => { diff --git a/packages/bot/src/utils/music/autoplay/candidateScorer.ts b/packages/bot/src/utils/music/autoplay/candidateScorer.ts index e7a5f3108..8a1effd9b 100644 --- a/packages/bot/src/utils/music/autoplay/candidateScorer.ts +++ b/packages/bot/src/utils/music/autoplay/candidateScorer.ts @@ -42,6 +42,13 @@ const GENRE_PENALTY_STRONG = -0.6 const GENRE_PENALTY_WEAK = -0.3 const GENRE_PENALTY_UNKNOWN = -0.1 const SCORE_SPOTIFY_PREFERRED = 0.4 +// On an unknown-genre candidate the spotify-preferred boost is halved rather +// than dropped, so the pool isn't starved when tags are missing. +const SPOTIFY_PREFERRED_UNKNOWN_MULTIPLIER = 0.5 +// Genre families dense/cohesive enough that a cross-family jump reads as drift. +// Used both for the cross-family penalty and the untagged-candidate fail-closed +// guard. Kept deliberately narrow (pop/soul are too broad to fail closed on). +const STRONG_GENRE_FAMILIES = ['rap_hiphop', 'rock_metal', 'latin'] const DISLIKE_WEIGHT_THRESHOLD = 0.5 const POPULAR_TRACK_THRESHOLD = 50 @@ -145,9 +152,8 @@ export function calculateGenreFamilyPenalty( } } - const strongGenres = ['rap_hiphop', 'rock_metal', 'latin'] const isStrongGenre = Array.from(currentFamilies).some((f) => - strongGenres.includes(f), + STRONG_GENRE_FAMILIES.includes(f), ) return isStrongGenre ? GENRE_PENALTY_STRONG : GENRE_PENALTY_WEAK @@ -200,6 +206,17 @@ export function calculateRecommendationScore(ctx: ScoringContext): { const currentTrackTags = genreContext.currentTrackTags ?? [] const sessionGenreFamilies = genreContext.sessionGenreFamilies ?? new Set() + // Reference families for the session: prefer the history-derived session + // families (present even when the current track is untagged), else fall + // back to the current track's own tags. Drives both the genre-conditional + // spotify boost and the untagged fail-closed guard below. + const referenceFamilies = + sessionGenreFamilies.size > 0 + ? sessionGenreFamilies + : getGenreFamilies(currentTrackTags) + const sessionHasStrongFamily = Array.from(referenceFamilies).some((f) => + STRONG_GENRE_FAMILIES.includes(f), + ) const currentArtist = currentTrack.author.toLowerCase() const candidateArtist = candidate.author.toLowerCase() const candidateArtistKey = normalizeText(cleanAuthor(candidate.author)) @@ -268,6 +285,18 @@ export function calculateRecommendationScore(ctx: ScoringContext): { } } + // Fail-closed for strong-family sessions when the candidate is UNTAGGED. + // The tagged cross-genre veto above can't judge a candidate with no Last.fm + // tags, so on a rap/rock/latin session an unknown mainstream track would + // otherwise slip through (fail-open — the core of the Prince→Drake drift). + // Treat untagged candidates as assumed cross-genre and demote them, but + // keep them in the pool (penalty, not hard reject) so the queue never + // stalls; relaxed during skip storms so the pool can broaden. + if (!inSkipStorm && candidateTags.length === 0 && sessionHasStrongFamily) { + score += GENRE_PENALTY_STRONG + signals.push('genre family drift') + } + if (preferredArtistKeys.has(candidateArtistKey)) { score += SCORE_PREFERRED_ARTIST signals.push('preferred artist') @@ -409,8 +438,34 @@ export function calculateRecommendationScore(ctx: ScoringContext): { } if (candidate.source === 'spotify') { - score += SCORE_SPOTIFY_PREFERRED - signals.push('spotify preferred') + // Genre-condition the spotify-preferred boost — the single largest, + // previously genre-blind term that let mainstream Spotify tracks (e.g. + // Drake on a Prince session) outrank seed-similar candidates. Full + // boost only when the candidate overlaps the session's families; half + // when its genre is unknown (don't starve the pool); none on a known + // cross-family candidate. + const candidateFamilies = + candidateTags.length > 0 + ? getGenreFamilies(candidateTags) + : new Set() + let spotifyBoost: number + if (referenceFamilies.size === 0 || candidateFamilies.size === 0) { + spotifyBoost = + SCORE_SPOTIFY_PREFERRED * SPOTIFY_PREFERRED_UNKNOWN_MULTIPLIER + } else { + let overlaps = false + for (const family of candidateFamilies) { + if (referenceFamilies.has(family)) { + overlaps = true + break + } + } + spotifyBoost = overlaps ? SCORE_SPOTIFY_PREFERRED : 0 + } + if (spotifyBoost > 0) { + score += spotifyBoost + signals.push('spotify preferred') + } } // Soft genre-family penalty (formerly inside enrichWithAudioFeatures via From 1cd988378fdf6210ab5df7a4fca20ac22b906100 Mon Sep 17 00:00:00 2001 From: Lucas Santana Date: Sun, 7 Jun 2026 20:25:14 -0300 Subject: [PATCH 3/4] fix(bot): always run seed-similar spine + cover replenisher wiring MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Self-review findings on the autoplay PR: - P2: drop the `candidates.size >= AUTOPLAY_BUFFER_SIZE` early-out in the seed-similar collector. collectRecommendationCandidates runs first and can fill the pool from (possibly drifted) history seeds; the early-out then let the current-track spine add nothing — exactly the compounding-drift case it exists to ground. The loop is still bounded by MAX_SEED_SIMILAR; a larger pool only widens diverse selection downstream. - P2: assert the replenisher actually invokes collectSeedSimilarCandidates (the integration point was untested — the real collector no-ops without LASTFM_API_KEY, so coverage was incidental). Add called/not-called tests gated on a resolvable requester. --- .../utils/music/autoplay/replenisher.spec.ts | 38 +++++++++++++++++++ .../music/autoplay/seedSimilarityCollector.ts | 7 +++- 2 files changed, 43 insertions(+), 2 deletions(-) diff --git a/packages/bot/src/utils/music/autoplay/replenisher.spec.ts b/packages/bot/src/utils/music/autoplay/replenisher.spec.ts index c75df21dd..a9af40481 100644 --- a/packages/bot/src/utils/music/autoplay/replenisher.spec.ts +++ b/packages/bot/src/utils/music/autoplay/replenisher.spec.ts @@ -75,6 +75,10 @@ jest.mock('./lastFmSeeder', () => ({ collectLastFmCandidates: jest.fn(), })) +jest.mock('./seedSimilarityCollector', () => ({ + collectSeedSimilarCandidates: jest.fn(), +})) + function createTrack(overrides: Partial = {}): Track { return { title: 'Test Song', @@ -177,6 +181,11 @@ describe('replenishQueue', () => { const { createArtistTagFetcher } = require('./artistTagCache') createArtistTagFetcher.mockReturnValue(jest.fn().mockResolvedValue([])) + + const { + collectSeedSimilarCandidates, + } = require('./seedSimilarityCollector') + collectSeedSimilarCandidates.mockResolvedValue(undefined) }) it('should be exported and callable', async () => { @@ -365,6 +374,35 @@ describe('replenishQueue', () => { expect.any(Function), ) }) + + it('runs the seed-similarity spine when a requester is known', async () => { + const queue = createGuildQueue({ + currentTrack: createTrack({ + requestedBy: { id: 'user-123' } as import('discord.js').User, + }), + }) + const { + collectSeedSimilarCandidates, + } = require('./seedSimilarityCollector') + + await replenishQueue(queue) + + expect(collectSeedSimilarCandidates).toHaveBeenCalled() + }) + + it('skips the seed-similarity spine when no requester is resolvable', async () => { + const queue = createGuildQueue({ + currentTrack: createTrack({ requestedBy: null }), + metadata: {}, + }) + const { + collectSeedSimilarCandidates, + } = require('./seedSimilarityCollector') + + await replenishQueue(queue) + + expect(collectSeedSimilarCandidates).not.toHaveBeenCalled() + }) }) describe('clearSessionMoodCache', () => { diff --git a/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts b/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts index f2228dff0..91eeebf3f 100644 --- a/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts +++ b/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts @@ -24,7 +24,6 @@ const MAX_SEED_SIMILAR = 10 // Hard ceiling on the Last.fm similar fetch so a slow/hanging request never // stalls the replenish pass; on timeout we fall through to the other sources. const SEED_SIMILAR_TIMEOUT_MS = 2000 -const AUTOPLAY_BUFFER_SIZE = 8 const SIMILAR_CACHE_TTL_MS = 60 * 60 * 1000 const SIMILAR_CACHE_MAX = 200 @@ -106,8 +105,12 @@ export async function collectSeedSimilarCandidates( const sessionGenreFamilies = ctx.genreContext.sessionGenreFamilies ?? new Set() + // No early-out on pool size: this is the seed-similarity SPINE, so the + // current-track anchor must always land — even when collectRecommendation- + // Candidates already filled the pool from (possibly drifted) history seeds. + // The loop is bounded by MAX_SEED_SIMILAR; the larger pool only widens the + // diverse-selection choice downstream. for (const s of similar.slice(0, MAX_SEED_SIMILAR)) { - if (candidates.size >= AUTOPLAY_BUFFER_SIZE) break const query = cleanSearchQuery(s.title, s.artist) const tracks = await searchLastFmQuery(ctx.queue, query, requestedBy) for (const track of tracks) { From fdfc3fae03816cfd71a9c0f3b1632eab37189d18 Mon Sep 17 00:00:00 2001 From: Lucas Santana Date: Sun, 7 Jun 2026 21:07:16 -0300 Subject: [PATCH 4/4] =?UTF-8?q?docs(bot):=20address=20coderabbit=20nits=20?= =?UTF-8?q?=E2=80=94=20ADR=20endpoint,=20cache=20eviction,=20comments?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - ADR: correct artist.getSimilar -> track.getSimilar (impl uses getSimilarTracks on the seed track; note the deliberate refinement from the original plan). - cache: delete the key before the size check so refreshing an expired non-oldest entry doesn't leave the cache under capacity. - drop the redundant similar.slice (getSimilarTracks already caps at MAX_SEED_SIMILAR); reword the match-weight comment to [0.5, 1.0]. --- .../2026-06-07-autoplay-seed-similarity-spine.md | 13 ++++++++----- .../music/autoplay/seedSimilarityCollector.ts | 16 +++++++++++----- 2 files changed, 19 insertions(+), 10 deletions(-) diff --git a/decisions/2026-06-07-autoplay-seed-similarity-spine.md b/decisions/2026-06-07-autoplay-seed-similarity-spine.md index d66276b2e..d287981a4 100644 --- a/decisions/2026-06-07-autoplay-seed-similarity-spine.md +++ b/decisions/2026-06-07-autoplay-seed-similarity-spine.md @@ -32,10 +32,13 @@ just Spotify-search-on-seed + genre-top-tracks fallback, with genre guards disab foundation; B stops the fail-open + genre-blind boost; the critic confirmed A alone is insufficient because B is what blocks drift when A's pool is thin. -**A — Seed-similarity spine (foundation).** Wire Last.fm `artist.getSimilar` (no user link; -`LASTFM_API_KEY` is set) on the **currently-playing seed artist** into the candidate collector -as a backbone source, boosted as "seed-similar". Demote the deprecated Spotify-recs source to a -best-effort extra (leave as a no-op-tolerant call; do not remove yet). +**A — Seed-similarity spine (foundation).** Wire Last.fm `track.getSimilar` (via +`getSimilarTracks`, `lastFmApi.ts:486`; no user link, `LASTFM_API_KEY` is set) on the +**currently-playing seed track** into the candidate collector as a backbone source, boosted as +"seed-similar". (Implementation refined the original `artist.getSimilar` plan to +`track.getSimilar` — it returns playable track candidates directly and reuses existing bot +infra, avoiding a shared-package ESM-exports change.) Demote the deprecated Spotify-recs source +to a best-effort extra (leave as a no-op-tolerant call; do not remove yet). - Guardrails: wrap in a 2s timeout + per-artist cache (reuse/extend `artistTagCache`, ~1h TTL) - 429 backoff. Cap to ~10 similar artists, one search each, to bound Last.fm calls (~5 req/s @@ -53,7 +56,7 @@ best-effort extra (leave as a no-op-tolerant call; do not remove yet). pool); no boost on a known cross-family candidate. **C — Fast-follow (only if A+B leave residual drift).** Replace the fallback's -`getTagTopTracks` with seed-artist-similar tracks (same `artist.getSimilar`) so even the last +`getTagTopTracks` with seed-similar tracks (same `track.getSimilar` as A) so even the last resort stays on-genre. **Non-negotiable guardrail across all:** autoplay must never stall. Keep the skip-storm diff --git a/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts b/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts index 91eeebf3f..8653eec83 100644 --- a/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts +++ b/packages/bot/src/utils/music/autoplay/seedSimilarityCollector.ts @@ -61,6 +61,11 @@ async function fetchSeedSimilar( }) if (tracks.length > 0) { + // Drop any existing (expired) entry for this key first so the size + // check reflects live capacity — Map.set updates in place without + // reordering, which would otherwise leave the cache under capacity + // when refreshing a non-oldest expired entry. + similarCache.delete(key) if (similarCache.size >= SIMILAR_CACHE_MAX) { const oldest = similarCache.keys().next().value if (oldest) similarCache.delete(oldest) @@ -108,9 +113,9 @@ export async function collectSeedSimilarCandidates( // No early-out on pool size: this is the seed-similarity SPINE, so the // current-track anchor must always land — even when collectRecommendation- // Candidates already filled the pool from (possibly drifted) history seeds. - // The loop is bounded by MAX_SEED_SIMILAR; the larger pool only widens the - // diverse-selection choice downstream. - for (const s of similar.slice(0, MAX_SEED_SIMILAR)) { + // `similar` is already capped at MAX_SEED_SIMILAR by the fetch; the larger + // pool only widens the diverse-selection choice downstream. + for (const s of similar) { const query = cleanSearchQuery(s.title, s.artist) const tracks = await searchLastFmQuery(ctx.queue, query, requestedBy) for (const track of tracks) { @@ -147,8 +152,9 @@ export async function collectSeedSimilarCandidates( }, }) // Last.fm returns `match` on a 0..1 scale. Weight by it but keep - // seed-similar competitive — never crush below 0.5× even on a weak - // match, so the spine still grounds a thin pool. + // seed-similar competitive — the weight is clamped to [0.5, 1.0] + // (never weighted below 0.5 even on a weak match), so the spine + // still grounds a thin pool. const matchWeight = 0.5 + 0.5 * Math.min(Math.max(s.match, 0), 1) upsertScoredCandidate( candidates,