Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
64 changes: 38 additions & 26 deletions packages/bot/src/utils/music/queueManipulation.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,12 @@ import {
} from './autoplay/lastFmSeeds'
import { detectSessionMood, type SessionMood } from './autoplay/sessionMood'
import { getSimilarTracks, getTagTopTracks } from '../../lastfm'
import { cleanSearchQuery, cleanTitle, cleanAuthor } from './searchQueryCleaner'
import {
cleanSearchQuery,
cleanTitle,
cleanAuthor,
extractSongCore,
} from './searchQueryCleaner'
import type { QueueMetadata } from '../../types/QueueMetadata'

const AUTOPLAY_BUFFER_SIZE = 8
Expand All @@ -54,46 +59,45 @@ async function getTrackAudioFeatures(
userId: string,
): Promise<SpotifyAudioFeatures | null> {
const cacheKey = normalizeTrackKey(track.title, track.author)

const cached = audioFeatureCache.get(cacheKey)
if (cached !== undefined) {
return cached
}

const token = await spotifyLinkService.getValidAccessToken(userId)
if (!token) {
audioFeatureCache.set(cacheKey, null)
return null
}

let spotifyId: string | null = null

if (track.url && track.url.includes('open.spotify.com/track/')) {
const match = track.url.match(/track\/([a-zA-Z0-9]+)/)
if (match) {
spotifyId = match[1]
}
}

if (!spotifyId) {
spotifyId = await searchSpotifyTrack(
token,
cleanTitle(track.title ?? ''),
cleanAuthor(track.author ?? ''),
)
}

if (!spotifyId) {
audioFeatureCache.set(cacheKey, null)
return null
}

const features = await getAudioFeatures(token, spotifyId).catch(() => null)
audioFeatureCache.set(cacheKey, features)
return features
}


type ScoredTrack = {
track: Track
score: number
Expand Down Expand Up @@ -399,9 +403,10 @@ async function _replenishQueue(
const beforeLastFm = candidates.size
const metadata = queue.metadata as QueueMetadata | undefined
const vcMemberIds = metadata?.vcMemberIds ?? []
const contributionWeights = vcMemberIds.length > 1
? buildVcContributionWeights(allHistoryTracks, vcMemberIds)
: new Map<string, number>()
const contributionWeights =
vcMemberIds.length > 1
? buildVcContributionWeights(allHistoryTracks, vcMemberIds)
: new Map<string, number>()
await collectLastFmCandidates(
queue,
requestedBy,
Expand Down Expand Up @@ -517,9 +522,9 @@ async function _replenishQueue(
(autoplayMode === 'discover' || autoplayMode === 'popular') &&
requestedBy?.id
) {
const token = await Promise.resolve(spotifyLinkService
.getValidAccessToken(requestedBy.id))
.catch(() => null)
const token = await Promise.resolve(
spotifyLinkService.getValidAccessToken(requestedBy.id),
).catch(() => null)
if (token) {
await Promise.all(
enriched.slice(0, 3).map(async (track) => {
Expand All @@ -528,10 +533,7 @@ async function _replenishQueue(
track.track.author,
).catch(() => null)
if (popularity === null) return
if (
autoplayMode === 'popular' &&
popularity >= 70
) {
if (autoplayMode === 'popular' && popularity >= 70) {
track.score += 0.12
} else if (
autoplayMode === 'discover' &&
Expand All @@ -548,7 +550,10 @@ async function _replenishQueue(
if (enriched.length === 0) {
warnLog({
message: 'Autoplay: no candidates selected — queue may stall',
data: { guildId: queue.guild.id, candidatePoolSize: candidates.size },
data: {
guildId: queue.guild.id,
candidatePoolSize: candidates.size,
},
})
replenishCounters.set(guildId, replenishCount + 1)
return
Expand Down Expand Up @@ -662,6 +667,8 @@ function buildExcludedKeys(
for (const t of allTracks) {
keys.push(normalizeTrackKey(t.title, t.author))
keys.push(normalizeTitleOnly(t.title))
const core = extractSongCore(t.title ?? '', t.author)
if (core) keys.push(normalizeText(core))
Comment on lines +670 to +671

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟠 Major

Core-key dedup still doesn't apply between candidates picked in the same replenish pass.

These additions only compare a track against the prebuilt exclusion set. If the same search batch returns both Beyoncé - Halo and Halo - Beyoncé, they can still survive together because upsertScoredCandidate() keys by full title/author and selectDiverseCandidates() only blocks normalizeTitleOnly(...). So this fixes queue/history duplicates, but not the “two variants added together right now” case.

Also applies to: 1453-1454

🤖 Prompt for AI Agents
Verify each finding against the current code and only fix it if needed.

In `@packages/bot/src/utils/music/queueManipulation.ts` around lines 670 - 671,
The current dedup logic only checks candidate cores against the prebuilt
exclusion set, so variants returned in the same replenish pass (e.g., "Beyoncé -
Halo" and "Halo - Beyoncé") can both be accepted; update the selection flow in
selectDiverseCandidates/upsertScoredCandidate to also maintain a local set of
core keys for the current pass (use extractSongCore and normalizeText to derive
keys, and also consider normalizeTitleOnly where relevant) and check each
candidate against both the prebuilt exclusion set and this local currentPassKeys
before accepting; when a candidate is accepted, add its core key to
currentPassKeys so subsequent candidates in the same pass are blocked.

}
return new Set(keys)
}
Expand Down Expand Up @@ -996,7 +1003,10 @@ async function collectLastFmCandidates(
})
}

const similar = await getSimilarTracks(seed.artist, cleanTitle(seed.title))
const similar = await getSimilarTracks(
seed.artist,
cleanTitle(seed.title),
)
for (const s of similar.slice(0, MAX_SIMILAR_LOOKUPS)) {
const query = cleanSearchQuery(s.title, s.artist)
const tracks = await searchLastFmQuery(queue, query, requestedBy)
Expand Down Expand Up @@ -1096,8 +1106,7 @@ function addGenreTrackCandidate(
return
const key = normalizeTrackKey(track.title, track.author)
const dislikedWeight = ctx.dislikedTrackKeys.get(key)
if (dislikedWeight !== undefined && dislikedWeight > 0.5)
return
if (dislikedWeight !== undefined && dislikedWeight > 0.5) return
const rec = calculateRecommendationScore(
track,
ctx.currentTrack,
Expand Down Expand Up @@ -1139,15 +1148,16 @@ async function collectGenreCandidates(
}
}


async function enrichWithAudioFeatures(
tracks: ScoredTrack[],
userId: string,
currentFeatures: SpotifyAudioFeatures | null,
): Promise<ScoredTrack[]> {
if (!currentFeatures || !userId) return tracks

const token = await Promise.resolve(spotifyLinkService.getValidAccessToken(userId)).catch(() => null)
const token = await Promise.resolve(
spotifyLinkService.getValidAccessToken(userId),
).catch(() => null)
if (!token) return tracks

const spotifyIds: string[] = []
Expand Down Expand Up @@ -1439,7 +1449,9 @@ function isDuplicateCandidate(
}
if (excludedKeys.has(normalizeTrackKey(track.title, track.author)))
return true
return excludedKeys.has(normalizeTitleOnly(track.title))
if (excludedKeys.has(normalizeTitleOnly(track.title))) return true
const core = extractSongCore(track.title ?? '', track.author)
return core !== null && excludedKeys.has(normalizeText(core))
}

function calculateRecommendationScore(
Expand Down
88 changes: 88 additions & 0 deletions packages/bot/src/utils/music/searchQueryCleaner.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@ import {
cleanAuthor,
cleanSearchQuery,
isSpamChannel,
extractSongCore,
} from './searchQueryCleaner'

describe('cleanTitle', () => {
Expand Down Expand Up @@ -117,6 +118,93 @@ describe('cleanSearchQuery', () => {
})
})

describe('cleanTitle — Brazilian noise', () => {
it('strips (Tradução) variants', () => {
expect(cleanTitle('Beyoncé - Halo (Tradução)')).toBe('Beyoncé - Halo')
expect(cleanTitle('Beyoncé - Halo (Tradução/Legendado)')).toBe(
'Beyoncé - Halo',
)
expect(cleanTitle('Beyoncé - Halo (Tradução PT-BR)')).toBe(
'Beyoncé - Halo',
)
})

it('strips standalone Legendado', () => {
expect(cleanTitle('Beyoncé - Halo Legendado')).toBe('Beyoncé - Halo')
})

it('strips (Clipe Oficial) variants', () => {
expect(cleanTitle('Beyoncé - Halo (Clipe Oficial)')).toBe(
'Beyoncé - Halo',
)
expect(cleanTitle('Beyoncé - Halo (Clipe Oficial HD)')).toBe(
'Beyoncé - Halo',
)
})

it('strips hashtags', () => {
expect(cleanTitle('Beyoncé - Halo #music #lyrics')).toBe(
'Beyoncé - Halo',
)
})

it('strips bare Lyrics word', () => {
expect(cleanTitle('Beyoncé - Halo Lyrics')).toBe('Beyoncé - Halo')
})

it('strips combined Brazilian noise', () => {
expect(
cleanTitle('Beyonce - Halo (Tradução)( legendado)(Clipe Oficial)'),
).toBe('Beyonce - Halo')
})
})

describe('extractSongCore', () => {
it('extracts right side of Artist - Song', () => {
expect(extractSongCore('Beyoncé - Halo')).toBe('Halo')
})

it('extracts left side when author matches right side (inverted format)', () => {
expect(
extractSongCore('Halo - Beyoncé (Lyrics)', 'Beyoncé - Topic'),
).toBe('Halo')
})

it('extracts right side when author matches left side', () => {
expect(
extractSongCore('Beyoncé - Halo (Tradução)', 'Beyoncé - Topic'),
).toBe('Halo')
})

it('trims secondary separators from the extracted core', () => {
expect(
extractSongCore(
'Beyoncé - Halo - VERSÃO FORROZINHO',
'Beyoncé - Topic',
),
).toBe('Halo')
})

it('returns null when no separator is found', () => {
expect(extractSongCore('Bohemian Rhapsody')).toBeNull()
})

it('strips noise from title before extracting', () => {
expect(
extractSongCore(
'Beyoncé - Halo (Tradução/Legendado)',
'Beyoncé - Topic',
),
).toBe('Halo')
})

it('defaults to right side when author does not match either part', () => {
expect(extractSongCore('Beyoncé - Halo', 'someuploader123')).toBe(
'Halo',
)
})
})

describe('isSpamChannel', () => {
it('flags "Best Songs" as spam', () => {
expect(isSpamChannel('Best Songs')).toBe(true)
Expand Down
59 changes: 59 additions & 0 deletions packages/bot/src/utils/music/searchQueryCleaner.ts
Original file line number Diff line number Diff line change
Expand Up @@ -99,6 +99,15 @@

// YouTube auto-generated "Topic" channel suffix
/\s{0,3}-\s{0,3}topic\b/gi,

// Brazilian/Portuguese YouTube title noise
/#\S+/g,
/\(tradu[çc][aã]o[^)]*\)/gi,
/\[tradu[çc][aã]o[^\]]*\]/gi,
/\blegendado\b/gi,
/\(clipe\s+oficial[^)]*\)/gi,
/\[clipe\s+oficial[^\]]*\]/gi,
/\blyrics\b/gi,
]

const HYPHENATED_VERSION_SUFFIXES: RegExp[] = [
Expand Down Expand Up @@ -200,6 +209,56 @@
return `${cleanedTitle} ${cleanedAuthor}`.trim()
}

/**
* Extract the song-title core from a YouTube-style "Artist - Song" title.
* Uses the author field to determine which side of the separator is the artist
* (handles both "Artist - Song" and inverted "Song - Artist" formats).
* Strips secondary separators from the core (e.g. "Halo - VERSÃO FORROZINHO" → "Halo").
* Returns null when no separator is found in the cleaned title.
*/
export function extractSongCore(title: string, author?: string): string | null {

Check failure on line 219 in packages/bot/src/utils/music/searchQueryCleaner.ts

View check run for this annotation

SonarQubeCloud / SonarCloud Code Analysis

Refactor this function to reduce its Cognitive Complexity from 22 to the 15 allowed.

See more on https://sonarcloud.io/project/issues?id=LucasSantana-Dev_Lucky&issues=AZ2FJ6n1NouVguQbly5i&open=AZ2FJ6n1NouVguQbly5i&pullRequest=582
const cleaned = cleanTitle(title)
for (const sep of [' - ', ' – ', ' — ']) {
const idx = cleaned.indexOf(sep)
if (idx < 2 || idx > 60) continue
const left = cleaned.slice(0, idx).trim()
if (/[()[\]]/.test(left)) continue
const right = cleaned.slice(idx + sep.length).trim()
if (left.length < 2 || right.length < 2) continue

let songPart: string
if (author) {
const norm = (s: string) =>
s.toLowerCase().replaceAll(/[^a-z0-9]+/g, '')
const authNorm = norm(cleanAuthor(author))
const overlaps = (a: string, b: string) =>
a.length >= 3 &&
b.length >= 3 &&
(a.includes(b.slice(0, 4)) || b.includes(a.slice(0, 4)))
if (overlaps(authNorm, norm(left))) {
songPart = right
} else if (overlaps(authNorm, norm(right))) {
songPart = left
} else {
songPart = right
}
Comment on lines +231 to +244

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟠 Major

extractSongCore stops disambiguating inverted titles for non-Latin artist names.

The overlap check drops every character outside [a-z0-9] before comparing author, left, and right. For artists like 아이유, 宇多田ヒカル, or Руки Вверх, both sides normalize to '', so an inverted title such as 좋은날 - 아이유 falls through to songPart = right and returns the artist instead of the song. That means the new core-key dedup still misses multilingual variants.

🤖 Prompt for AI Agents
Verify each finding against the current code and only fix it if needed.

In `@packages/bot/src/utils/music/searchQueryCleaner.ts` around lines 231 - 244,
The current normalization in extractSongCore (norm + cleanAuthor) strips
non-ASCII letters so non-Latin artist names become empty and overlaps() can't
disambiguate; update norm to preserve Unicode letters and numbers (use a
Unicode-aware regex like excluding [^\p{L}\p{N}] with the u flag) so authNorm,
norm(left), and norm(right) retain non-Latin characters, and keep the overlaps
logic (overlaps, songPart, left, right) intact so inverted titles for
multilingual artists are correctly detected.

} else {
songPart = right
}

for (const innerSep of [' - ', ' – ', ' — ']) {
const innerIdx = songPart.indexOf(innerSep)
if (innerIdx > 0) {
songPart = songPart.slice(0, innerIdx).trim()
break
}
}

return songPart.length >= 2 ? songPart : null
}
return null
}

/**
* True if the author is a known noise/compilation channel.
* Callers can use this to invalidate a match and trigger a retry with a
Expand Down
Loading