Match LRCLIB on the song title when the channel is not the artist, and stop generic words like Christian or worship from vouching for an artist

This commit is contained in:
Jonathan Sykes
2026-09-20 18:06:04 +08:00
parent 2987f46037
commit 89d35562d3

View File

@@ -99,27 +99,64 @@ def http_json(url, tries=4):
return None return None
# Words that say nothing about WHO recorded a song. A lyric-video channel
# called "Christian Lyrics" otherwise "verifies" any track featuring someone
# named Christian — which is exactly how "Still" matched Nicky Romero.
GENERIC = {
'the', 'and', 'of', 'a', 'feat', 'featuring', 'ft', 'with', 'band', 'music', 'musica',
'worship', 'ministries', 'ministry', 'christian', 'gospel', 'praise', 'church', 'choir',
'lyrics', 'lyric', 'live', 'official', 'records', 'recordings', 'group', 'project', 'team',
}
def artist_ok(lrc_artist, channel, video_title): def artist_ok(lrc_artist, channel, video_title):
"""True when the LRCLIB artist plausibly matches the video. """True when the LRCLIB artist plausibly matches the video.
Lyric-video channels ("Christian Lyrics", a person's name) carry no artist, Lyric-video channels ("Christian Lyrics", a person's name) carry no artist,
so the artist is also looked for in the video title. Songs whose title is so the artist is also looked for in the video title. Songs whose title is
shared across genres ("Still") otherwise match the wrong recording.""" shared across genres ("Still") otherwise match the wrong recording."""
a = set(norm(lrc_artist).split()) - {'the', 'and', 'of', 'band', 'music', 'worship', 'ministries'} a = set(norm(lrc_artist).split()) - GENERIC
if not a: if not a:
return False return False
hay = set(norm(channel).split()) | set(norm(video_title).split()) hay = set(norm(channel).split()) | set(norm(video_title).split())
return bool(a & hay) return bool(a & hay)
def lrclib_lookup(title, artist, duration, tolerance=6): def title_run(track, video_title):
"""Best LRCLIB entry for a song, or None. Prefers synced lyrics.""" """True when LRCLIB's track name is the video's title, allowing the video
to carry extra words around it ("Lakewood Live - Holy You Are").
Compared word by word, never as a substring: "Still" IS a substring of
"(You Can Still) Rock in America", and that is exactly how a one-word
title matches the wrong song."""
a, b = norm(track).split(), norm(video_title).split()
if not a or not b:
return False
if a == b:
return True
if len(a) < 2:
return False # one word matches by accident far too often
return any(b[i:i + len(a)] == a for i in range(len(b) - len(a) + 1))
def lrclib_lookup(title, artist, duration, tolerance=6, verify=None):
"""Best LRCLIB entry for a song, or None. Prefers synced lyrics, and
strongly prefers a candidate whose artist verifies against the video —
"Still" returns both Hillsong Worship and Night Ranger."""
q = {'track_name': title, 'artist_name': artist or ''} q = {'track_name': title, 'artist_name': artist or ''}
if duration: if duration:
q['duration'] = str(int(round(duration))) q['duration'] = str(int(round(duration)))
hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q)) hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q))
if not hit: if not hit:
results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': f'{title} {artist}'.strip()})) or [] # The "artist" is really the YouTube channel ("Integrity Worship",
# "Christian Lyrics"), so a title+artist search often finds nothing
# where a title-only one finds the song. Try both, widest last.
results = []
for query in ([f'{title} {artist}'.strip(), title] if artist else [title]):
results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': query})) or []
if results:
break
time.sleep(0.4)
want = norm(title) want = norm(title)
scored = [] scored = []
for x in results: for x in results:
@@ -130,7 +167,8 @@ def lrclib_lookup(title, artist, duration, tolerance=6):
title_hit = 2 if t == want else 1 if (want in t or t in want) else 0 title_hit = 2 if t == want else 1 if (want in t or t in want) else 0
if not title_hit or (duration and dd > tolerance): if not title_hit or (duration and dd > tolerance):
continue continue
scored.append((title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x)) ok = 50 if (verify and verify(x.get('artistName'))) else 0
scored.append((ok + title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x))
if not scored: if not scored:
return None return None
hit = max(scored, key=lambda p: p[0])[1] hit = max(scored, key=lambda p: p[0])[1]
@@ -205,8 +243,9 @@ def main():
vid, meta, cur = p['id'], p['meta'], p['cur'] vid, meta, cur = p['id'], p['meta'], p['cur']
title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel')) title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel'))
dur = float(meta.get('duration') or 0) dur = float(meta.get('duration') or 0)
verify = lambda a: artist_ok(a, meta.get('channel'), meta.get('title'))
try: try:
hit = lrclib_lookup(title, artist, dur, args.tolerance) hit = lrclib_lookup(title, artist, dur, args.tolerance, verify)
except Exception as e: # network hiccup — keep going except Exception as e: # network hiccup — keep going
print(f'{vid} lookup failed: {e}') print(f'{vid} lookup failed: {e}')
continue continue
@@ -216,12 +255,19 @@ def main():
print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})') print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})')
continue continue
h = hit['hit'] h = hit['hit']
sure = artist_ok(h.get('artistName'), meta.get('channel'), meta.get('title')) sure = verify(h.get('artistName'))
mark = 'LRCLIB' if sure else 'UNSURE' mark = 'LRCLIB' if sure else 'UNSURE'
print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}') # Worship uploads are often credited to a lyric-video channel, so the
# artist can't be checked. The same song title at the same length to
# within 3 s is evidence in its own right — a different recording of a
# same-named song is essentially never that close.
dd = abs((h.get('duration') or 0) - dur) if dur else 99
if not sure and dd <= 3 and title_run(h.get('trackName'), meta.get('title')):
sure, mark = True, 'LENGTH'
print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}' + (f' | Δ{dd:.1f}s' if dd < 99 else ''))
if not sure and not args.loose: if not sure and not args.loose:
skipped += 1 skipped += 1
print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title — left alone (use --loose to accept)') print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title and the length differs by {dd:.0f}s — left alone (use --loose to accept)')
continue continue
if not args.apply: if not args.apply:
continue continue