diff --git a/scripts/lyrics/lrclib_regen.py b/scripts/lyrics/lrclib_regen.py index 73f048c..7cc7f9d 100644 --- a/scripts/lyrics/lrclib_regen.py +++ b/scripts/lyrics/lrclib_regen.py @@ -99,27 +99,64 @@ def http_json(url, tries=4): return None +# Words that say nothing about WHO recorded a song. A lyric-video channel +# called "Christian Lyrics" otherwise "verifies" any track featuring someone +# named Christian — which is exactly how "Still" matched Nicky Romero. +GENERIC = { + 'the', 'and', 'of', 'a', 'feat', 'featuring', 'ft', 'with', 'band', 'music', 'musica', + 'worship', 'ministries', 'ministry', 'christian', 'gospel', 'praise', 'church', 'choir', + 'lyrics', 'lyric', 'live', 'official', 'records', 'recordings', 'group', 'project', 'team', +} + + def artist_ok(lrc_artist, channel, video_title): """True when the LRCLIB artist plausibly matches the video. Lyric-video channels ("Christian Lyrics", a person's name) carry no artist, so the artist is also looked for in the video title. Songs whose title is shared across genres ("Still") otherwise match the wrong recording.""" - a = set(norm(lrc_artist).split()) - {'the', 'and', 'of', 'band', 'music', 'worship', 'ministries'} + a = set(norm(lrc_artist).split()) - GENERIC if not a: return False hay = set(norm(channel).split()) | set(norm(video_title).split()) return bool(a & hay) -def lrclib_lookup(title, artist, duration, tolerance=6): - """Best LRCLIB entry for a song, or None. Prefers synced lyrics.""" +def title_run(track, video_title): + """True when LRCLIB's track name is the video's title, allowing the video + to carry extra words around it ("Lakewood Live - Holy You Are"). + + Compared word by word, never as a substring: "Still" IS a substring of + "(You Can Still) Rock in America", and that is exactly how a one-word + title matches the wrong song.""" + a, b = norm(track).split(), norm(video_title).split() + if not a or not b: + return False + if a == b: + return True + if len(a) < 2: + return False # one word matches by accident far too often + return any(b[i:i + len(a)] == a for i in range(len(b) - len(a) + 1)) + + +def lrclib_lookup(title, artist, duration, tolerance=6, verify=None): + """Best LRCLIB entry for a song, or None. Prefers synced lyrics, and + strongly prefers a candidate whose artist verifies against the video — + "Still" returns both Hillsong Worship and Night Ranger.""" q = {'track_name': title, 'artist_name': artist or ''} if duration: q['duration'] = str(int(round(duration))) hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q)) if not hit: - results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': f'{title} {artist}'.strip()})) or [] + # The "artist" is really the YouTube channel ("Integrity Worship", + # "Christian Lyrics"), so a title+artist search often finds nothing + # where a title-only one finds the song. Try both, widest last. + results = [] + for query in ([f'{title} {artist}'.strip(), title] if artist else [title]): + results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': query})) or [] + if results: + break + time.sleep(0.4) want = norm(title) scored = [] for x in results: @@ -130,7 +167,8 @@ def lrclib_lookup(title, artist, duration, tolerance=6): title_hit = 2 if t == want else 1 if (want in t or t in want) else 0 if not title_hit or (duration and dd > tolerance): continue - scored.append((title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x)) + ok = 50 if (verify and verify(x.get('artistName'))) else 0 + scored.append((ok + title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x)) if not scored: return None hit = max(scored, key=lambda p: p[0])[1] @@ -205,8 +243,9 @@ def main(): vid, meta, cur = p['id'], p['meta'], p['cur'] title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel')) dur = float(meta.get('duration') or 0) + verify = lambda a: artist_ok(a, meta.get('channel'), meta.get('title')) try: - hit = lrclib_lookup(title, artist, dur, args.tolerance) + hit = lrclib_lookup(title, artist, dur, args.tolerance, verify) except Exception as e: # network hiccup — keep going print(f'{vid} lookup failed: {e}') continue @@ -216,12 +255,19 @@ def main(): print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})') continue h = hit['hit'] - sure = artist_ok(h.get('artistName'), meta.get('channel'), meta.get('title')) + sure = verify(h.get('artistName')) mark = 'LRCLIB' if sure else 'UNSURE' - print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}') + # Worship uploads are often credited to a lyric-video channel, so the + # artist can't be checked. The same song title at the same length to + # within 3 s is evidence in its own right — a different recording of a + # same-named song is essentially never that close. + dd = abs((h.get('duration') or 0) - dur) if dur else 99 + if not sure and dd <= 3 and title_run(h.get('trackName'), meta.get('title')): + sure, mark = True, 'LENGTH' + print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}' + (f' | Δ{dd:.1f}s' if dd < 99 else '')) if not sure and not args.loose: skipped += 1 - print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title — left alone (use --loose to accept)') + print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title and the length differs by {dd:.0f}s — left alone (use --loose to accept)') continue if not args.apply: continue