Match LRCLIB on the song title when the channel is not the artist, and stop generic words like Christian or worship from vouching for an artist
This commit is contained in:
@@ -99,27 +99,64 @@ def http_json(url, tries=4):
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
# Words that say nothing about WHO recorded a song. A lyric-video channel
|
||||||
|
# called "Christian Lyrics" otherwise "verifies" any track featuring someone
|
||||||
|
# named Christian — which is exactly how "Still" matched Nicky Romero.
|
||||||
|
GENERIC = {
|
||||||
|
'the', 'and', 'of', 'a', 'feat', 'featuring', 'ft', 'with', 'band', 'music', 'musica',
|
||||||
|
'worship', 'ministries', 'ministry', 'christian', 'gospel', 'praise', 'church', 'choir',
|
||||||
|
'lyrics', 'lyric', 'live', 'official', 'records', 'recordings', 'group', 'project', 'team',
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def artist_ok(lrc_artist, channel, video_title):
|
def artist_ok(lrc_artist, channel, video_title):
|
||||||
"""True when the LRCLIB artist plausibly matches the video.
|
"""True when the LRCLIB artist plausibly matches the video.
|
||||||
|
|
||||||
Lyric-video channels ("Christian Lyrics", a person's name) carry no artist,
|
Lyric-video channels ("Christian Lyrics", a person's name) carry no artist,
|
||||||
so the artist is also looked for in the video title. Songs whose title is
|
so the artist is also looked for in the video title. Songs whose title is
|
||||||
shared across genres ("Still") otherwise match the wrong recording."""
|
shared across genres ("Still") otherwise match the wrong recording."""
|
||||||
a = set(norm(lrc_artist).split()) - {'the', 'and', 'of', 'band', 'music', 'worship', 'ministries'}
|
a = set(norm(lrc_artist).split()) - GENERIC
|
||||||
if not a:
|
if not a:
|
||||||
return False
|
return False
|
||||||
hay = set(norm(channel).split()) | set(norm(video_title).split())
|
hay = set(norm(channel).split()) | set(norm(video_title).split())
|
||||||
return bool(a & hay)
|
return bool(a & hay)
|
||||||
|
|
||||||
|
|
||||||
def lrclib_lookup(title, artist, duration, tolerance=6):
|
def title_run(track, video_title):
|
||||||
"""Best LRCLIB entry for a song, or None. Prefers synced lyrics."""
|
"""True when LRCLIB's track name is the video's title, allowing the video
|
||||||
|
to carry extra words around it ("Lakewood Live - Holy You Are").
|
||||||
|
|
||||||
|
Compared word by word, never as a substring: "Still" IS a substring of
|
||||||
|
"(You Can Still) Rock in America", and that is exactly how a one-word
|
||||||
|
title matches the wrong song."""
|
||||||
|
a, b = norm(track).split(), norm(video_title).split()
|
||||||
|
if not a or not b:
|
||||||
|
return False
|
||||||
|
if a == b:
|
||||||
|
return True
|
||||||
|
if len(a) < 2:
|
||||||
|
return False # one word matches by accident far too often
|
||||||
|
return any(b[i:i + len(a)] == a for i in range(len(b) - len(a) + 1))
|
||||||
|
|
||||||
|
|
||||||
|
def lrclib_lookup(title, artist, duration, tolerance=6, verify=None):
|
||||||
|
"""Best LRCLIB entry for a song, or None. Prefers synced lyrics, and
|
||||||
|
strongly prefers a candidate whose artist verifies against the video —
|
||||||
|
"Still" returns both Hillsong Worship and Night Ranger."""
|
||||||
q = {'track_name': title, 'artist_name': artist or ''}
|
q = {'track_name': title, 'artist_name': artist or ''}
|
||||||
if duration:
|
if duration:
|
||||||
q['duration'] = str(int(round(duration)))
|
q['duration'] = str(int(round(duration)))
|
||||||
hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q))
|
hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q))
|
||||||
if not hit:
|
if not hit:
|
||||||
results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': f'{title} {artist}'.strip()})) or []
|
# The "artist" is really the YouTube channel ("Integrity Worship",
|
||||||
|
# "Christian Lyrics"), so a title+artist search often finds nothing
|
||||||
|
# where a title-only one finds the song. Try both, widest last.
|
||||||
|
results = []
|
||||||
|
for query in ([f'{title} {artist}'.strip(), title] if artist else [title]):
|
||||||
|
results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': query})) or []
|
||||||
|
if results:
|
||||||
|
break
|
||||||
|
time.sleep(0.4)
|
||||||
want = norm(title)
|
want = norm(title)
|
||||||
scored = []
|
scored = []
|
||||||
for x in results:
|
for x in results:
|
||||||
@@ -130,7 +167,8 @@ def lrclib_lookup(title, artist, duration, tolerance=6):
|
|||||||
title_hit = 2 if t == want else 1 if (want in t or t in want) else 0
|
title_hit = 2 if t == want else 1 if (want in t or t in want) else 0
|
||||||
if not title_hit or (duration and dd > tolerance):
|
if not title_hit or (duration and dd > tolerance):
|
||||||
continue
|
continue
|
||||||
scored.append((title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x))
|
ok = 50 if (verify and verify(x.get('artistName'))) else 0
|
||||||
|
scored.append((ok + title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x))
|
||||||
if not scored:
|
if not scored:
|
||||||
return None
|
return None
|
||||||
hit = max(scored, key=lambda p: p[0])[1]
|
hit = max(scored, key=lambda p: p[0])[1]
|
||||||
@@ -205,8 +243,9 @@ def main():
|
|||||||
vid, meta, cur = p['id'], p['meta'], p['cur']
|
vid, meta, cur = p['id'], p['meta'], p['cur']
|
||||||
title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel'))
|
title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel'))
|
||||||
dur = float(meta.get('duration') or 0)
|
dur = float(meta.get('duration') or 0)
|
||||||
|
verify = lambda a: artist_ok(a, meta.get('channel'), meta.get('title'))
|
||||||
try:
|
try:
|
||||||
hit = lrclib_lookup(title, artist, dur, args.tolerance)
|
hit = lrclib_lookup(title, artist, dur, args.tolerance, verify)
|
||||||
except Exception as e: # network hiccup — keep going
|
except Exception as e: # network hiccup — keep going
|
||||||
print(f'{vid} lookup failed: {e}')
|
print(f'{vid} lookup failed: {e}')
|
||||||
continue
|
continue
|
||||||
@@ -216,12 +255,19 @@ def main():
|
|||||||
print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})')
|
print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})')
|
||||||
continue
|
continue
|
||||||
h = hit['hit']
|
h = hit['hit']
|
||||||
sure = artist_ok(h.get('artistName'), meta.get('channel'), meta.get('title'))
|
sure = verify(h.get('artistName'))
|
||||||
mark = 'LRCLIB' if sure else 'UNSURE'
|
mark = 'LRCLIB' if sure else 'UNSURE'
|
||||||
print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}')
|
# Worship uploads are often credited to a lyric-video channel, so the
|
||||||
|
# artist can't be checked. The same song title at the same length to
|
||||||
|
# within 3 s is evidence in its own right — a different recording of a
|
||||||
|
# same-named song is essentially never that close.
|
||||||
|
dd = abs((h.get('duration') or 0) - dur) if dur else 99
|
||||||
|
if not sure and dd <= 3 and title_run(h.get('trackName'), meta.get('title')):
|
||||||
|
sure, mark = True, 'LENGTH'
|
||||||
|
print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}' + (f' | Δ{dd:.1f}s' if dd < 99 else ''))
|
||||||
if not sure and not args.loose:
|
if not sure and not args.loose:
|
||||||
skipped += 1
|
skipped += 1
|
||||||
print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title — left alone (use --loose to accept)')
|
print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title and the length differs by {dd:.0f}s — left alone (use --loose to accept)')
|
||||||
continue
|
continue
|
||||||
if not args.apply:
|
if not args.apply:
|
||||||
continue
|
continue
|
||||||
|
|||||||
Reference in New Issue
Block a user