Files
ytplayer/scripts/lyrics/lrclib_regen.py

289 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""lrclib_regen.py — back up and regenerate lyrics on worship.hesed.sbs from LRCLIB.
Machine transcripts (Whisper/Scribe) mishear words and drift; published lyrics
on LRCLIB are usually correct and often SYNCED. This tool:
1. writes a JSON backup of the CURRENT lyrics of every song it will touch
(the server also keeps each previous version as a revision — nothing is
ever lost, this file is just an offline copy),
2. looks each song up on LRCLIB by cleaned title + artist + duration,
3. replaces the lyrics only when a confident match is found.
Song metadata (title / artist / duration) comes from the public
/api/streams endpoint, so this works without admin access; writing needs
YTP_ADMIN_PASSWORD or YTP_TOKEN.
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --ids A,B --dry-run
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --since '2026-09-19 01:45' --until '2026-09-19 02:00'
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --tagged auto-transcribed --apply
"""
import argparse
import datetime as dt
import json
import os
import re
import sys
import time
import urllib.parse
import urllib.request
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from auto_lyrics import Api # noqa: E402
LRCLIB = 'https://lrclib.net/api'
UA = {'User-Agent': 'ytplayer-lyrics-regen (https://worship.hesed.sbs)'}
NOISE = re.compile(r'\b(official|video|audio|lyrics?|lyric|hd|hq|4k|live|mv|karaoke|minus[\s-]?one|instrumental|cover|remaster(ed)?|visualizer|performance|version)\b', re.I)
def clean_title(raw):
t = str(raw or '').split('|')[0]
t = re.sub(r'\([^)]*\)|\[[^\]]*\]', lambda m: ' ' if NOISE.search(m.group()) else m.group(), t)
t = NOISE.sub(' ', t)
t = re.sub(r'\(\s*\)|\[\s*\]', ' ', t)
return re.sub(r'\s{2,}', ' ', t).strip(' -–—,')
def clean_artist(raw):
a = re.sub(r'\s*-\s*Topic$', '', str(raw or ''), flags=re.I)
a = re.sub(r'VEVO$', '', a, flags=re.I)
return re.sub(r'\s{2,}', ' ', NOISE.sub(' ', a)).strip()
def parse_lrc(text):
out = []
for raw in str(text or '').replace('\r', '').split('\n'):
rest = raw.strip()
if not rest or re.fullmatch(r'\[[a-z]+:[^\]]*\]', rest, re.I):
continue
stamps = []
while True:
m = re.match(r'^\[(\d{1,3}):(\d{1,2})(?:[.:](\d{1,3}))?\]', rest)
if not m:
break
stamps.append(int(m.group(1)) * 60 + int(m.group(2)) + (float('0.' + m.group(3)) if m.group(3) else 0))
rest = rest[m.end():].strip()
if not rest:
continue
if not stamps:
out.append({'t': None, 'text': rest[:300], 'kind': 'line'})
for t in stamps:
out.append({'t': round(t, 2), 'text': rest[:300], 'kind': 'line'})
if out and all(l['t'] is not None for l in out):
out.sort(key=lambda l: l['t'])
return out
def norm(s):
return re.sub(r'\s+', ' ', re.sub(r'[^a-z0-9 ]+', ' ', str(s or '').lower())).strip()
def http_json(url, tries=4):
"""LRCLIB is free and rate-limits bursts with 503/429 — back off and retry."""
for attempt in range(tries):
try:
with urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=30) as r:
return json.loads(r.read() or b'null')
except urllib.error.HTTPError as e:
if e.code == 404:
return None
if e.code in (429, 503) and attempt < tries - 1:
time.sleep(2 * (attempt + 1))
continue
raise
except Exception:
if attempt < tries - 1:
time.sleep(2 * (attempt + 1))
continue
raise
return None
# Words that say nothing about WHO recorded a song. A lyric-video channel
# called "Christian Lyrics" otherwise "verifies" any track featuring someone
# named Christian — which is exactly how "Still" matched Nicky Romero.
GENERIC = {
'the', 'and', 'of', 'a', 'feat', 'featuring', 'ft', 'with', 'band', 'music', 'musica',
'worship', 'ministries', 'ministry', 'christian', 'gospel', 'praise', 'church', 'choir',
'lyrics', 'lyric', 'live', 'official', 'records', 'recordings', 'group', 'project', 'team',
}
def artist_ok(lrc_artist, channel, video_title):
"""True when the LRCLIB artist plausibly matches the video.
Lyric-video channels ("Christian Lyrics", a person's name) carry no artist,
so the artist is also looked for in the video title. Songs whose title is
shared across genres ("Still") otherwise match the wrong recording."""
a = set(norm(lrc_artist).split()) - GENERIC
if not a:
return False
hay = set(norm(channel).split()) | set(norm(video_title).split())
return bool(a & hay)
def title_run(track, video_title):
"""True when LRCLIB's track name is the video's title, allowing the video
to carry extra words around it ("Lakewood Live - Holy You Are").
Compared word by word, never as a substring: "Still" IS a substring of
"(You Can Still) Rock in America", and that is exactly how a one-word
title matches the wrong song."""
a, b = norm(track).split(), norm(video_title).split()
if not a or not b:
return False
if a == b:
return True
if len(a) < 2:
return False # one word matches by accident far too often
return any(b[i:i + len(a)] == a for i in range(len(b) - len(a) + 1))
def lrclib_lookup(title, artist, duration, tolerance=6, verify=None):
"""Best LRCLIB entry for a song, or None. Prefers synced lyrics, and
strongly prefers a candidate whose artist verifies against the video —
"Still" returns both Hillsong Worship and Night Ranger."""
q = {'track_name': title, 'artist_name': artist or ''}
if duration:
q['duration'] = str(int(round(duration)))
hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q))
if not hit:
# The "artist" is really the YouTube channel ("Integrity Worship",
# "Christian Lyrics"), so a title+artist search often finds nothing
# where a title-only one finds the song. Try both, widest last.
results = []
for query in ([f'{title} {artist}'.strip(), title] if artist else [title]):
results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': query})) or []
if results:
break
time.sleep(0.4)
want = norm(title)
scored = []
for x in results:
if x.get('instrumental') or not (x.get('syncedLyrics') or x.get('plainLyrics')):
continue
t = norm(x.get('trackName'))
dd = abs((x.get('duration') or 0) - duration) if duration else 99
title_hit = 2 if t == want else 1 if (want in t or t in want) else 0
if not title_hit or (duration and dd > tolerance):
continue
ok = 50 if (verify and verify(x.get('artistName'))) else 0
scored.append((ok + title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x))
if not scored:
return None
hit = max(scored, key=lambda p: p[0])[1]
lines = parse_lrc(hit.get('syncedLyrics') or '')
synced = bool(lines)
if not lines:
lines = parse_lrc(hit.get('plainLyrics') or '')
return {'lines': lines, 'synced': synced, 'hit': hit} if lines else None
def describe(hit, cur_lines):
h = hit['hit']
return f'{"synced" if hit["synced"] else "plain "} | {h.get("artistName", "")[:22]:22} – {h.get("trackName", "")[:26]:26} | {len(hit["lines"])} lines (was {cur_lines})'
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
ap.add_argument('--ids', help='comma-separated video ids')
ap.add_argument('--since', help='only songs whose lyrics were last saved after this local time (YYYY-MM-DD HH:MM)')
ap.add_argument('--until', help='…and before this one')
ap.add_argument('--tagged', help='only songs whose lyrics carry this tag, e.g. auto-transcribed')
ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', ''), help='where to write the backup JSON (default: Windows Documents on WSL, else ~/)')
ap.add_argument('--apply', action='store_true', help='actually write the new lyrics (default: dry run)')
ap.add_argument('--tolerance', type=float, default=6, help='max duration difference in seconds')
ap.add_argument('--loose', action='store_true', help="accept matches whose artist doesn't line up with the video (risky)")
args = ap.parse_args()
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
# Which songs? Explicit ids, or every song the server has lyrics for,
# filtered by when they were saved and/or their tag.
ids = [x.strip() for x in (args.ids or '').split(',') if x.strip()]
if not ids:
st, r = api.call('GET', '/api/admin/media')
if st != 200:
sys.exit(f'need --ids (listing needs the newer server build: {r.get("error")})')
ids = [m['id'] for m in r['media'] if m.get('lyricsLines')]
to_ts = lambda s: dt.datetime.strptime(s, '%Y-%m-%d %H:%M').timestamp() if s else None
since, until = to_ts(args.since), to_ts(args.until)
picked, backup = [], {}
for vid in ids:
st, n = api.call('GET', f'/api/notes/{vid}')
cur = (n or {}).get('lyrics') if st == 200 else None
if not cur:
continue
when = cur['updatedAt']
if (since and when < since) or (until and when > until):
continue
if args.tagged and not any(args.tagged.lower() in t.lower() for t in cur['data'].get('tags', [])):
continue
st, sres = api.call('GET', f'/api/streams?v={vid}')
meta = ((sres or {}).get('data') or {}).get('meta') or {}
picked.append({'id': vid, 'cur': cur, 'meta': meta})
backup[vid] = {'savedAt': when, 'rev': cur['rev'], 'title': meta.get('title', ''), 'data': cur['data']}
if not picked:
sys.exit('nothing matched')
# 1) Backup first — always, even on a dry run.
out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~'))
stamp = dt.datetime.now().strftime('%Y%m%d-%H%M%S')
path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{stamp}.json')
with open(path, 'w', encoding='utf-8') as f:
json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base, 'songs': backup}, f, ensure_ascii=False, indent=1)
print(f'backup: {len(backup)} songs → {path}\n')
# 2) Look each one up and (optionally) replace.
changed = skipped = 0
for p in picked:
vid, meta, cur = p['id'], p['meta'], p['cur']
title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel'))
dur = float(meta.get('duration') or 0)
verify = lambda a: artist_ok(a, meta.get('channel'), meta.get('title'))
try:
hit = lrclib_lookup(title, artist, dur, args.tolerance, verify)
except Exception as e: # network hiccup — keep going
print(f'{vid} lookup failed: {e}')
continue
time.sleep(0.4) # be polite to a free service
if not hit:
skipped += 1
print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})')
continue
h = hit['hit']
sure = verify(h.get('artistName'))
mark = 'LRCLIB' if sure else 'UNSURE'
# Worship uploads are often credited to a lyric-video channel, so the
# artist can't be checked. The same song title at the same length to
# within 3 s is evidence in its own right — a different recording of a
# same-named song is essentially never that close.
dd = abs((h.get('duration') or 0) - dur) if dur else 99
if not sure and dd <= 3 and title_run(h.get('trackName'), meta.get('title')):
sure, mark = True, 'LENGTH'
print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}' + (f' | Δ{dd:.1f}s' if dd < 99 else ''))
if not sure and not args.loose:
skipped += 1
print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title and the length differs by {dd:.0f}s — left alone (use --loose to accept)')
continue
if not args.apply:
continue
doc = {
'lines': hit['lines'],
'tags': [f'from LRCLIB ({"synced" if hit["synced"] else "plain text"})'],
'offset': 0,
}
st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': cur['rev']})
if st == 200:
changed += 1
else:
print(f' write failed {st}: {rr.get("error")}')
print(f'\n{changed} replaced, {skipped} left alone' + ('' if args.apply else ' (dry run — add --apply)'))
if __name__ == '__main__':
main()