289 lines
13 KiB
Python
289 lines
13 KiB
Python
#!/usr/bin/env python3
|
||
"""lrclib_regen.py — back up and regenerate lyrics on worship.hesed.sbs from LRCLIB.
|
||
|
||
Machine transcripts (Whisper/Scribe) mishear words and drift; published lyrics
|
||
on LRCLIB are usually correct and often SYNCED. This tool:
|
||
|
||
1. writes a JSON backup of the CURRENT lyrics of every song it will touch
|
||
(the server also keeps each previous version as a revision — nothing is
|
||
ever lost, this file is just an offline copy),
|
||
2. looks each song up on LRCLIB by cleaned title + artist + duration,
|
||
3. replaces the lyrics only when a confident match is found.
|
||
|
||
Song metadata (title / artist / duration) comes from the public
|
||
/api/streams endpoint, so this works without admin access; writing needs
|
||
YTP_ADMIN_PASSWORD or YTP_TOKEN.
|
||
|
||
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --ids A,B --dry-run
|
||
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --since '2026-09-19 01:45' --until '2026-09-19 02:00'
|
||
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --tagged auto-transcribed --apply
|
||
"""
|
||
import argparse
|
||
import datetime as dt
|
||
import json
|
||
import os
|
||
import re
|
||
import sys
|
||
import time
|
||
import urllib.parse
|
||
import urllib.request
|
||
|
||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||
from auto_lyrics import Api # noqa: E402
|
||
|
||
LRCLIB = 'https://lrclib.net/api'
|
||
UA = {'User-Agent': 'ytplayer-lyrics-regen (https://worship.hesed.sbs)'}
|
||
NOISE = re.compile(r'\b(official|video|audio|lyrics?|lyric|hd|hq|4k|live|mv|karaoke|minus[\s-]?one|instrumental|cover|remaster(ed)?|visualizer|performance|version)\b', re.I)
|
||
|
||
|
||
def clean_title(raw):
|
||
t = str(raw or '').split('|')[0]
|
||
t = re.sub(r'\([^)]*\)|\[[^\]]*\]', lambda m: ' ' if NOISE.search(m.group()) else m.group(), t)
|
||
t = NOISE.sub(' ', t)
|
||
t = re.sub(r'\(\s*\)|\[\s*\]', ' ', t)
|
||
return re.sub(r'\s{2,}', ' ', t).strip(' -–—,')
|
||
|
||
|
||
def clean_artist(raw):
|
||
a = re.sub(r'\s*-\s*Topic$', '', str(raw or ''), flags=re.I)
|
||
a = re.sub(r'VEVO$', '', a, flags=re.I)
|
||
return re.sub(r'\s{2,}', ' ', NOISE.sub(' ', a)).strip()
|
||
|
||
|
||
def parse_lrc(text):
|
||
out = []
|
||
for raw in str(text or '').replace('\r', '').split('\n'):
|
||
rest = raw.strip()
|
||
if not rest or re.fullmatch(r'\[[a-z]+:[^\]]*\]', rest, re.I):
|
||
continue
|
||
stamps = []
|
||
while True:
|
||
m = re.match(r'^\[(\d{1,3}):(\d{1,2})(?:[.:](\d{1,3}))?\]', rest)
|
||
if not m:
|
||
break
|
||
stamps.append(int(m.group(1)) * 60 + int(m.group(2)) + (float('0.' + m.group(3)) if m.group(3) else 0))
|
||
rest = rest[m.end():].strip()
|
||
if not rest:
|
||
continue
|
||
if not stamps:
|
||
out.append({'t': None, 'text': rest[:300], 'kind': 'line'})
|
||
for t in stamps:
|
||
out.append({'t': round(t, 2), 'text': rest[:300], 'kind': 'line'})
|
||
if out and all(l['t'] is not None for l in out):
|
||
out.sort(key=lambda l: l['t'])
|
||
return out
|
||
|
||
|
||
def norm(s):
|
||
return re.sub(r'\s+', ' ', re.sub(r'[^a-z0-9 ]+', ' ', str(s or '').lower())).strip()
|
||
|
||
|
||
def http_json(url, tries=4):
|
||
"""LRCLIB is free and rate-limits bursts with 503/429 — back off and retry."""
|
||
for attempt in range(tries):
|
||
try:
|
||
with urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=30) as r:
|
||
return json.loads(r.read() or b'null')
|
||
except urllib.error.HTTPError as e:
|
||
if e.code == 404:
|
||
return None
|
||
if e.code in (429, 503) and attempt < tries - 1:
|
||
time.sleep(2 * (attempt + 1))
|
||
continue
|
||
raise
|
||
except Exception:
|
||
if attempt < tries - 1:
|
||
time.sleep(2 * (attempt + 1))
|
||
continue
|
||
raise
|
||
return None
|
||
|
||
|
||
# Words that say nothing about WHO recorded a song. A lyric-video channel
|
||
# called "Christian Lyrics" otherwise "verifies" any track featuring someone
|
||
# named Christian — which is exactly how "Still" matched Nicky Romero.
|
||
GENERIC = {
|
||
'the', 'and', 'of', 'a', 'feat', 'featuring', 'ft', 'with', 'band', 'music', 'musica',
|
||
'worship', 'ministries', 'ministry', 'christian', 'gospel', 'praise', 'church', 'choir',
|
||
'lyrics', 'lyric', 'live', 'official', 'records', 'recordings', 'group', 'project', 'team',
|
||
}
|
||
|
||
|
||
def artist_ok(lrc_artist, channel, video_title):
|
||
"""True when the LRCLIB artist plausibly matches the video.
|
||
|
||
Lyric-video channels ("Christian Lyrics", a person's name) carry no artist,
|
||
so the artist is also looked for in the video title. Songs whose title is
|
||
shared across genres ("Still") otherwise match the wrong recording."""
|
||
a = set(norm(lrc_artist).split()) - GENERIC
|
||
if not a:
|
||
return False
|
||
hay = set(norm(channel).split()) | set(norm(video_title).split())
|
||
return bool(a & hay)
|
||
|
||
|
||
def title_run(track, video_title):
|
||
"""True when LRCLIB's track name is the video's title, allowing the video
|
||
to carry extra words around it ("Lakewood Live - Holy You Are").
|
||
|
||
Compared word by word, never as a substring: "Still" IS a substring of
|
||
"(You Can Still) Rock in America", and that is exactly how a one-word
|
||
title matches the wrong song."""
|
||
a, b = norm(track).split(), norm(video_title).split()
|
||
if not a or not b:
|
||
return False
|
||
if a == b:
|
||
return True
|
||
if len(a) < 2:
|
||
return False # one word matches by accident far too often
|
||
return any(b[i:i + len(a)] == a for i in range(len(b) - len(a) + 1))
|
||
|
||
|
||
def lrclib_lookup(title, artist, duration, tolerance=6, verify=None):
|
||
"""Best LRCLIB entry for a song, or None. Prefers synced lyrics, and
|
||
strongly prefers a candidate whose artist verifies against the video —
|
||
"Still" returns both Hillsong Worship and Night Ranger."""
|
||
q = {'track_name': title, 'artist_name': artist or ''}
|
||
if duration:
|
||
q['duration'] = str(int(round(duration)))
|
||
hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q))
|
||
if not hit:
|
||
# The "artist" is really the YouTube channel ("Integrity Worship",
|
||
# "Christian Lyrics"), so a title+artist search often finds nothing
|
||
# where a title-only one finds the song. Try both, widest last.
|
||
results = []
|
||
for query in ([f'{title} {artist}'.strip(), title] if artist else [title]):
|
||
results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': query})) or []
|
||
if results:
|
||
break
|
||
time.sleep(0.4)
|
||
want = norm(title)
|
||
scored = []
|
||
for x in results:
|
||
if x.get('instrumental') or not (x.get('syncedLyrics') or x.get('plainLyrics')):
|
||
continue
|
||
t = norm(x.get('trackName'))
|
||
dd = abs((x.get('duration') or 0) - duration) if duration else 99
|
||
title_hit = 2 if t == want else 1 if (want in t or t in want) else 0
|
||
if not title_hit or (duration and dd > tolerance):
|
||
continue
|
||
ok = 50 if (verify and verify(x.get('artistName'))) else 0
|
||
scored.append((ok + title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x))
|
||
if not scored:
|
||
return None
|
||
hit = max(scored, key=lambda p: p[0])[1]
|
||
lines = parse_lrc(hit.get('syncedLyrics') or '')
|
||
synced = bool(lines)
|
||
if not lines:
|
||
lines = parse_lrc(hit.get('plainLyrics') or '')
|
||
return {'lines': lines, 'synced': synced, 'hit': hit} if lines else None
|
||
|
||
|
||
def describe(hit, cur_lines):
|
||
h = hit['hit']
|
||
return f'{"synced" if hit["synced"] else "plain "} | {h.get("artistName", "")[:22]:22} – {h.get("trackName", "")[:26]:26} | {len(hit["lines"])} lines (was {cur_lines})'
|
||
|
||
|
||
def main():
|
||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
|
||
ap.add_argument('--ids', help='comma-separated video ids')
|
||
ap.add_argument('--since', help='only songs whose lyrics were last saved after this local time (YYYY-MM-DD HH:MM)')
|
||
ap.add_argument('--until', help='…and before this one')
|
||
ap.add_argument('--tagged', help='only songs whose lyrics carry this tag, e.g. auto-transcribed')
|
||
ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', ''), help='where to write the backup JSON (default: Windows Documents on WSL, else ~/)')
|
||
ap.add_argument('--apply', action='store_true', help='actually write the new lyrics (default: dry run)')
|
||
ap.add_argument('--tolerance', type=float, default=6, help='max duration difference in seconds')
|
||
ap.add_argument('--loose', action='store_true', help="accept matches whose artist doesn't line up with the video (risky)")
|
||
args = ap.parse_args()
|
||
|
||
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
|
||
|
||
# Which songs? Explicit ids, or every song the server has lyrics for,
|
||
# filtered by when they were saved and/or their tag.
|
||
ids = [x.strip() for x in (args.ids or '').split(',') if x.strip()]
|
||
if not ids:
|
||
st, r = api.call('GET', '/api/admin/media')
|
||
if st != 200:
|
||
sys.exit(f'need --ids (listing needs the newer server build: {r.get("error")})')
|
||
ids = [m['id'] for m in r['media'] if m.get('lyricsLines')]
|
||
to_ts = lambda s: dt.datetime.strptime(s, '%Y-%m-%d %H:%M').timestamp() if s else None
|
||
since, until = to_ts(args.since), to_ts(args.until)
|
||
|
||
picked, backup = [], {}
|
||
for vid in ids:
|
||
st, n = api.call('GET', f'/api/notes/{vid}')
|
||
cur = (n or {}).get('lyrics') if st == 200 else None
|
||
if not cur:
|
||
continue
|
||
when = cur['updatedAt']
|
||
if (since and when < since) or (until and when > until):
|
||
continue
|
||
if args.tagged and not any(args.tagged.lower() in t.lower() for t in cur['data'].get('tags', [])):
|
||
continue
|
||
st, sres = api.call('GET', f'/api/streams?v={vid}')
|
||
meta = ((sres or {}).get('data') or {}).get('meta') or {}
|
||
picked.append({'id': vid, 'cur': cur, 'meta': meta})
|
||
backup[vid] = {'savedAt': when, 'rev': cur['rev'], 'title': meta.get('title', ''), 'data': cur['data']}
|
||
|
||
if not picked:
|
||
sys.exit('nothing matched')
|
||
|
||
# 1) Backup first — always, even on a dry run.
|
||
out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~'))
|
||
stamp = dt.datetime.now().strftime('%Y%m%d-%H%M%S')
|
||
path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{stamp}.json')
|
||
with open(path, 'w', encoding='utf-8') as f:
|
||
json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base, 'songs': backup}, f, ensure_ascii=False, indent=1)
|
||
print(f'backup: {len(backup)} songs → {path}\n')
|
||
|
||
# 2) Look each one up and (optionally) replace.
|
||
changed = skipped = 0
|
||
for p in picked:
|
||
vid, meta, cur = p['id'], p['meta'], p['cur']
|
||
title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel'))
|
||
dur = float(meta.get('duration') or 0)
|
||
verify = lambda a: artist_ok(a, meta.get('channel'), meta.get('title'))
|
||
try:
|
||
hit = lrclib_lookup(title, artist, dur, args.tolerance, verify)
|
||
except Exception as e: # network hiccup — keep going
|
||
print(f'{vid} lookup failed: {e}')
|
||
continue
|
||
time.sleep(0.4) # be polite to a free service
|
||
if not hit:
|
||
skipped += 1
|
||
print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})')
|
||
continue
|
||
h = hit['hit']
|
||
sure = verify(h.get('artistName'))
|
||
mark = 'LRCLIB' if sure else 'UNSURE'
|
||
# Worship uploads are often credited to a lyric-video channel, so the
|
||
# artist can't be checked. The same song title at the same length to
|
||
# within 3 s is evidence in its own right — a different recording of a
|
||
# same-named song is essentially never that close.
|
||
dd = abs((h.get('duration') or 0) - dur) if dur else 99
|
||
if not sure and dd <= 3 and title_run(h.get('trackName'), meta.get('title')):
|
||
sure, mark = True, 'LENGTH'
|
||
print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}' + (f' | Δ{dd:.1f}s' if dd < 99 else ''))
|
||
if not sure and not args.loose:
|
||
skipped += 1
|
||
print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title and the length differs by {dd:.0f}s — left alone (use --loose to accept)')
|
||
continue
|
||
if not args.apply:
|
||
continue
|
||
doc = {
|
||
'lines': hit['lines'],
|
||
'tags': [f'from LRCLIB ({"synced" if hit["synced"] else "plain text"})'],
|
||
'offset': 0,
|
||
}
|
||
st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': cur['rev']})
|
||
if st == 200:
|
||
changed += 1
|
||
else:
|
||
print(f' write failed {st}: {rr.get("error")}')
|
||
print(f'\n{changed} replaced, {skipped} left alone' + ('' if args.apply else ' (dry run — add --apply)'))
|
||
|
||
|
||
if __name__ == '__main__':
|
||
main()
|