Add a scrubbable timed lyric editor to the admin page, LRCLIB regeneration tooling with lyrics skills, and keep the five playback buttons on one row on phones
This commit is contained in:
242
scripts/lyrics/lrclib_regen.py
Normal file
242
scripts/lyrics/lrclib_regen.py
Normal file
@@ -0,0 +1,242 @@
|
||||
#!/usr/bin/env python3
|
||||
"""lrclib_regen.py — back up and regenerate lyrics on worship.hesed.sbs from LRCLIB.
|
||||
|
||||
Machine transcripts (Whisper/Scribe) mishear words and drift; published lyrics
|
||||
on LRCLIB are usually correct and often SYNCED. This tool:
|
||||
|
||||
1. writes a JSON backup of the CURRENT lyrics of every song it will touch
|
||||
(the server also keeps each previous version as a revision — nothing is
|
||||
ever lost, this file is just an offline copy),
|
||||
2. looks each song up on LRCLIB by cleaned title + artist + duration,
|
||||
3. replaces the lyrics only when a confident match is found.
|
||||
|
||||
Song metadata (title / artist / duration) comes from the public
|
||||
/api/streams endpoint, so this works without admin access; writing needs
|
||||
YTP_ADMIN_PASSWORD or YTP_TOKEN.
|
||||
|
||||
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --ids A,B --dry-run
|
||||
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --since '2026-09-19 01:45' --until '2026-09-19 02:00'
|
||||
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --tagged auto-transcribed --apply
|
||||
"""
|
||||
import argparse
|
||||
import datetime as dt
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from auto_lyrics import Api # noqa: E402
|
||||
|
||||
LRCLIB = 'https://lrclib.net/api'
|
||||
UA = {'User-Agent': 'ytplayer-lyrics-regen (https://worship.hesed.sbs)'}
|
||||
NOISE = re.compile(r'\b(official|video|audio|lyrics?|lyric|hd|hq|4k|live|mv|karaoke|minus[\s-]?one|instrumental|cover|remaster(ed)?|visualizer|performance|version)\b', re.I)
|
||||
|
||||
|
||||
def clean_title(raw):
|
||||
t = str(raw or '').split('|')[0]
|
||||
t = re.sub(r'\([^)]*\)|\[[^\]]*\]', lambda m: ' ' if NOISE.search(m.group()) else m.group(), t)
|
||||
t = NOISE.sub(' ', t)
|
||||
t = re.sub(r'\(\s*\)|\[\s*\]', ' ', t)
|
||||
return re.sub(r'\s{2,}', ' ', t).strip(' -–—,')
|
||||
|
||||
|
||||
def clean_artist(raw):
|
||||
a = re.sub(r'\s*-\s*Topic$', '', str(raw or ''), flags=re.I)
|
||||
a = re.sub(r'VEVO$', '', a, flags=re.I)
|
||||
return re.sub(r'\s{2,}', ' ', NOISE.sub(' ', a)).strip()
|
||||
|
||||
|
||||
def parse_lrc(text):
|
||||
out = []
|
||||
for raw in str(text or '').replace('\r', '').split('\n'):
|
||||
rest = raw.strip()
|
||||
if not rest or re.fullmatch(r'\[[a-z]+:[^\]]*\]', rest, re.I):
|
||||
continue
|
||||
stamps = []
|
||||
while True:
|
||||
m = re.match(r'^\[(\d{1,3}):(\d{1,2})(?:[.:](\d{1,3}))?\]', rest)
|
||||
if not m:
|
||||
break
|
||||
stamps.append(int(m.group(1)) * 60 + int(m.group(2)) + (float('0.' + m.group(3)) if m.group(3) else 0))
|
||||
rest = rest[m.end():].strip()
|
||||
if not rest:
|
||||
continue
|
||||
if not stamps:
|
||||
out.append({'t': None, 'text': rest[:300], 'kind': 'line'})
|
||||
for t in stamps:
|
||||
out.append({'t': round(t, 2), 'text': rest[:300], 'kind': 'line'})
|
||||
if out and all(l['t'] is not None for l in out):
|
||||
out.sort(key=lambda l: l['t'])
|
||||
return out
|
||||
|
||||
|
||||
def norm(s):
|
||||
return re.sub(r'\s+', ' ', re.sub(r'[^a-z0-9 ]+', ' ', str(s or '').lower())).strip()
|
||||
|
||||
|
||||
def http_json(url, tries=4):
|
||||
"""LRCLIB is free and rate-limits bursts with 503/429 — back off and retry."""
|
||||
for attempt in range(tries):
|
||||
try:
|
||||
with urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=30) as r:
|
||||
return json.loads(r.read() or b'null')
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code == 404:
|
||||
return None
|
||||
if e.code in (429, 503) and attempt < tries - 1:
|
||||
time.sleep(2 * (attempt + 1))
|
||||
continue
|
||||
raise
|
||||
except Exception:
|
||||
if attempt < tries - 1:
|
||||
time.sleep(2 * (attempt + 1))
|
||||
continue
|
||||
raise
|
||||
return None
|
||||
|
||||
|
||||
def artist_ok(lrc_artist, channel, video_title):
|
||||
"""True when the LRCLIB artist plausibly matches the video.
|
||||
|
||||
Lyric-video channels ("Christian Lyrics", a person's name) carry no artist,
|
||||
so the artist is also looked for in the video title. Songs whose title is
|
||||
shared across genres ("Still") otherwise match the wrong recording."""
|
||||
a = set(norm(lrc_artist).split()) - {'the', 'and', 'of', 'band', 'music', 'worship', 'ministries'}
|
||||
if not a:
|
||||
return False
|
||||
hay = set(norm(channel).split()) | set(norm(video_title).split())
|
||||
return bool(a & hay)
|
||||
|
||||
|
||||
def lrclib_lookup(title, artist, duration, tolerance=6):
|
||||
"""Best LRCLIB entry for a song, or None. Prefers synced lyrics."""
|
||||
q = {'track_name': title, 'artist_name': artist or ''}
|
||||
if duration:
|
||||
q['duration'] = str(int(round(duration)))
|
||||
hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q))
|
||||
if not hit:
|
||||
results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': f'{title} {artist}'.strip()})) or []
|
||||
want = norm(title)
|
||||
scored = []
|
||||
for x in results:
|
||||
if x.get('instrumental') or not (x.get('syncedLyrics') or x.get('plainLyrics')):
|
||||
continue
|
||||
t = norm(x.get('trackName'))
|
||||
dd = abs((x.get('duration') or 0) - duration) if duration else 99
|
||||
title_hit = 2 if t == want else 1 if (want in t or t in want) else 0
|
||||
if not title_hit or (duration and dd > tolerance):
|
||||
continue
|
||||
scored.append((title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x))
|
||||
if not scored:
|
||||
return None
|
||||
hit = max(scored, key=lambda p: p[0])[1]
|
||||
lines = parse_lrc(hit.get('syncedLyrics') or '')
|
||||
synced = bool(lines)
|
||||
if not lines:
|
||||
lines = parse_lrc(hit.get('plainLyrics') or '')
|
||||
return {'lines': lines, 'synced': synced, 'hit': hit} if lines else None
|
||||
|
||||
|
||||
def describe(hit, cur_lines):
|
||||
h = hit['hit']
|
||||
return f'{"synced" if hit["synced"] else "plain "} | {h.get("artistName", "")[:22]:22} – {h.get("trackName", "")[:26]:26} | {len(hit["lines"])} lines (was {cur_lines})'
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
|
||||
ap.add_argument('--ids', help='comma-separated video ids')
|
||||
ap.add_argument('--since', help='only songs whose lyrics were last saved after this local time (YYYY-MM-DD HH:MM)')
|
||||
ap.add_argument('--until', help='…and before this one')
|
||||
ap.add_argument('--tagged', help='only songs whose lyrics carry this tag, e.g. auto-transcribed')
|
||||
ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', ''), help='where to write the backup JSON (default: Windows Documents on WSL, else ~/)')
|
||||
ap.add_argument('--apply', action='store_true', help='actually write the new lyrics (default: dry run)')
|
||||
ap.add_argument('--tolerance', type=float, default=6, help='max duration difference in seconds')
|
||||
ap.add_argument('--loose', action='store_true', help="accept matches whose artist doesn't line up with the video (risky)")
|
||||
args = ap.parse_args()
|
||||
|
||||
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
|
||||
|
||||
# Which songs? Explicit ids, or every song the server has lyrics for,
|
||||
# filtered by when they were saved and/or their tag.
|
||||
ids = [x.strip() for x in (args.ids or '').split(',') if x.strip()]
|
||||
if not ids:
|
||||
st, r = api.call('GET', '/api/admin/media')
|
||||
if st != 200:
|
||||
sys.exit(f'need --ids (listing needs the newer server build: {r.get("error")})')
|
||||
ids = [m['id'] for m in r['media'] if m.get('lyricsLines')]
|
||||
to_ts = lambda s: dt.datetime.strptime(s, '%Y-%m-%d %H:%M').timestamp() if s else None
|
||||
since, until = to_ts(args.since), to_ts(args.until)
|
||||
|
||||
picked, backup = [], {}
|
||||
for vid in ids:
|
||||
st, n = api.call('GET', f'/api/notes/{vid}')
|
||||
cur = (n or {}).get('lyrics') if st == 200 else None
|
||||
if not cur:
|
||||
continue
|
||||
when = cur['updatedAt']
|
||||
if (since and when < since) or (until and when > until):
|
||||
continue
|
||||
if args.tagged and not any(args.tagged.lower() in t.lower() for t in cur['data'].get('tags', [])):
|
||||
continue
|
||||
st, sres = api.call('GET', f'/api/streams?v={vid}')
|
||||
meta = ((sres or {}).get('data') or {}).get('meta') or {}
|
||||
picked.append({'id': vid, 'cur': cur, 'meta': meta})
|
||||
backup[vid] = {'savedAt': when, 'rev': cur['rev'], 'title': meta.get('title', ''), 'data': cur['data']}
|
||||
|
||||
if not picked:
|
||||
sys.exit('nothing matched')
|
||||
|
||||
# 1) Backup first — always, even on a dry run.
|
||||
out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~'))
|
||||
stamp = dt.datetime.now().strftime('%Y%m%d-%H%M%S')
|
||||
path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{stamp}.json')
|
||||
with open(path, 'w', encoding='utf-8') as f:
|
||||
json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base, 'songs': backup}, f, ensure_ascii=False, indent=1)
|
||||
print(f'backup: {len(backup)} songs → {path}\n')
|
||||
|
||||
# 2) Look each one up and (optionally) replace.
|
||||
changed = skipped = 0
|
||||
for p in picked:
|
||||
vid, meta, cur = p['id'], p['meta'], p['cur']
|
||||
title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel'))
|
||||
dur = float(meta.get('duration') or 0)
|
||||
try:
|
||||
hit = lrclib_lookup(title, artist, dur, args.tolerance)
|
||||
except Exception as e: # network hiccup — keep going
|
||||
print(f'{vid} lookup failed: {e}')
|
||||
continue
|
||||
time.sleep(0.4) # be polite to a free service
|
||||
if not hit:
|
||||
skipped += 1
|
||||
print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})')
|
||||
continue
|
||||
h = hit['hit']
|
||||
sure = artist_ok(h.get('artistName'), meta.get('channel'), meta.get('title'))
|
||||
mark = 'LRCLIB' if sure else 'UNSURE'
|
||||
print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}')
|
||||
if not sure and not args.loose:
|
||||
skipped += 1
|
||||
print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title — left alone (use --loose to accept)')
|
||||
continue
|
||||
if not args.apply:
|
||||
continue
|
||||
doc = {
|
||||
'lines': hit['lines'],
|
||||
'tags': [f'from LRCLIB ({"synced" if hit["synced"] else "plain text"})'],
|
||||
'offset': 0,
|
||||
}
|
||||
st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': cur['rev']})
|
||||
if st == 200:
|
||||
changed += 1
|
||||
else:
|
||||
print(f' write failed {st}: {rr.get("error")}')
|
||||
print(f'\n{changed} replaced, {skipped} left alone' + ('' if args.apply else ' (dry run — add --apply)'))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in New Issue
Block a user