Add a scrubbable timed lyric editor to the admin page, LRCLIB regeneration tooling with lyrics skills, and keep the five playback buttons on one row on phones

This commit is contained in:
Jonathan Sykes
2026-09-20 17:55:49 +08:00
parent b983c6ca3c
commit 2987f46037
8 changed files with 919 additions and 3 deletions

View File

@@ -0,0 +1,242 @@
#!/usr/bin/env python3
"""lrclib_regen.py — back up and regenerate lyrics on worship.hesed.sbs from LRCLIB.
Machine transcripts (Whisper/Scribe) mishear words and drift; published lyrics
on LRCLIB are usually correct and often SYNCED. This tool:
1. writes a JSON backup of the CURRENT lyrics of every song it will touch
(the server also keeps each previous version as a revision — nothing is
ever lost, this file is just an offline copy),
2. looks each song up on LRCLIB by cleaned title + artist + duration,
3. replaces the lyrics only when a confident match is found.
Song metadata (title / artist / duration) comes from the public
/api/streams endpoint, so this works without admin access; writing needs
YTP_ADMIN_PASSWORD or YTP_TOKEN.
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --ids A,B --dry-run
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --since '2026-09-19 01:45' --until '2026-09-19 02:00'
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --tagged auto-transcribed --apply
"""
import argparse
import datetime as dt
import json
import os
import re
import sys
import time
import urllib.parse
import urllib.request
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from auto_lyrics import Api # noqa: E402
LRCLIB = 'https://lrclib.net/api'
UA = {'User-Agent': 'ytplayer-lyrics-regen (https://worship.hesed.sbs)'}
NOISE = re.compile(r'\b(official|video|audio|lyrics?|lyric|hd|hq|4k|live|mv|karaoke|minus[\s-]?one|instrumental|cover|remaster(ed)?|visualizer|performance|version)\b', re.I)
def clean_title(raw):
t = str(raw or '').split('|')[0]
t = re.sub(r'\([^)]*\)|\[[^\]]*\]', lambda m: ' ' if NOISE.search(m.group()) else m.group(), t)
t = NOISE.sub(' ', t)
t = re.sub(r'\(\s*\)|\[\s*\]', ' ', t)
return re.sub(r'\s{2,}', ' ', t).strip(' -–—,')
def clean_artist(raw):
a = re.sub(r'\s*-\s*Topic$', '', str(raw or ''), flags=re.I)
a = re.sub(r'VEVO$', '', a, flags=re.I)
return re.sub(r'\s{2,}', ' ', NOISE.sub(' ', a)).strip()
def parse_lrc(text):
out = []
for raw in str(text or '').replace('\r', '').split('\n'):
rest = raw.strip()
if not rest or re.fullmatch(r'\[[a-z]+:[^\]]*\]', rest, re.I):
continue
stamps = []
while True:
m = re.match(r'^\[(\d{1,3}):(\d{1,2})(?:[.:](\d{1,3}))?\]', rest)
if not m:
break
stamps.append(int(m.group(1)) * 60 + int(m.group(2)) + (float('0.' + m.group(3)) if m.group(3) else 0))
rest = rest[m.end():].strip()
if not rest:
continue
if not stamps:
out.append({'t': None, 'text': rest[:300], 'kind': 'line'})
for t in stamps:
out.append({'t': round(t, 2), 'text': rest[:300], 'kind': 'line'})
if out and all(l['t'] is not None for l in out):
out.sort(key=lambda l: l['t'])
return out
def norm(s):
return re.sub(r'\s+', ' ', re.sub(r'[^a-z0-9 ]+', ' ', str(s or '').lower())).strip()
def http_json(url, tries=4):
"""LRCLIB is free and rate-limits bursts with 503/429 — back off and retry."""
for attempt in range(tries):
try:
with urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=30) as r:
return json.loads(r.read() or b'null')
except urllib.error.HTTPError as e:
if e.code == 404:
return None
if e.code in (429, 503) and attempt < tries - 1:
time.sleep(2 * (attempt + 1))
continue
raise
except Exception:
if attempt < tries - 1:
time.sleep(2 * (attempt + 1))
continue
raise
return None
def artist_ok(lrc_artist, channel, video_title):
"""True when the LRCLIB artist plausibly matches the video.
Lyric-video channels ("Christian Lyrics", a person's name) carry no artist,
so the artist is also looked for in the video title. Songs whose title is
shared across genres ("Still") otherwise match the wrong recording."""
a = set(norm(lrc_artist).split()) - {'the', 'and', 'of', 'band', 'music', 'worship', 'ministries'}
if not a:
return False
hay = set(norm(channel).split()) | set(norm(video_title).split())
return bool(a & hay)
def lrclib_lookup(title, artist, duration, tolerance=6):
"""Best LRCLIB entry for a song, or None. Prefers synced lyrics."""
q = {'track_name': title, 'artist_name': artist or ''}
if duration:
q['duration'] = str(int(round(duration)))
hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q))
if not hit:
results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': f'{title} {artist}'.strip()})) or []
want = norm(title)
scored = []
for x in results:
if x.get('instrumental') or not (x.get('syncedLyrics') or x.get('plainLyrics')):
continue
t = norm(x.get('trackName'))
dd = abs((x.get('duration') or 0) - duration) if duration else 99
title_hit = 2 if t == want else 1 if (want in t or t in want) else 0
if not title_hit or (duration and dd > tolerance):
continue
scored.append((title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x))
if not scored:
return None
hit = max(scored, key=lambda p: p[0])[1]
lines = parse_lrc(hit.get('syncedLyrics') or '')
synced = bool(lines)
if not lines:
lines = parse_lrc(hit.get('plainLyrics') or '')
return {'lines': lines, 'synced': synced, 'hit': hit} if lines else None
def describe(hit, cur_lines):
h = hit['hit']
return f'{"synced" if hit["synced"] else "plain "} | {h.get("artistName", "")[:22]:22} – {h.get("trackName", "")[:26]:26} | {len(hit["lines"])} lines (was {cur_lines})'
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
ap.add_argument('--ids', help='comma-separated video ids')
ap.add_argument('--since', help='only songs whose lyrics were last saved after this local time (YYYY-MM-DD HH:MM)')
ap.add_argument('--until', help='…and before this one')
ap.add_argument('--tagged', help='only songs whose lyrics carry this tag, e.g. auto-transcribed')
ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', ''), help='where to write the backup JSON (default: Windows Documents on WSL, else ~/)')
ap.add_argument('--apply', action='store_true', help='actually write the new lyrics (default: dry run)')
ap.add_argument('--tolerance', type=float, default=6, help='max duration difference in seconds')
ap.add_argument('--loose', action='store_true', help="accept matches whose artist doesn't line up with the video (risky)")
args = ap.parse_args()
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
# Which songs? Explicit ids, or every song the server has lyrics for,
# filtered by when they were saved and/or their tag.
ids = [x.strip() for x in (args.ids or '').split(',') if x.strip()]
if not ids:
st, r = api.call('GET', '/api/admin/media')
if st != 200:
sys.exit(f'need --ids (listing needs the newer server build: {r.get("error")})')
ids = [m['id'] for m in r['media'] if m.get('lyricsLines')]
to_ts = lambda s: dt.datetime.strptime(s, '%Y-%m-%d %H:%M').timestamp() if s else None
since, until = to_ts(args.since), to_ts(args.until)
picked, backup = [], {}
for vid in ids:
st, n = api.call('GET', f'/api/notes/{vid}')
cur = (n or {}).get('lyrics') if st == 200 else None
if not cur:
continue
when = cur['updatedAt']
if (since and when < since) or (until and when > until):
continue
if args.tagged and not any(args.tagged.lower() in t.lower() for t in cur['data'].get('tags', [])):
continue
st, sres = api.call('GET', f'/api/streams?v={vid}')
meta = ((sres or {}).get('data') or {}).get('meta') or {}
picked.append({'id': vid, 'cur': cur, 'meta': meta})
backup[vid] = {'savedAt': when, 'rev': cur['rev'], 'title': meta.get('title', ''), 'data': cur['data']}
if not picked:
sys.exit('nothing matched')
# 1) Backup first — always, even on a dry run.
out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~'))
stamp = dt.datetime.now().strftime('%Y%m%d-%H%M%S')
path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{stamp}.json')
with open(path, 'w', encoding='utf-8') as f:
json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base, 'songs': backup}, f, ensure_ascii=False, indent=1)
print(f'backup: {len(backup)} songs → {path}\n')
# 2) Look each one up and (optionally) replace.
changed = skipped = 0
for p in picked:
vid, meta, cur = p['id'], p['meta'], p['cur']
title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel'))
dur = float(meta.get('duration') or 0)
try:
hit = lrclib_lookup(title, artist, dur, args.tolerance)
except Exception as e: # network hiccup — keep going
print(f'{vid} lookup failed: {e}')
continue
time.sleep(0.4) # be polite to a free service
if not hit:
skipped += 1
print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})')
continue
h = hit['hit']
sure = artist_ok(h.get('artistName'), meta.get('channel'), meta.get('title'))
mark = 'LRCLIB' if sure else 'UNSURE'
print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}')
if not sure and not args.loose:
skipped += 1
print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title — left alone (use --loose to accept)')
continue
if not args.apply:
continue
doc = {
'lines': hit['lines'],
'tags': [f'from LRCLIB ({"synced" if hit["synced"] else "plain text"})'],
'offset': 0,
}
st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': cur['rev']})
if st == 200:
changed += 1
else:
print(f' write failed {st}: {rr.get("error")}')
print(f'\n{changed} replaced, {skipped} left alone' + ('' if args.apply else ' (dry run — add --apply)'))
if __name__ == '__main__':
main()