#!/usr/bin/env python3 """lrclib_regen.py — back up and regenerate lyrics on worship.hesed.sbs from LRCLIB. Machine transcripts (Whisper/Scribe) mishear words and drift; published lyrics on LRCLIB are usually correct and often SYNCED. This tool: 1. writes a JSON backup of the CURRENT lyrics of every song it will touch (the server also keeps each previous version as a revision — nothing is ever lost, this file is just an offline copy), 2. looks each song up on LRCLIB by cleaned title + artist + duration, 3. replaces the lyrics only when a confident match is found. Song metadata (title / artist / duration) comes from the public /api/streams endpoint, so this works without admin access; writing needs YTP_ADMIN_PASSWORD or YTP_TOKEN. YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --ids A,B --dry-run YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --since '2026-09-19 01:45' --until '2026-09-19 02:00' YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/lrclib_regen.py --tagged auto-transcribed --apply """ import argparse import datetime as dt import json import os import re import sys import time import urllib.parse import urllib.request sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from auto_lyrics import Api # noqa: E402 LRCLIB = 'https://lrclib.net/api' UA = {'User-Agent': 'ytplayer-lyrics-regen (https://worship.hesed.sbs)'} NOISE = re.compile(r'\b(official|video|audio|lyrics?|lyric|hd|hq|4k|live|mv|karaoke|minus[\s-]?one|instrumental|cover|remaster(ed)?|visualizer|performance|version)\b', re.I) def clean_title(raw): t = str(raw or '').split('|')[0] t = re.sub(r'\([^)]*\)|\[[^\]]*\]', lambda m: ' ' if NOISE.search(m.group()) else m.group(), t) t = NOISE.sub(' ', t) t = re.sub(r'\(\s*\)|\[\s*\]', ' ', t) return re.sub(r'\s{2,}', ' ', t).strip(' -–—,') def clean_artist(raw): a = re.sub(r'\s*-\s*Topic$', '', str(raw or ''), flags=re.I) a = re.sub(r'VEVO$', '', a, flags=re.I) return re.sub(r'\s{2,}', ' ', NOISE.sub(' ', a)).strip() def parse_lrc(text): out = [] for raw in str(text or '').replace('\r', '').split('\n'): rest = raw.strip() if not rest or re.fullmatch(r'\[[a-z]+:[^\]]*\]', rest, re.I): continue stamps = [] while True: m = re.match(r'^\[(\d{1,3}):(\d{1,2})(?:[.:](\d{1,3}))?\]', rest) if not m: break stamps.append(int(m.group(1)) * 60 + int(m.group(2)) + (float('0.' + m.group(3)) if m.group(3) else 0)) rest = rest[m.end():].strip() if not rest: continue if not stamps: out.append({'t': None, 'text': rest[:300], 'kind': 'line'}) for t in stamps: out.append({'t': round(t, 2), 'text': rest[:300], 'kind': 'line'}) if out and all(l['t'] is not None for l in out): out.sort(key=lambda l: l['t']) return out def norm(s): return re.sub(r'\s+', ' ', re.sub(r'[^a-z0-9 ]+', ' ', str(s or '').lower())).strip() def http_json(url, tries=4): """LRCLIB is free and rate-limits bursts with 503/429 — back off and retry.""" for attempt in range(tries): try: with urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=30) as r: return json.loads(r.read() or b'null') except urllib.error.HTTPError as e: if e.code == 404: return None if e.code in (429, 503) and attempt < tries - 1: time.sleep(2 * (attempt + 1)) continue raise except Exception: if attempt < tries - 1: time.sleep(2 * (attempt + 1)) continue raise return None def artist_ok(lrc_artist, channel, video_title): """True when the LRCLIB artist plausibly matches the video. Lyric-video channels ("Christian Lyrics", a person's name) carry no artist, so the artist is also looked for in the video title. Songs whose title is shared across genres ("Still") otherwise match the wrong recording.""" a = set(norm(lrc_artist).split()) - {'the', 'and', 'of', 'band', 'music', 'worship', 'ministries'} if not a: return False hay = set(norm(channel).split()) | set(norm(video_title).split()) return bool(a & hay) def lrclib_lookup(title, artist, duration, tolerance=6): """Best LRCLIB entry for a song, or None. Prefers synced lyrics.""" q = {'track_name': title, 'artist_name': artist or ''} if duration: q['duration'] = str(int(round(duration))) hit = http_json(f'{LRCLIB}/get?' + urllib.parse.urlencode(q)) if not hit: results = http_json(f'{LRCLIB}/search?' + urllib.parse.urlencode({'q': f'{title} {artist}'.strip()})) or [] want = norm(title) scored = [] for x in results: if x.get('instrumental') or not (x.get('syncedLyrics') or x.get('plainLyrics')): continue t = norm(x.get('trackName')) dd = abs((x.get('duration') or 0) - duration) if duration else 99 title_hit = 2 if t == want else 1 if (want in t or t in want) else 0 if not title_hit or (duration and dd > tolerance): continue scored.append((title_hit * 10 + (3 if x.get('syncedLyrics') else 0) - min(9, dd), x)) if not scored: return None hit = max(scored, key=lambda p: p[0])[1] lines = parse_lrc(hit.get('syncedLyrics') or '') synced = bool(lines) if not lines: lines = parse_lrc(hit.get('plainLyrics') or '') return {'lines': lines, 'synced': synced, 'hit': hit} if lines else None def describe(hit, cur_lines): h = hit['hit'] return f'{"synced" if hit["synced"] else "plain "} | {h.get("artistName", "")[:22]:22} – {h.get("trackName", "")[:26]:26} | {len(hit["lines"])} lines (was {cur_lines})' def main(): ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs')) ap.add_argument('--ids', help='comma-separated video ids') ap.add_argument('--since', help='only songs whose lyrics were last saved after this local time (YYYY-MM-DD HH:MM)') ap.add_argument('--until', help='…and before this one') ap.add_argument('--tagged', help='only songs whose lyrics carry this tag, e.g. auto-transcribed') ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', ''), help='where to write the backup JSON (default: Windows Documents on WSL, else ~/)') ap.add_argument('--apply', action='store_true', help='actually write the new lyrics (default: dry run)') ap.add_argument('--tolerance', type=float, default=6, help='max duration difference in seconds') ap.add_argument('--loose', action='store_true', help="accept matches whose artist doesn't line up with the video (risky)") args = ap.parse_args() api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')) # Which songs? Explicit ids, or every song the server has lyrics for, # filtered by when they were saved and/or their tag. ids = [x.strip() for x in (args.ids or '').split(',') if x.strip()] if not ids: st, r = api.call('GET', '/api/admin/media') if st != 200: sys.exit(f'need --ids (listing needs the newer server build: {r.get("error")})') ids = [m['id'] for m in r['media'] if m.get('lyricsLines')] to_ts = lambda s: dt.datetime.strptime(s, '%Y-%m-%d %H:%M').timestamp() if s else None since, until = to_ts(args.since), to_ts(args.until) picked, backup = [], {} for vid in ids: st, n = api.call('GET', f'/api/notes/{vid}') cur = (n or {}).get('lyrics') if st == 200 else None if not cur: continue when = cur['updatedAt'] if (since and when < since) or (until and when > until): continue if args.tagged and not any(args.tagged.lower() in t.lower() for t in cur['data'].get('tags', [])): continue st, sres = api.call('GET', f'/api/streams?v={vid}') meta = ((sres or {}).get('data') or {}).get('meta') or {} picked.append({'id': vid, 'cur': cur, 'meta': meta}) backup[vid] = {'savedAt': when, 'rev': cur['rev'], 'title': meta.get('title', ''), 'data': cur['data']} if not picked: sys.exit('nothing matched') # 1) Backup first — always, even on a dry run. out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~')) stamp = dt.datetime.now().strftime('%Y%m%d-%H%M%S') path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{stamp}.json') with open(path, 'w', encoding='utf-8') as f: json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base, 'songs': backup}, f, ensure_ascii=False, indent=1) print(f'backup: {len(backup)} songs → {path}\n') # 2) Look each one up and (optionally) replace. changed = skipped = 0 for p in picked: vid, meta, cur = p['id'], p['meta'], p['cur'] title, artist = clean_title(meta.get('title')), clean_artist(meta.get('channel')) dur = float(meta.get('duration') or 0) try: hit = lrclib_lookup(title, artist, dur, args.tolerance) except Exception as e: # network hiccup — keep going print(f'{vid} lookup failed: {e}') continue time.sleep(0.4) # be polite to a free service if not hit: skipped += 1 print(f'{vid} no match | {title[:42]:42} | {artist[:20]:20} | keeping {len(cur["data"]["lines"])} lines ({",".join(cur["data"].get("tags") or [])[:24]})') continue h = hit['hit'] sure = artist_ok(h.get('artistName'), meta.get('channel'), meta.get('title')) mark = 'LRCLIB' if sure else 'UNSURE' print(f'{vid} {mark} {describe(hit, len(cur["data"]["lines"]))}') if not sure and not args.loose: skipped += 1 print(f' ↳ artist doesn\'t match "{meta.get("channel", "")}" / the video title — left alone (use --loose to accept)') continue if not args.apply: continue doc = { 'lines': hit['lines'], 'tags': [f'from LRCLIB ({"synced" if hit["synced"] else "plain text"})'], 'offset': 0, } st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': cur['rev']}) if st == 200: changed += 1 else: print(f' write failed {st}: {rr.get("error")}') print(f'\n{changed} replaced, {skipped} left alone' + ('' if args.apply else ' (dry run — add --apply)')) if __name__ == '__main__': main()