#!/usr/bin/env python3 """auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics. Lyrics come from LRCLIB first (free, key-less, often SYNCED) and otherwise from faster-whisper (CTranslate2, int8, CPU) running on this machine — no API key, no credits either way. Audio comes from the server's own cache (/api/media/?a=1), results go to /api/notes//lyrics with baseRev, so a song someone already has lyrics for is never overwritten unless --overwrite. Setup (once): uv venv ~/.local/share/lyrics-asr/.venv --python 3.12 ~/.local/share/lyrics-asr/.venv/bin/pip install -r scripts/lyrics/requirements.txt Run: YTP_TOKEN=ytp_… .venv/bin/python scripts/lyrics/auto_lyrics.py --missing YTP_ADMIN_PASSWORD=… … --ids KcrXlpg0LKI,65vdbMHh4mo --dry-run Measured on the 16-core WSL laptop: large-v3-turbo ≈ 0.65× real time (6.5-min song in ~4 min); openai-whisper medium was ~2.5× real time. Karaoke / minus-one tracks have no vocals; they are reported as "instrumental" and skipped (their words are on screen — see the repo CLAUDE.md). """ import argparse import json import os import re import sys import tempfile import time import urllib.error import urllib.request import http.cookiejar FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I) KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ', 'He', 'His', 'Him', 'Thee', 'Thy', 'Thou', 'Father', 'Spirit', 'Holy', 'Savior', 'Saviour', 'King', 'Lamb', 'Earth', 'Heaven', 'Panginoon', 'Diyos', 'Hesus', 'Ikaw', 'Iyo', 'Iyong'} def segment(words, max_words=9, gap_break=1.0): """Word timings -> sung lines. Whisper punctuates songs sparsely, so lines break on sentence ends, pauses, capitalised line starts and a length cap; 1-2 word fragments (a held note split a phrase) fold into the next line.""" ws = [w for w in words if w['text'].strip()] groups, cur = [], [] for w in ws: if cur: gap = w['start'] - cur[-1]['end'] prev = cur[-1]['text'].strip() word = w['text'].strip() n = len(cur) cap_start = word[:1].isupper() and word.strip('",.!?') not in KEEP_CAP if (re.search(r'[.!?]$', prev) or gap >= gap_break or n >= max_words or (re.search(r'[,;:]$', prev) and n >= 4) or (cap_start and n >= 3) or (word in ('You', 'I') and n >= 5)): groups.append(cur) cur = [] cur.append(w) if cur: groups.append(cur) folded, i = [], 0 while i < len(groups): g = groups[i] if (len(g) <= 2 and i + 1 < len(groups) and not re.search(r'[.!?]$', g[-1]['text'].strip()) and groups[i + 1][0]['start'] - g[-1]['end'] < 4): groups[i + 1] = g + groups[i + 1] else: folded.append(g) i += 1 lines = [] for g in folded: text = re.sub(r'\s+([,.!?;:])', r'\1', ' '.join(w['text'].strip() for w in g)).strip().rstrip(',;:') if re.sub(r'[\W_]+', '', text): lines.append({'t': round(g[0]['start'], 2), 'end': g[-1]['end'], 'text': text[:300], 'n': len(g)}) out = [] for i, l in enumerate(lines): nxt = lines[i + 1]['t'] if i + 1 < len(lines) else l['end'] + 99 prv = lines[i - 1]['end'] if i else -99 if l['n'] == 1 and nxt - l['end'] > 4 and l['t'] - prv > 4: continue # isolated one-word line: intro/outro hallucination if FILLER.match(l['text']): continue out.append({'t': l['t'], 'text': l['text'], 'kind': 'line'}) return out class Api: def __init__(self, base, token=None, password=None): self.base = base.rstrip('/') self.token = token self.opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(http.cookiejar.CookieJar())) if not token and password: st, r = self.call('POST', '/api/admin/login', {'password': password}) if st != 200: sys.exit(f'admin login failed: {r}') def call(self, method, path, body=None, raw=False): headers = {'Content-Type': 'application/json'} if self.token: headers['Authorization'] = f'Bearer {self.token}' req = urllib.request.Request(self.base + path, method=method, headers=headers, data=json.dumps(body).encode() if body is not None else None) try: with self.opener.open(req, timeout=600) as r: data = r.read() return r.status, data if raw else json.loads(data or b'{}') except urllib.error.HTTPError as e: data = e.read() try: return e.code, json.loads(data or b'{}') except ValueError: return e.code, {'error': data[:200].decode('utf-8', 'replace')} def main(): ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs')) pick = ap.add_mutually_exclusive_group(required=True) pick.add_argument('--missing', action='store_true', help='every saved video without lyrics') pick.add_argument('--ids', help='comma-separated video ids') ap.add_argument('--overwrite', action='store_true', help='replace existing lyrics (kept in history)') ap.add_argument('--model', default='large-v3-turbo') ap.add_argument('--language', default=None, help='e.g. en, tl (default: auto-detect)') ap.add_argument('--threads', type=int, default=os.cpu_count() or 4) ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload') ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental') ap.add_argument('--watch', type=int, default=0, metavar='SECONDS', help='keep running: re-check for songs without lyrics every SECONDS (worker mode)') ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)') ap.add_argument('--no-web', action='store_true', help='skip the LRCLIB lookup and always transcribe') args = ap.parse_args() if args.watch: return watch(args) run_once(args, Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))) def load_state(path): try: with open(path) as f: return json.load(f) except (OSError, ValueError): return {} def save_state(path, state): if not path: return tmp = path + '.tmp' with open(tmp, 'w') as f: json.dump(state, f) os.replace(tmp, path) def watch(args): """Worker mode: poll for saved songs without lyrics and transcribe them one at a time, forever. Runs at low CPU priority; the state file remembers instrumentals (never retried) and failures (retried with backoff).""" try: os.nice(10) except OSError: pass token, password = os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD') if not (token or password): print('lyrics worker: no YTP_TOKEN set (LYRICS_WORKER_TOKEN in the server env) — idle', flush=True) while True: time.sleep(3600) args.missing, args.ids, args.overwrite, args.dry_run = True, None, False, False model = None while True: try: api = Api(args.base, token, password) st, r = api.call('GET', '/api/admin/media') if st != 200: raise RuntimeError(f'listing failed ({st}): {r.get("error")}') state = load_state(args.state) now = time.time() todo = [m for m in r['media'] if not m['lyricsLines'] and state.get(m['id'], {}).get('status') != 'instrumental' and state.get(m['id'], {}).get('retry_at', 0) <= now] if todo: if model is None: from faster_whisper import WhisperModel print(f'lyrics worker: loading {args.model}', flush=True) model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads) m = todo[0] # one song per cycle keeps the worker's footprint small result = transcribe_one(args, api, model, m['id']) entry = state.get(m['id'], {}) if result.startswith('instrumental'): state[m['id']] = {'status': 'instrumental', 'at': now} elif result.startswith('saved') or result.startswith('skip'): state.pop(m['id'], None) else: fails = entry.get('fails', 0) + 1 state[m['id']] = {'status': 'failed', 'fails': fails, 'retry_at': now + min(86400, 900 * 2 ** fails), 'error': result[:200]} save_state(args.state, state) print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True) continue # straight on to the next song except Exception as e: # never die: the next cycle retries print(f'lyrics worker: {e}', flush=True) time.sleep(args.watch) def run_once(args, api): if args.missing: st, r = api.call('GET', '/api/admin/media') if st != 200: sys.exit(f'listing saved videos failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD') todo = [m['id'] for m in r['media'] if args.overwrite or not m['lyricsLines']] else: todo = [x.strip() for x in args.ids.split(',') if x.strip()] print(f'{len(todo)} video(s) to transcribe with {args.model}', flush=True) if not todo: return from faster_whisper import WhisperModel # imported late: listing works without it model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads) summary = [(vid, transcribe_one(args, api, model, vid)) for vid in todo] print('\n'.join(f'{v} {s}' for v, s in summary)) def transcribe_one(args, api, model, vid): """Give one saved song lyrics. Published (often synced) lyrics from LRCLIB beat a machine transcript, so that is tried first; transcription is the fallback. Returns a one-line result.""" st, cur = api.call('GET', f'/api/notes/{vid}') live = (cur or {}).get('lyrics') if st == 200 else None if live and live['data']['lines'] and not args.overwrite: return 'skip: has lyrics' if not getattr(args, 'no_web', False): st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)}) if st == 200: m = r.get('match') or {} return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}" st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True) if st != 200: return f'no cached audio ({st})' with tempfile.NamedTemporaryFile(suffix='.m4a') as f: f.write(audio) f.flush() t0 = time.time() # vad_filter must stay OFF: it classifies sung music as non-speech # and silently drops the whole song. segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False, beam_size=5, condition_on_previous_text=False) words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end} for s in segs for w in (s.words or []) if w.word.strip()] took = time.time() - t0 if len(words) < args.min_words: return f'instrumental? only {len(words)} words — skipped' lines = segment(words) doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0} head = ' / '.join(l['text'] for l in lines[:3]) print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True) if args.dry_run: print(json.dumps(doc, ensure_ascii=False)[:2000]) return f'dry-run {len(lines)} lines' body = {'data': doc, 'baseRev': live['rev'] if live else 0} st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body) return f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}' if __name__ == '__main__': main()