#!/usr/bin/env python3 """auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics. Lyrics come from LRCLIB first (free, key-less, often SYNCED) and otherwise from faster-whisper (CTranslate2, int8, CPU) running on this machine — no API key, no credits either way. Audio comes from the server's own cache (/api/media/?a=1), results go to /api/notes//lyrics with baseRev, so a song someone already has lyrics for is never overwritten unless --overwrite. Setup (once): uv venv ~/.local/share/lyrics-asr/.venv --python 3.12 ~/.local/share/lyrics-asr/.venv/bin/pip install -r scripts/lyrics/requirements.txt Run: YTP_TOKEN=ytp_… .venv/bin/python scripts/lyrics/auto_lyrics.py --missing YTP_ADMIN_PASSWORD=… … --ids KcrXlpg0LKI,65vdbMHh4mo --dry-run Measured on the 16-core WSL laptop: large-v3-turbo ≈ 0.65× real time (6.5-min song in ~4 min); openai-whisper medium was ~2.5× real time. Karaoke / minus-one tracks have no vocals; they are reported as "instrumental" and skipped (their words are on screen — see the repo CLAUDE.md). """ import argparse import json import os import re import sys import tempfile import time import urllib.error import urllib.request import glob import http.cookiejar import threading FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I) KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ', 'He', 'His', 'Him', 'Thee', 'Thy', 'Thou', 'Father', 'Spirit', 'Holy', 'Savior', 'Saviour', 'King', 'Lamb', 'Earth', 'Heaven', 'Panginoon', 'Diyos', 'Hesus', 'Ikaw', 'Iyo', 'Iyong'} def segment(words, max_words=9, gap_break=1.0): """Word timings -> sung lines. Whisper punctuates songs sparsely, so lines break on sentence ends, pauses, capitalised line starts and a length cap; 1-2 word fragments (a held note split a phrase) fold into the next line.""" ws = [w for w in words if w['text'].strip()] groups, cur = [], [] for w in ws: if cur: gap = w['start'] - cur[-1]['end'] prev = cur[-1]['text'].strip() word = w['text'].strip() n = len(cur) cap_start = word[:1].isupper() and word.strip('",.!?') not in KEEP_CAP if (re.search(r'[.!?]$', prev) or gap >= gap_break or n >= max_words or (re.search(r'[,;:]$', prev) and n >= 4) or (cap_start and n >= 3) or (word in ('You', 'I') and n >= 5)): groups.append(cur) cur = [] cur.append(w) if cur: groups.append(cur) folded, i = [], 0 while i < len(groups): g = groups[i] if (len(g) <= 2 and i + 1 < len(groups) and not re.search(r'[.!?]$', g[-1]['text'].strip()) and groups[i + 1][0]['start'] - g[-1]['end'] < 4): groups[i + 1] = g + groups[i + 1] else: folded.append(g) i += 1 lines = [] for g in folded: text = re.sub(r'\s+([,.!?;:])', r'\1', ' '.join(w['text'].strip() for w in g)).strip().rstrip(',;:') if re.sub(r'[\W_]+', '', text): lines.append({'t': round(g[0]['start'], 2), 'end': g[-1]['end'], 'text': text[:300], 'n': len(g)}) out = [] for i, l in enumerate(lines): nxt = lines[i + 1]['t'] if i + 1 < len(lines) else l['end'] + 99 prv = lines[i - 1]['end'] if i else -99 if l['n'] == 1 and nxt - l['end'] > 4 and l['t'] - prv > 4: continue # isolated one-word line: intro/outro hallucination if FILLER.match(l['text']): continue out.append({'t': l['t'], 'text': l['text'], 'kind': 'line'}) return out class Api: def __init__(self, base, token=None, password=None): self.base = base.rstrip('/') self.token = token self.opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(http.cookiejar.CookieJar())) if not token and password: st, r = self.call('POST', '/api/admin/login', {'password': password}) if st != 200: sys.exit(f'admin login failed: {r}') def call(self, method, path, body=None, raw=False): headers = {'Content-Type': 'application/json'} if self.token: headers['Authorization'] = f'Bearer {self.token}' req = urllib.request.Request(self.base + path, method=method, headers=headers, data=json.dumps(body).encode() if body is not None else None) try: with self.opener.open(req, timeout=600) as r: data = r.read(MAX_AUDIO_BYTES + 1) if raw else r.read() return r.status, data if raw else json.loads(data or b'{}') except urllib.error.HTTPError as e: data = e.read() try: return e.code, json.loads(data or b'{}') except ValueError: return e.code, {'error': data[:200].decode('utf-8', 'replace')} def main(): ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs')) pick = ap.add_mutually_exclusive_group(required=True) pick.add_argument('--missing', action='store_true', help='every saved video without lyrics') pick.add_argument('--ids', help='comma-separated video ids') ap.add_argument('--overwrite', action='store_true', help='replace existing lyrics (kept in history)') ap.add_argument('--model', default='large-v3-turbo') ap.add_argument('--language', default=None, help='e.g. en, tl (default: auto-detect)') ap.add_argument('--threads', type=int, default=os.cpu_count() or 4) ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload') ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental') ap.add_argument('--watch', type=int, default=0, metavar='SECONDS', help='keep running: re-check for songs without lyrics every SECONDS (worker mode)') ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)') ap.add_argument('--no-web', action='store_true', help='skip the LRCLIB lookup and always transcribe') args = ap.parse_args() if args.watch: return watch(args) run_once(args, Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))) # A whisper run killed by the container's memory cap skips the tempfile # cleanup, and the restart used to retry the same song forever — 490 OOM # kills left 61 GB of copies of one long video in /tmp. Long audio is skipped # up front, stale temp files are swept on start, and a song that was in # progress when the worker died is recorded as a failure (with backoff). TMP_PREFIX = 'ytp-lyrics-' MAX_AUDIO_BYTES = int(os.environ.get('LYRICS_MAX_AUDIO_MB', '40')) * 1048576 def sweep_stale_tmp(): for f in glob.glob(os.path.join(tempfile.gettempdir(), 'tmp*.m4a')) + \ glob.glob(os.path.join(tempfile.gettempdir(), TMP_PREFIX + '*')): try: os.remove(f) except OSError: pass def load_state(path): try: with open(path) as f: return json.load(f) except (OSError, ValueError): return {} def save_state(path, state): if not path: return tmp = path + '.tmp' with open(tmp, 'w') as f: json.dump(state, f) os.replace(tmp, path) def watch(args): """Worker mode: poll for saved songs without lyrics and transcribe them one at a time, forever. Runs at low CPU priority; the state file remembers instrumentals (never retried) and failures (retried with backoff).""" try: os.nice(10) except OSError: pass token, password = os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD') if not (token or password): print('lyrics worker: no YTP_TOKEN set (LYRICS_WORKER_TOKEN in the server env) — idle', flush=True) while True: time.sleep(3600) args.missing, args.ids, args.overwrite, args.dry_run = True, None, False, False model = None sweep_stale_tmp() state = load_state(args.state) for vid, entry in state.items(): if entry.get('status') == 'in_progress': # the worker died mid-song fails = entry.get('fails', 0) + 1 state[vid] = {'status': 'failed', 'fails': fails, 'retry_at': time.time() + min(86400, 900 * 2 ** fails), 'error': 'worker died while transcribing (likely out of memory)'} print(f'lyrics worker: {vid} crashed the previous run — backing off', flush=True) save_state(args.state, state) next_auto = 0 def get_model(): nonlocal model if model is None: from faster_whisper import WhisperModel print(f'lyrics worker: loading {args.model}', flush=True) model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads) return model while True: try: api = Api(args.base, token, password) # Explicit admin requests take priority and always use Whisper. # Poll every ten seconds, independently of the automatic interval. if token: st, request = api.call('POST', '/api/lyrics-worker/claim', {}) if st == 200 and request.get('job'): process_requested(args, api, request['job'], get_model) continue if time.time() < next_auto: time.sleep(min(args.watch, 10)) continue st, r = api.call('GET', '/api/admin/media') if st != 200: raise RuntimeError(f'listing failed ({st}): {r.get("error")}') state = load_state(args.state) now = time.time() todo = [m for m in r['media'] if not m['lyricsLines'] and state.get(m['id'], {}).get('status') != 'instrumental' and state.get(m['id'], {}).get('retry_at', 0) <= now] if todo: get_model() m = todo[0] # one song per cycle keeps the worker's footprint small entry = state.get(m['id'], {}) state[m['id']] = {**entry, 'status': 'in_progress'} save_state(args.state, state) result = transcribe_one(args, api, model, m['id']) if result.startswith('instrumental') or result.startswith('too long'): state[m['id']] = {'status': 'instrumental', 'at': now} elif result.startswith('saved') or result.startswith('skip'): state.pop(m['id'], None) else: fails = entry.get('fails', 0) + 1 state[m['id']] = {'status': 'failed', 'fails': fails, 'retry_at': now + min(86400, 900 * 2 ** fails), 'error': result[:200]} save_state(args.state, state) print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True) continue # straight on to the next song next_auto = time.time() + args.watch except Exception as e: # never die: the next cycle retries print(f'lyrics worker: {e}', flush=True) next_auto = time.time() + args.watch time.sleep(min(args.watch, 10)) def process_requested(args, api, job, get_model): """Lease a manual request, heartbeat through model loading, return a draft.""" stopped = threading.Event() stage = ['loading-model'] path = f'/api/lyrics-worker/jobs/{job["id"]}' def report(extra=None): return api.call('POST', path, {'lease': job['lease'], 'stage': stage[0], **(extra or {})}) def heartbeat(): while not stopped.wait(25): try: st, _ = report() if st == 409: stopped.set() except Exception: pass # transient connection failure; the durable lease handles recovery thread = threading.Thread(target=heartbeat, daemon=True) thread.start() try: model = get_model() def progress(value): stage[0] = value st, _ = report() if st == 409: raise RuntimeError('The transcription lease expired.') result = transcribe_one(args, api, model, job['videoId'], draft_job=job, on_stage=progress) stopped.set() thread.join(timeout=2) payload = {'status': 'complete', 'result': result} if isinstance(result, dict) else {'status': 'failed', 'error': result} st, response = report(payload) if st == 400 and payload['status'] == 'complete': report({'status': 'failed', 'error': response.get('error', 'The transcript was rejected.')[:300]}) if st != 200: print(f'lyrics worker: request {job["id"]} could not finish ({st}): {response.get("error", "failed")}', flush=True) except Exception as e: stopped.set() try: report({'status': 'failed', 'error': str(e)[:300]}) except Exception: pass # the next worker can reclaim an expired request finally: stopped.set() thread.join(timeout=2) def run_once(args, api): if args.missing: st, r = api.call('GET', '/api/admin/media') if st != 200: sys.exit(f'listing saved videos failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD') todo = [m['id'] for m in r['media'] if args.overwrite or not m['lyricsLines']] else: todo = [x.strip() for x in args.ids.split(',') if x.strip()] print(f'{len(todo)} video(s) to transcribe with {args.model}', flush=True) if not todo: return from faster_whisper import WhisperModel # imported late: listing works without it model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads) summary = [(vid, transcribe_one(args, api, model, vid)) for vid in todo] print('\n'.join(f'{v} {s}' for v, s in summary)) def transcribe_one(args, api, model, vid, draft_job=None, on_stage=None): """Give one saved song lyrics. Published (often synced) lyrics from LRCLIB beat a machine transcript, so that is tried first; transcription is the fallback. Returns a one-line result.""" st, cur = api.call('GET', f'/api/notes/{vid}') live = (cur or {}).get('lyrics') if st == 200 else None if not draft_job and live and live['data']['lines'] and not args.overwrite: return 'skip: has lyrics' if not draft_job and not getattr(args, 'no_web', False): st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)}) if st == 200: m = r.get('match') or {} return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}" if on_stage: on_stage('downloading-audio') st, audio = api.call('GET', draft_job['audioPath'] if draft_job else f'/api/media/{vid}?a=1', raw=True) if st != 200: return f'no cached audio ({st})' if len(audio) > MAX_AUDIO_BYTES: return f'too long: {len(audio) // 1048576} MB audio (cap {MAX_AUDIO_BYTES // 1048576} MB) — skipped' with tempfile.NamedTemporaryFile(suffix='.m4a', prefix=TMP_PREFIX) as f: f.write(audio) f.flush() t0 = time.time() if on_stage: on_stage('transcribing') # vad_filter must stay OFF: it classifies sung music as non-speech # and silently drops the whole song. segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False, beam_size=5, condition_on_previous_text=False) words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end} for s in segs for w in (s.words or []) if w.word.strip()] took = time.time() - t0 if len(words) < args.min_words: return f'instrumental? only {len(words)} words — skipped' lines = segment(words) doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0} head = ' / '.join(l['text'] for l in lines[:3]) print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True) if draft_job: return doc if args.dry_run: print(json.dumps(doc, ensure_ascii=False)[:2000]) return f'dry-run {len(lines)} lines' body = {'data': doc, 'baseRev': live['rev'] if live else 0} st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body) return f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}' if __name__ == '__main__': main()