Files
ytplayer/scripts/lyrics/auto_lyrics.py
Jonathan Sykes 2b38717c05 Add admin analytics, metadata collection, and grouped lyric cues
Queue reviewable Whisper drafts from the song list and lyrics editor. Preserve line breaks within one timed cue across editing, saving, reporting, and service views.

Add storage and listening analytics with a durable metadata collector, related-search depth, video limits, thumbnail storage, and a browsable metadata library.
2026-10-03 07:55:33 +08:00

368 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics.
Lyrics come from LRCLIB first (free, key-less, often SYNCED) and otherwise from
faster-whisper (CTranslate2, int8, CPU) running on this machine — no API key, no
credits either way. Audio comes from the server's own cache
(/api/media/<id>?a=1), results go to /api/notes/<id>/lyrics with baseRev, so a
song someone already has lyrics for is never overwritten unless --overwrite.
Setup (once):
uv venv ~/.local/share/lyrics-asr/.venv --python 3.12
~/.local/share/lyrics-asr/.venv/bin/pip install -r scripts/lyrics/requirements.txt
Run:
YTP_TOKEN=ytp_… .venv/bin/python scripts/lyrics/auto_lyrics.py --missing
YTP_ADMIN_PASSWORD=… … --ids KcrXlpg0LKI,65vdbMHh4mo --dry-run
Measured on the 16-core WSL laptop: large-v3-turbo ≈ 0.65× real time
(6.5-min song in ~4 min); openai-whisper medium was ~2.5× real time.
Karaoke / minus-one tracks have no vocals; they are reported as "instrumental"
and skipped (their words are on screen — see the repo CLAUDE.md).
"""
import argparse
import json
import os
import re
import sys
import tempfile
import time
import urllib.error
import urllib.request
import glob
import http.cookiejar
import threading
FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I)
KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ',
'He', 'His', 'Him', 'Thee', 'Thy', 'Thou', 'Father', 'Spirit', 'Holy', 'Savior', 'Saviour', 'King', 'Lamb',
'Earth', 'Heaven', 'Panginoon', 'Diyos', 'Hesus', 'Ikaw', 'Iyo', 'Iyong'}
def segment(words, max_words=9, gap_break=1.0):
"""Word timings -> sung lines. Whisper punctuates songs sparsely, so lines
break on sentence ends, pauses, capitalised line starts and a length cap;
1-2 word fragments (a held note split a phrase) fold into the next line."""
ws = [w for w in words if w['text'].strip()]
groups, cur = [], []
for w in ws:
if cur:
gap = w['start'] - cur[-1]['end']
prev = cur[-1]['text'].strip()
word = w['text'].strip()
n = len(cur)
cap_start = word[:1].isupper() and word.strip('",.!?') not in KEEP_CAP
if (re.search(r'[.!?]$', prev) or gap >= gap_break or n >= max_words
or (re.search(r'[,;:]$', prev) and n >= 4) or (cap_start and n >= 3)
or (word in ('You', 'I') and n >= 5)):
groups.append(cur)
cur = []
cur.append(w)
if cur:
groups.append(cur)
folded, i = [], 0
while i < len(groups):
g = groups[i]
if (len(g) <= 2 and i + 1 < len(groups) and not re.search(r'[.!?]$', g[-1]['text'].strip())
and groups[i + 1][0]['start'] - g[-1]['end'] < 4):
groups[i + 1] = g + groups[i + 1]
else:
folded.append(g)
i += 1
lines = []
for g in folded:
text = re.sub(r'\s+([,.!?;:])', r'\1', ' '.join(w['text'].strip() for w in g)).strip().rstrip(',;:')
if re.sub(r'[\W_]+', '', text):
lines.append({'t': round(g[0]['start'], 2), 'end': g[-1]['end'], 'text': text[:300], 'n': len(g)})
out = []
for i, l in enumerate(lines):
nxt = lines[i + 1]['t'] if i + 1 < len(lines) else l['end'] + 99
prv = lines[i - 1]['end'] if i else -99
if l['n'] == 1 and nxt - l['end'] > 4 and l['t'] - prv > 4:
continue # isolated one-word line: intro/outro hallucination
if FILLER.match(l['text']):
continue
out.append({'t': l['t'], 'text': l['text'], 'kind': 'line'})
return out
class Api:
def __init__(self, base, token=None, password=None):
self.base = base.rstrip('/')
self.token = token
self.opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(http.cookiejar.CookieJar()))
if not token and password:
st, r = self.call('POST', '/api/admin/login', {'password': password})
if st != 200:
sys.exit(f'admin login failed: {r}')
def call(self, method, path, body=None, raw=False):
headers = {'Content-Type': 'application/json'}
if self.token:
headers['Authorization'] = f'Bearer {self.token}'
req = urllib.request.Request(self.base + path, method=method, headers=headers,
data=json.dumps(body).encode() if body is not None else None)
try:
with self.opener.open(req, timeout=600) as r:
data = r.read(MAX_AUDIO_BYTES + 1) if raw else r.read()
return r.status, data if raw else json.loads(data or b'{}')
except urllib.error.HTTPError as e:
data = e.read()
try:
return e.code, json.loads(data or b'{}')
except ValueError:
return e.code, {'error': data[:200].decode('utf-8', 'replace')}
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
pick = ap.add_mutually_exclusive_group(required=True)
pick.add_argument('--missing', action='store_true', help='every saved video without lyrics')
pick.add_argument('--ids', help='comma-separated video ids')
ap.add_argument('--overwrite', action='store_true', help='replace existing lyrics (kept in history)')
ap.add_argument('--model', default='large-v3-turbo')
ap.add_argument('--language', default=None, help='e.g. en, tl (default: auto-detect)')
ap.add_argument('--threads', type=int, default=os.cpu_count() or 4)
ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload')
ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental')
ap.add_argument('--watch', type=int, default=0, metavar='SECONDS',
help='keep running: re-check for songs without lyrics every SECONDS (worker mode)')
ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)')
ap.add_argument('--no-web', action='store_true', help='skip the LRCLIB lookup and always transcribe')
args = ap.parse_args()
if args.watch:
return watch(args)
run_once(args, Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')))
# A whisper run killed by the container's memory cap skips the tempfile
# cleanup, and the restart used to retry the same song forever — 490 OOM
# kills left 61 GB of copies of one long video in /tmp. Long audio is skipped
# up front, stale temp files are swept on start, and a song that was in
# progress when the worker died is recorded as a failure (with backoff).
TMP_PREFIX = 'ytp-lyrics-'
MAX_AUDIO_BYTES = int(os.environ.get('LYRICS_MAX_AUDIO_MB', '40')) * 1048576
def sweep_stale_tmp():
for f in glob.glob(os.path.join(tempfile.gettempdir(), 'tmp*.m4a')) + \
glob.glob(os.path.join(tempfile.gettempdir(), TMP_PREFIX + '*')):
try:
os.remove(f)
except OSError:
pass
def load_state(path):
try:
with open(path) as f:
return json.load(f)
except (OSError, ValueError):
return {}
def save_state(path, state):
if not path:
return
tmp = path + '.tmp'
with open(tmp, 'w') as f:
json.dump(state, f)
os.replace(tmp, path)
def watch(args):
"""Worker mode: poll for saved songs without lyrics and transcribe them one
at a time, forever. Runs at low CPU priority; the state file remembers
instrumentals (never retried) and failures (retried with backoff)."""
try:
os.nice(10)
except OSError:
pass
token, password = os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')
if not (token or password):
print('lyrics worker: no YTP_TOKEN set (LYRICS_WORKER_TOKEN in the server env) — idle', flush=True)
while True:
time.sleep(3600)
args.missing, args.ids, args.overwrite, args.dry_run = True, None, False, False
model = None
sweep_stale_tmp()
state = load_state(args.state)
for vid, entry in state.items():
if entry.get('status') == 'in_progress': # the worker died mid-song
fails = entry.get('fails', 0) + 1
state[vid] = {'status': 'failed', 'fails': fails, 'retry_at': time.time() + min(86400, 900 * 2 ** fails),
'error': 'worker died while transcribing (likely out of memory)'}
print(f'lyrics worker: {vid} crashed the previous run — backing off', flush=True)
save_state(args.state, state)
next_auto = 0
def get_model():
nonlocal model
if model is None:
from faster_whisper import WhisperModel
print(f'lyrics worker: loading {args.model}', flush=True)
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
return model
while True:
try:
api = Api(args.base, token, password)
# Explicit admin requests take priority and always use Whisper.
# Poll every ten seconds, independently of the automatic interval.
if token:
st, request = api.call('POST', '/api/lyrics-worker/claim', {})
if st == 200 and request.get('job'):
process_requested(args, api, request['job'], get_model)
continue
if time.time() < next_auto:
time.sleep(min(args.watch, 10))
continue
st, r = api.call('GET', '/api/admin/media')
if st != 200:
raise RuntimeError(f'listing failed ({st}): {r.get("error")}')
state = load_state(args.state)
now = time.time()
todo = [m for m in r['media'] if not m['lyricsLines']
and state.get(m['id'], {}).get('status') != 'instrumental'
and state.get(m['id'], {}).get('retry_at', 0) <= now]
if todo:
get_model()
m = todo[0] # one song per cycle keeps the worker's footprint small
entry = state.get(m['id'], {})
state[m['id']] = {**entry, 'status': 'in_progress'}
save_state(args.state, state)
result = transcribe_one(args, api, model, m['id'])
if result.startswith('instrumental') or result.startswith('too long'):
state[m['id']] = {'status': 'instrumental', 'at': now}
elif result.startswith('saved') or result.startswith('skip'):
state.pop(m['id'], None)
else:
fails = entry.get('fails', 0) + 1
state[m['id']] = {'status': 'failed', 'fails': fails, 'retry_at': now + min(86400, 900 * 2 ** fails), 'error': result[:200]}
save_state(args.state, state)
print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True)
continue # straight on to the next song
next_auto = time.time() + args.watch
except Exception as e: # never die: the next cycle retries
print(f'lyrics worker: {e}', flush=True)
next_auto = time.time() + args.watch
time.sleep(min(args.watch, 10))
def process_requested(args, api, job, get_model):
"""Lease a manual request, heartbeat through model loading, return a draft."""
stopped = threading.Event()
stage = ['loading-model']
path = f'/api/lyrics-worker/jobs/{job["id"]}'
def report(extra=None):
return api.call('POST', path, {'lease': job['lease'], 'stage': stage[0], **(extra or {})})
def heartbeat():
while not stopped.wait(25):
try:
st, _ = report()
if st == 409:
stopped.set()
except Exception:
pass # transient connection failure; the durable lease handles recovery
thread = threading.Thread(target=heartbeat, daemon=True)
thread.start()
try:
model = get_model()
def progress(value):
stage[0] = value
st, _ = report()
if st == 409:
raise RuntimeError('The transcription lease expired.')
result = transcribe_one(args, api, model, job['videoId'], draft_job=job, on_stage=progress)
stopped.set()
thread.join(timeout=2)
payload = {'status': 'complete', 'result': result} if isinstance(result, dict) else {'status': 'failed', 'error': result}
st, response = report(payload)
if st == 400 and payload['status'] == 'complete':
report({'status': 'failed', 'error': response.get('error', 'The transcript was rejected.')[:300]})
if st != 200:
print(f'lyrics worker: request {job["id"]} could not finish ({st}): {response.get("error", "failed")}', flush=True)
except Exception as e:
stopped.set()
try:
report({'status': 'failed', 'error': str(e)[:300]})
except Exception:
pass # the next worker can reclaim an expired request
finally:
stopped.set()
thread.join(timeout=2)
def run_once(args, api):
if args.missing:
st, r = api.call('GET', '/api/admin/media')
if st != 200:
sys.exit(f'listing saved videos failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
todo = [m['id'] for m in r['media'] if args.overwrite or not m['lyricsLines']]
else:
todo = [x.strip() for x in args.ids.split(',') if x.strip()]
print(f'{len(todo)} video(s) to transcribe with {args.model}', flush=True)
if not todo:
return
from faster_whisper import WhisperModel # imported late: listing works without it
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
summary = [(vid, transcribe_one(args, api, model, vid)) for vid in todo]
print('\n'.join(f'{v} {s}' for v, s in summary))
def transcribe_one(args, api, model, vid, draft_job=None, on_stage=None):
"""Give one saved song lyrics. Published (often synced) lyrics from LRCLIB
beat a machine transcript, so that is tried first; transcription is the
fallback. Returns a one-line result."""
st, cur = api.call('GET', f'/api/notes/{vid}')
live = (cur or {}).get('lyrics') if st == 200 else None
if not draft_job and live and live['data']['lines'] and not args.overwrite:
return 'skip: has lyrics'
if not draft_job and not getattr(args, 'no_web', False):
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
if st == 200:
m = r.get('match') or {}
return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}"
if on_stage:
on_stage('downloading-audio')
st, audio = api.call('GET', draft_job['audioPath'] if draft_job else f'/api/media/{vid}?a=1', raw=True)
if st != 200:
return f'no cached audio ({st})'
if len(audio) > MAX_AUDIO_BYTES:
return f'too long: {len(audio) // 1048576} MB audio (cap {MAX_AUDIO_BYTES // 1048576} MB) — skipped'
with tempfile.NamedTemporaryFile(suffix='.m4a', prefix=TMP_PREFIX) as f:
f.write(audio)
f.flush()
t0 = time.time()
if on_stage:
on_stage('transcribing')
# vad_filter must stay OFF: it classifies sung music as non-speech
# and silently drops the whole song.
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
beam_size=5, condition_on_previous_text=False)
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
for s in segs for w in (s.words or []) if w.word.strip()]
took = time.time() - t0
if len(words) < args.min_words:
return f'instrumental? only {len(words)} words — skipped'
lines = segment(words)
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
head = ' / '.join(l['text'] for l in lines[:3])
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
if draft_job:
return doc
if args.dry_run:
print(json.dumps(doc, ensure_ascii=False)[:2000])
return f'dry-run {len(lines)} lines'
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
return f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'
if __name__ == '__main__':
main()