Files
ytplayer/scripts/lyrics/auto_lyrics.py

263 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics.
Lyrics come from LRCLIB first (free, key-less, often SYNCED) and otherwise from
faster-whisper (CTranslate2, int8, CPU) running on this machine — no API key, no
credits either way. Audio comes from the server's own cache
(/api/media/<id>?a=1), results go to /api/notes/<id>/lyrics with baseRev, so a
song someone already has lyrics for is never overwritten unless --overwrite.
Setup (once):
uv venv ~/.local/share/lyrics-asr/.venv --python 3.12
~/.local/share/lyrics-asr/.venv/bin/pip install -r scripts/lyrics/requirements.txt
Run:
YTP_TOKEN=ytp_… .venv/bin/python scripts/lyrics/auto_lyrics.py --missing
YTP_ADMIN_PASSWORD=… … --ids KcrXlpg0LKI,65vdbMHh4mo --dry-run
Measured on the 16-core WSL laptop: large-v3-turbo ≈ 0.65× real time
(6.5-min song in ~4 min); openai-whisper medium was ~2.5× real time.
Karaoke / minus-one tracks have no vocals; they are reported as "instrumental"
and skipped (their words are on screen — see the repo CLAUDE.md).
"""
import argparse
import json
import os
import re
import sys
import tempfile
import time
import urllib.error
import urllib.request
import http.cookiejar
FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I)
KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ',
'He', 'His', 'Him', 'Thee', 'Thy', 'Thou', 'Father', 'Spirit', 'Holy', 'Savior', 'Saviour', 'King', 'Lamb',
'Earth', 'Heaven', 'Panginoon', 'Diyos', 'Hesus', 'Ikaw', 'Iyo', 'Iyong'}
def segment(words, max_words=9, gap_break=1.0):
"""Word timings -> sung lines. Whisper punctuates songs sparsely, so lines
break on sentence ends, pauses, capitalised line starts and a length cap;
1-2 word fragments (a held note split a phrase) fold into the next line."""
ws = [w for w in words if w['text'].strip()]
groups, cur = [], []
for w in ws:
if cur:
gap = w['start'] - cur[-1]['end']
prev = cur[-1]['text'].strip()
word = w['text'].strip()
n = len(cur)
cap_start = word[:1].isupper() and word.strip('",.!?') not in KEEP_CAP
if (re.search(r'[.!?]$', prev) or gap >= gap_break or n >= max_words
or (re.search(r'[,;:]$', prev) and n >= 4) or (cap_start and n >= 3)
or (word in ('You', 'I') and n >= 5)):
groups.append(cur)
cur = []
cur.append(w)
if cur:
groups.append(cur)
folded, i = [], 0
while i < len(groups):
g = groups[i]
if (len(g) <= 2 and i + 1 < len(groups) and not re.search(r'[.!?]$', g[-1]['text'].strip())
and groups[i + 1][0]['start'] - g[-1]['end'] < 4):
groups[i + 1] = g + groups[i + 1]
else:
folded.append(g)
i += 1
lines = []
for g in folded:
text = re.sub(r'\s+([,.!?;:])', r'\1', ' '.join(w['text'].strip() for w in g)).strip().rstrip(',;:')
if re.sub(r'[\W_]+', '', text):
lines.append({'t': round(g[0]['start'], 2), 'end': g[-1]['end'], 'text': text[:300], 'n': len(g)})
out = []
for i, l in enumerate(lines):
nxt = lines[i + 1]['t'] if i + 1 < len(lines) else l['end'] + 99
prv = lines[i - 1]['end'] if i else -99
if l['n'] == 1 and nxt - l['end'] > 4 and l['t'] - prv > 4:
continue # isolated one-word line: intro/outro hallucination
if FILLER.match(l['text']):
continue
out.append({'t': l['t'], 'text': l['text'], 'kind': 'line'})
return out
class Api:
def __init__(self, base, token=None, password=None):
self.base = base.rstrip('/')
self.token = token
self.opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(http.cookiejar.CookieJar()))
if not token and password:
st, r = self.call('POST', '/api/admin/login', {'password': password})
if st != 200:
sys.exit(f'admin login failed: {r}')
def call(self, method, path, body=None, raw=False):
headers = {'Content-Type': 'application/json'}
if self.token:
headers['Authorization'] = f'Bearer {self.token}'
req = urllib.request.Request(self.base + path, method=method, headers=headers,
data=json.dumps(body).encode() if body is not None else None)
try:
with self.opener.open(req, timeout=600) as r:
data = r.read()
return r.status, data if raw else json.loads(data or b'{}')
except urllib.error.HTTPError as e:
data = e.read()
try:
return e.code, json.loads(data or b'{}')
except ValueError:
return e.code, {'error': data[:200].decode('utf-8', 'replace')}
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
pick = ap.add_mutually_exclusive_group(required=True)
pick.add_argument('--missing', action='store_true', help='every saved video without lyrics')
pick.add_argument('--ids', help='comma-separated video ids')
ap.add_argument('--overwrite', action='store_true', help='replace existing lyrics (kept in history)')
ap.add_argument('--model', default='large-v3-turbo')
ap.add_argument('--language', default=None, help='e.g. en, tl (default: auto-detect)')
ap.add_argument('--threads', type=int, default=os.cpu_count() or 4)
ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload')
ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental')
ap.add_argument('--watch', type=int, default=0, metavar='SECONDS',
help='keep running: re-check for songs without lyrics every SECONDS (worker mode)')
ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)')
ap.add_argument('--no-web', action='store_true', help='skip the LRCLIB lookup and always transcribe')
args = ap.parse_args()
if args.watch:
return watch(args)
run_once(args, Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')))
def load_state(path):
try:
with open(path) as f:
return json.load(f)
except (OSError, ValueError):
return {}
def save_state(path, state):
if not path:
return
tmp = path + '.tmp'
with open(tmp, 'w') as f:
json.dump(state, f)
os.replace(tmp, path)
def watch(args):
"""Worker mode: poll for saved songs without lyrics and transcribe them one
at a time, forever. Runs at low CPU priority; the state file remembers
instrumentals (never retried) and failures (retried with backoff)."""
try:
os.nice(10)
except OSError:
pass
token, password = os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')
if not (token or password):
print('lyrics worker: no YTP_TOKEN set (LYRICS_WORKER_TOKEN in the server env) — idle', flush=True)
while True:
time.sleep(3600)
args.missing, args.ids, args.overwrite, args.dry_run = True, None, False, False
model = None
while True:
try:
api = Api(args.base, token, password)
st, r = api.call('GET', '/api/admin/media')
if st != 200:
raise RuntimeError(f'listing failed ({st}): {r.get("error")}')
state = load_state(args.state)
now = time.time()
todo = [m for m in r['media'] if not m['lyricsLines']
and state.get(m['id'], {}).get('status') != 'instrumental'
and state.get(m['id'], {}).get('retry_at', 0) <= now]
if todo:
if model is None:
from faster_whisper import WhisperModel
print(f'lyrics worker: loading {args.model}', flush=True)
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
m = todo[0] # one song per cycle keeps the worker's footprint small
result = transcribe_one(args, api, model, m['id'])
entry = state.get(m['id'], {})
if result.startswith('instrumental'):
state[m['id']] = {'status': 'instrumental', 'at': now}
elif result.startswith('saved') or result.startswith('skip'):
state.pop(m['id'], None)
else:
fails = entry.get('fails', 0) + 1
state[m['id']] = {'status': 'failed', 'fails': fails, 'retry_at': now + min(86400, 900 * 2 ** fails), 'error': result[:200]}
save_state(args.state, state)
print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True)
continue # straight on to the next song
except Exception as e: # never die: the next cycle retries
print(f'lyrics worker: {e}', flush=True)
time.sleep(args.watch)
def run_once(args, api):
if args.missing:
st, r = api.call('GET', '/api/admin/media')
if st != 200:
sys.exit(f'listing saved videos failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
todo = [m['id'] for m in r['media'] if args.overwrite or not m['lyricsLines']]
else:
todo = [x.strip() for x in args.ids.split(',') if x.strip()]
print(f'{len(todo)} video(s) to transcribe with {args.model}', flush=True)
if not todo:
return
from faster_whisper import WhisperModel # imported late: listing works without it
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
summary = [(vid, transcribe_one(args, api, model, vid)) for vid in todo]
print('\n'.join(f'{v} {s}' for v, s in summary))
def transcribe_one(args, api, model, vid):
"""Give one saved song lyrics. Published (often synced) lyrics from LRCLIB
beat a machine transcript, so that is tried first; transcription is the
fallback. Returns a one-line result."""
st, cur = api.call('GET', f'/api/notes/{vid}')
live = (cur or {}).get('lyrics') if st == 200 else None
if live and live['data']['lines'] and not args.overwrite:
return 'skip: has lyrics'
if not getattr(args, 'no_web', False):
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
if st == 200:
m = r.get('match') or {}
return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}"
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
if st != 200:
return f'no cached audio ({st})'
with tempfile.NamedTemporaryFile(suffix='.m4a') as f:
f.write(audio)
f.flush()
t0 = time.time()
# vad_filter must stay OFF: it classifies sung music as non-speech
# and silently drops the whole song.
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
beam_size=5, condition_on_previous_text=False)
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
for s in segs for w in (s.words or []) if w.word.strip()]
took = time.time() - t0
if len(words) < args.min_words:
return f'instrumental? only {len(words)} words — skipped'
lines = segment(words)
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
head = ' / '.join(l['text'] for l in lines[:3])
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
if args.dry_run:
print(json.dumps(doc, ensure_ascii=False)[:2000])
return f'dry-run {len(lines)} lines'
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
return f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'
if __name__ == '__main__':
main()