263 lines
12 KiB
Python
263 lines
12 KiB
Python
#!/usr/bin/env python3
|
||
"""auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics.
|
||
|
||
Lyrics come from LRCLIB first (free, key-less, often SYNCED) and otherwise from
|
||
faster-whisper (CTranslate2, int8, CPU) running on this machine — no API key, no
|
||
credits either way. Audio comes from the server's own cache
|
||
(/api/media/<id>?a=1), results go to /api/notes/<id>/lyrics with baseRev, so a
|
||
song someone already has lyrics for is never overwritten unless --overwrite.
|
||
|
||
Setup (once):
|
||
uv venv ~/.local/share/lyrics-asr/.venv --python 3.12
|
||
~/.local/share/lyrics-asr/.venv/bin/pip install -r scripts/lyrics/requirements.txt
|
||
Run:
|
||
YTP_TOKEN=ytp_… .venv/bin/python scripts/lyrics/auto_lyrics.py --missing
|
||
YTP_ADMIN_PASSWORD=… … --ids KcrXlpg0LKI,65vdbMHh4mo --dry-run
|
||
Measured on the 16-core WSL laptop: large-v3-turbo ≈ 0.65× real time
|
||
(6.5-min song in ~4 min); openai-whisper medium was ~2.5× real time.
|
||
Karaoke / minus-one tracks have no vocals; they are reported as "instrumental"
|
||
and skipped (their words are on screen — see the repo CLAUDE.md).
|
||
"""
|
||
import argparse
|
||
import json
|
||
import os
|
||
import re
|
||
import sys
|
||
import tempfile
|
||
import time
|
||
import urllib.error
|
||
import urllib.request
|
||
import http.cookiejar
|
||
|
||
FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I)
|
||
KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ',
|
||
'He', 'His', 'Him', 'Thee', 'Thy', 'Thou', 'Father', 'Spirit', 'Holy', 'Savior', 'Saviour', 'King', 'Lamb',
|
||
'Earth', 'Heaven', 'Panginoon', 'Diyos', 'Hesus', 'Ikaw', 'Iyo', 'Iyong'}
|
||
|
||
|
||
def segment(words, max_words=9, gap_break=1.0):
|
||
"""Word timings -> sung lines. Whisper punctuates songs sparsely, so lines
|
||
break on sentence ends, pauses, capitalised line starts and a length cap;
|
||
1-2 word fragments (a held note split a phrase) fold into the next line."""
|
||
ws = [w for w in words if w['text'].strip()]
|
||
groups, cur = [], []
|
||
for w in ws:
|
||
if cur:
|
||
gap = w['start'] - cur[-1]['end']
|
||
prev = cur[-1]['text'].strip()
|
||
word = w['text'].strip()
|
||
n = len(cur)
|
||
cap_start = word[:1].isupper() and word.strip('",.!?') not in KEEP_CAP
|
||
if (re.search(r'[.!?]$', prev) or gap >= gap_break or n >= max_words
|
||
or (re.search(r'[,;:]$', prev) and n >= 4) or (cap_start and n >= 3)
|
||
or (word in ('You', 'I') and n >= 5)):
|
||
groups.append(cur)
|
||
cur = []
|
||
cur.append(w)
|
||
if cur:
|
||
groups.append(cur)
|
||
folded, i = [], 0
|
||
while i < len(groups):
|
||
g = groups[i]
|
||
if (len(g) <= 2 and i + 1 < len(groups) and not re.search(r'[.!?]$', g[-1]['text'].strip())
|
||
and groups[i + 1][0]['start'] - g[-1]['end'] < 4):
|
||
groups[i + 1] = g + groups[i + 1]
|
||
else:
|
||
folded.append(g)
|
||
i += 1
|
||
lines = []
|
||
for g in folded:
|
||
text = re.sub(r'\s+([,.!?;:])', r'\1', ' '.join(w['text'].strip() for w in g)).strip().rstrip(',;:')
|
||
if re.sub(r'[\W_]+', '', text):
|
||
lines.append({'t': round(g[0]['start'], 2), 'end': g[-1]['end'], 'text': text[:300], 'n': len(g)})
|
||
out = []
|
||
for i, l in enumerate(lines):
|
||
nxt = lines[i + 1]['t'] if i + 1 < len(lines) else l['end'] + 99
|
||
prv = lines[i - 1]['end'] if i else -99
|
||
if l['n'] == 1 and nxt - l['end'] > 4 and l['t'] - prv > 4:
|
||
continue # isolated one-word line: intro/outro hallucination
|
||
if FILLER.match(l['text']):
|
||
continue
|
||
out.append({'t': l['t'], 'text': l['text'], 'kind': 'line'})
|
||
return out
|
||
|
||
|
||
class Api:
|
||
def __init__(self, base, token=None, password=None):
|
||
self.base = base.rstrip('/')
|
||
self.token = token
|
||
self.opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(http.cookiejar.CookieJar()))
|
||
if not token and password:
|
||
st, r = self.call('POST', '/api/admin/login', {'password': password})
|
||
if st != 200:
|
||
sys.exit(f'admin login failed: {r}')
|
||
|
||
def call(self, method, path, body=None, raw=False):
|
||
headers = {'Content-Type': 'application/json'}
|
||
if self.token:
|
||
headers['Authorization'] = f'Bearer {self.token}'
|
||
req = urllib.request.Request(self.base + path, method=method, headers=headers,
|
||
data=json.dumps(body).encode() if body is not None else None)
|
||
try:
|
||
with self.opener.open(req, timeout=600) as r:
|
||
data = r.read()
|
||
return r.status, data if raw else json.loads(data or b'{}')
|
||
except urllib.error.HTTPError as e:
|
||
data = e.read()
|
||
try:
|
||
return e.code, json.loads(data or b'{}')
|
||
except ValueError:
|
||
return e.code, {'error': data[:200].decode('utf-8', 'replace')}
|
||
|
||
|
||
def main():
|
||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
|
||
pick = ap.add_mutually_exclusive_group(required=True)
|
||
pick.add_argument('--missing', action='store_true', help='every saved video without lyrics')
|
||
pick.add_argument('--ids', help='comma-separated video ids')
|
||
ap.add_argument('--overwrite', action='store_true', help='replace existing lyrics (kept in history)')
|
||
ap.add_argument('--model', default='large-v3-turbo')
|
||
ap.add_argument('--language', default=None, help='e.g. en, tl (default: auto-detect)')
|
||
ap.add_argument('--threads', type=int, default=os.cpu_count() or 4)
|
||
ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload')
|
||
ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental')
|
||
ap.add_argument('--watch', type=int, default=0, metavar='SECONDS',
|
||
help='keep running: re-check for songs without lyrics every SECONDS (worker mode)')
|
||
ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)')
|
||
ap.add_argument('--no-web', action='store_true', help='skip the LRCLIB lookup and always transcribe')
|
||
args = ap.parse_args()
|
||
if args.watch:
|
||
return watch(args)
|
||
|
||
run_once(args, Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')))
|
||
|
||
|
||
def load_state(path):
|
||
try:
|
||
with open(path) as f:
|
||
return json.load(f)
|
||
except (OSError, ValueError):
|
||
return {}
|
||
|
||
|
||
def save_state(path, state):
|
||
if not path:
|
||
return
|
||
tmp = path + '.tmp'
|
||
with open(tmp, 'w') as f:
|
||
json.dump(state, f)
|
||
os.replace(tmp, path)
|
||
|
||
|
||
def watch(args):
|
||
"""Worker mode: poll for saved songs without lyrics and transcribe them one
|
||
at a time, forever. Runs at low CPU priority; the state file remembers
|
||
instrumentals (never retried) and failures (retried with backoff)."""
|
||
try:
|
||
os.nice(10)
|
||
except OSError:
|
||
pass
|
||
token, password = os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')
|
||
if not (token or password):
|
||
print('lyrics worker: no YTP_TOKEN set (LYRICS_WORKER_TOKEN in the server env) — idle', flush=True)
|
||
while True:
|
||
time.sleep(3600)
|
||
args.missing, args.ids, args.overwrite, args.dry_run = True, None, False, False
|
||
model = None
|
||
while True:
|
||
try:
|
||
api = Api(args.base, token, password)
|
||
st, r = api.call('GET', '/api/admin/media')
|
||
if st != 200:
|
||
raise RuntimeError(f'listing failed ({st}): {r.get("error")}')
|
||
state = load_state(args.state)
|
||
now = time.time()
|
||
todo = [m for m in r['media'] if not m['lyricsLines']
|
||
and state.get(m['id'], {}).get('status') != 'instrumental'
|
||
and state.get(m['id'], {}).get('retry_at', 0) <= now]
|
||
if todo:
|
||
if model is None:
|
||
from faster_whisper import WhisperModel
|
||
print(f'lyrics worker: loading {args.model}', flush=True)
|
||
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
|
||
m = todo[0] # one song per cycle keeps the worker's footprint small
|
||
result = transcribe_one(args, api, model, m['id'])
|
||
entry = state.get(m['id'], {})
|
||
if result.startswith('instrumental'):
|
||
state[m['id']] = {'status': 'instrumental', 'at': now}
|
||
elif result.startswith('saved') or result.startswith('skip'):
|
||
state.pop(m['id'], None)
|
||
else:
|
||
fails = entry.get('fails', 0) + 1
|
||
state[m['id']] = {'status': 'failed', 'fails': fails, 'retry_at': now + min(86400, 900 * 2 ** fails), 'error': result[:200]}
|
||
save_state(args.state, state)
|
||
print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True)
|
||
continue # straight on to the next song
|
||
except Exception as e: # never die: the next cycle retries
|
||
print(f'lyrics worker: {e}', flush=True)
|
||
time.sleep(args.watch)
|
||
|
||
|
||
def run_once(args, api):
|
||
if args.missing:
|
||
st, r = api.call('GET', '/api/admin/media')
|
||
if st != 200:
|
||
sys.exit(f'listing saved videos failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
|
||
todo = [m['id'] for m in r['media'] if args.overwrite or not m['lyricsLines']]
|
||
else:
|
||
todo = [x.strip() for x in args.ids.split(',') if x.strip()]
|
||
print(f'{len(todo)} video(s) to transcribe with {args.model}', flush=True)
|
||
if not todo:
|
||
return
|
||
|
||
from faster_whisper import WhisperModel # imported late: listing works without it
|
||
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
|
||
summary = [(vid, transcribe_one(args, api, model, vid)) for vid in todo]
|
||
print('\n'.join(f'{v} {s}' for v, s in summary))
|
||
|
||
|
||
def transcribe_one(args, api, model, vid):
|
||
"""Give one saved song lyrics. Published (often synced) lyrics from LRCLIB
|
||
beat a machine transcript, so that is tried first; transcription is the
|
||
fallback. Returns a one-line result."""
|
||
st, cur = api.call('GET', f'/api/notes/{vid}')
|
||
live = (cur or {}).get('lyrics') if st == 200 else None
|
||
if live and live['data']['lines'] and not args.overwrite:
|
||
return 'skip: has lyrics'
|
||
if not getattr(args, 'no_web', False):
|
||
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
|
||
if st == 200:
|
||
m = r.get('match') or {}
|
||
return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}"
|
||
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
|
||
if st != 200:
|
||
return f'no cached audio ({st})'
|
||
with tempfile.NamedTemporaryFile(suffix='.m4a') as f:
|
||
f.write(audio)
|
||
f.flush()
|
||
t0 = time.time()
|
||
# vad_filter must stay OFF: it classifies sung music as non-speech
|
||
# and silently drops the whole song.
|
||
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
|
||
beam_size=5, condition_on_previous_text=False)
|
||
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
|
||
for s in segs for w in (s.words or []) if w.word.strip()]
|
||
took = time.time() - t0
|
||
if len(words) < args.min_words:
|
||
return f'instrumental? only {len(words)} words — skipped'
|
||
lines = segment(words)
|
||
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
|
||
head = ' / '.join(l['text'] for l in lines[:3])
|
||
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
|
||
if args.dry_run:
|
||
print(json.dumps(doc, ensure_ascii=False)[:2000])
|
||
return f'dry-run {len(lines)} lines'
|
||
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
|
||
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
|
||
return f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'
|
||
|
||
|
||
if __name__ == '__main__':
|
||
main()
|