Queue reviewable Whisper drafts from the song list and lyrics editor. Preserve line breaks within one timed cue across editing, saving, reporting, and service views. Add storage and listening analytics with a durable metadata collector, related-search depth, video limits, thumbnail storage, and a browsable metadata library.
368 lines
17 KiB
Python
368 lines
17 KiB
Python
#!/usr/bin/env python3
|
||
"""auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics.
|
||
|
||
Lyrics come from LRCLIB first (free, key-less, often SYNCED) and otherwise from
|
||
faster-whisper (CTranslate2, int8, CPU) running on this machine — no API key, no
|
||
credits either way. Audio comes from the server's own cache
|
||
(/api/media/<id>?a=1), results go to /api/notes/<id>/lyrics with baseRev, so a
|
||
song someone already has lyrics for is never overwritten unless --overwrite.
|
||
|
||
Setup (once):
|
||
uv venv ~/.local/share/lyrics-asr/.venv --python 3.12
|
||
~/.local/share/lyrics-asr/.venv/bin/pip install -r scripts/lyrics/requirements.txt
|
||
Run:
|
||
YTP_TOKEN=ytp_… .venv/bin/python scripts/lyrics/auto_lyrics.py --missing
|
||
YTP_ADMIN_PASSWORD=… … --ids KcrXlpg0LKI,65vdbMHh4mo --dry-run
|
||
Measured on the 16-core WSL laptop: large-v3-turbo ≈ 0.65× real time
|
||
(6.5-min song in ~4 min); openai-whisper medium was ~2.5× real time.
|
||
Karaoke / minus-one tracks have no vocals; they are reported as "instrumental"
|
||
and skipped (their words are on screen — see the repo CLAUDE.md).
|
||
"""
|
||
import argparse
|
||
import json
|
||
import os
|
||
import re
|
||
import sys
|
||
import tempfile
|
||
import time
|
||
import urllib.error
|
||
import urllib.request
|
||
import glob
|
||
import http.cookiejar
|
||
import threading
|
||
|
||
FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I)
|
||
KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ',
|
||
'He', 'His', 'Him', 'Thee', 'Thy', 'Thou', 'Father', 'Spirit', 'Holy', 'Savior', 'Saviour', 'King', 'Lamb',
|
||
'Earth', 'Heaven', 'Panginoon', 'Diyos', 'Hesus', 'Ikaw', 'Iyo', 'Iyong'}
|
||
|
||
|
||
def segment(words, max_words=9, gap_break=1.0):
|
||
"""Word timings -> sung lines. Whisper punctuates songs sparsely, so lines
|
||
break on sentence ends, pauses, capitalised line starts and a length cap;
|
||
1-2 word fragments (a held note split a phrase) fold into the next line."""
|
||
ws = [w for w in words if w['text'].strip()]
|
||
groups, cur = [], []
|
||
for w in ws:
|
||
if cur:
|
||
gap = w['start'] - cur[-1]['end']
|
||
prev = cur[-1]['text'].strip()
|
||
word = w['text'].strip()
|
||
n = len(cur)
|
||
cap_start = word[:1].isupper() and word.strip('",.!?') not in KEEP_CAP
|
||
if (re.search(r'[.!?]$', prev) or gap >= gap_break or n >= max_words
|
||
or (re.search(r'[,;:]$', prev) and n >= 4) or (cap_start and n >= 3)
|
||
or (word in ('You', 'I') and n >= 5)):
|
||
groups.append(cur)
|
||
cur = []
|
||
cur.append(w)
|
||
if cur:
|
||
groups.append(cur)
|
||
folded, i = [], 0
|
||
while i < len(groups):
|
||
g = groups[i]
|
||
if (len(g) <= 2 and i + 1 < len(groups) and not re.search(r'[.!?]$', g[-1]['text'].strip())
|
||
and groups[i + 1][0]['start'] - g[-1]['end'] < 4):
|
||
groups[i + 1] = g + groups[i + 1]
|
||
else:
|
||
folded.append(g)
|
||
i += 1
|
||
lines = []
|
||
for g in folded:
|
||
text = re.sub(r'\s+([,.!?;:])', r'\1', ' '.join(w['text'].strip() for w in g)).strip().rstrip(',;:')
|
||
if re.sub(r'[\W_]+', '', text):
|
||
lines.append({'t': round(g[0]['start'], 2), 'end': g[-1]['end'], 'text': text[:300], 'n': len(g)})
|
||
out = []
|
||
for i, l in enumerate(lines):
|
||
nxt = lines[i + 1]['t'] if i + 1 < len(lines) else l['end'] + 99
|
||
prv = lines[i - 1]['end'] if i else -99
|
||
if l['n'] == 1 and nxt - l['end'] > 4 and l['t'] - prv > 4:
|
||
continue # isolated one-word line: intro/outro hallucination
|
||
if FILLER.match(l['text']):
|
||
continue
|
||
out.append({'t': l['t'], 'text': l['text'], 'kind': 'line'})
|
||
return out
|
||
|
||
|
||
class Api:
|
||
def __init__(self, base, token=None, password=None):
|
||
self.base = base.rstrip('/')
|
||
self.token = token
|
||
self.opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(http.cookiejar.CookieJar()))
|
||
if not token and password:
|
||
st, r = self.call('POST', '/api/admin/login', {'password': password})
|
||
if st != 200:
|
||
sys.exit(f'admin login failed: {r}')
|
||
|
||
def call(self, method, path, body=None, raw=False):
|
||
headers = {'Content-Type': 'application/json'}
|
||
if self.token:
|
||
headers['Authorization'] = f'Bearer {self.token}'
|
||
req = urllib.request.Request(self.base + path, method=method, headers=headers,
|
||
data=json.dumps(body).encode() if body is not None else None)
|
||
try:
|
||
with self.opener.open(req, timeout=600) as r:
|
||
data = r.read(MAX_AUDIO_BYTES + 1) if raw else r.read()
|
||
return r.status, data if raw else json.loads(data or b'{}')
|
||
except urllib.error.HTTPError as e:
|
||
data = e.read()
|
||
try:
|
||
return e.code, json.loads(data or b'{}')
|
||
except ValueError:
|
||
return e.code, {'error': data[:200].decode('utf-8', 'replace')}
|
||
|
||
|
||
def main():
|
||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
|
||
pick = ap.add_mutually_exclusive_group(required=True)
|
||
pick.add_argument('--missing', action='store_true', help='every saved video without lyrics')
|
||
pick.add_argument('--ids', help='comma-separated video ids')
|
||
ap.add_argument('--overwrite', action='store_true', help='replace existing lyrics (kept in history)')
|
||
ap.add_argument('--model', default='large-v3-turbo')
|
||
ap.add_argument('--language', default=None, help='e.g. en, tl (default: auto-detect)')
|
||
ap.add_argument('--threads', type=int, default=os.cpu_count() or 4)
|
||
ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload')
|
||
ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental')
|
||
ap.add_argument('--watch', type=int, default=0, metavar='SECONDS',
|
||
help='keep running: re-check for songs without lyrics every SECONDS (worker mode)')
|
||
ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)')
|
||
ap.add_argument('--no-web', action='store_true', help='skip the LRCLIB lookup and always transcribe')
|
||
args = ap.parse_args()
|
||
if args.watch:
|
||
return watch(args)
|
||
|
||
run_once(args, Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')))
|
||
|
||
|
||
# A whisper run killed by the container's memory cap skips the tempfile
|
||
# cleanup, and the restart used to retry the same song forever — 490 OOM
|
||
# kills left 61 GB of copies of one long video in /tmp. Long audio is skipped
|
||
# up front, stale temp files are swept on start, and a song that was in
|
||
# progress when the worker died is recorded as a failure (with backoff).
|
||
TMP_PREFIX = 'ytp-lyrics-'
|
||
MAX_AUDIO_BYTES = int(os.environ.get('LYRICS_MAX_AUDIO_MB', '40')) * 1048576
|
||
|
||
|
||
def sweep_stale_tmp():
|
||
for f in glob.glob(os.path.join(tempfile.gettempdir(), 'tmp*.m4a')) + \
|
||
glob.glob(os.path.join(tempfile.gettempdir(), TMP_PREFIX + '*')):
|
||
try:
|
||
os.remove(f)
|
||
except OSError:
|
||
pass
|
||
|
||
|
||
def load_state(path):
|
||
try:
|
||
with open(path) as f:
|
||
return json.load(f)
|
||
except (OSError, ValueError):
|
||
return {}
|
||
|
||
|
||
def save_state(path, state):
|
||
if not path:
|
||
return
|
||
tmp = path + '.tmp'
|
||
with open(tmp, 'w') as f:
|
||
json.dump(state, f)
|
||
os.replace(tmp, path)
|
||
|
||
|
||
def watch(args):
|
||
"""Worker mode: poll for saved songs without lyrics and transcribe them one
|
||
at a time, forever. Runs at low CPU priority; the state file remembers
|
||
instrumentals (never retried) and failures (retried with backoff)."""
|
||
try:
|
||
os.nice(10)
|
||
except OSError:
|
||
pass
|
||
token, password = os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')
|
||
if not (token or password):
|
||
print('lyrics worker: no YTP_TOKEN set (LYRICS_WORKER_TOKEN in the server env) — idle', flush=True)
|
||
while True:
|
||
time.sleep(3600)
|
||
args.missing, args.ids, args.overwrite, args.dry_run = True, None, False, False
|
||
model = None
|
||
sweep_stale_tmp()
|
||
state = load_state(args.state)
|
||
for vid, entry in state.items():
|
||
if entry.get('status') == 'in_progress': # the worker died mid-song
|
||
fails = entry.get('fails', 0) + 1
|
||
state[vid] = {'status': 'failed', 'fails': fails, 'retry_at': time.time() + min(86400, 900 * 2 ** fails),
|
||
'error': 'worker died while transcribing (likely out of memory)'}
|
||
print(f'lyrics worker: {vid} crashed the previous run — backing off', flush=True)
|
||
save_state(args.state, state)
|
||
next_auto = 0
|
||
|
||
def get_model():
|
||
nonlocal model
|
||
if model is None:
|
||
from faster_whisper import WhisperModel
|
||
print(f'lyrics worker: loading {args.model}', flush=True)
|
||
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
|
||
return model
|
||
|
||
while True:
|
||
try:
|
||
api = Api(args.base, token, password)
|
||
# Explicit admin requests take priority and always use Whisper.
|
||
# Poll every ten seconds, independently of the automatic interval.
|
||
if token:
|
||
st, request = api.call('POST', '/api/lyrics-worker/claim', {})
|
||
if st == 200 and request.get('job'):
|
||
process_requested(args, api, request['job'], get_model)
|
||
continue
|
||
if time.time() < next_auto:
|
||
time.sleep(min(args.watch, 10))
|
||
continue
|
||
st, r = api.call('GET', '/api/admin/media')
|
||
if st != 200:
|
||
raise RuntimeError(f'listing failed ({st}): {r.get("error")}')
|
||
state = load_state(args.state)
|
||
now = time.time()
|
||
todo = [m for m in r['media'] if not m['lyricsLines']
|
||
and state.get(m['id'], {}).get('status') != 'instrumental'
|
||
and state.get(m['id'], {}).get('retry_at', 0) <= now]
|
||
if todo:
|
||
get_model()
|
||
m = todo[0] # one song per cycle keeps the worker's footprint small
|
||
entry = state.get(m['id'], {})
|
||
state[m['id']] = {**entry, 'status': 'in_progress'}
|
||
save_state(args.state, state)
|
||
result = transcribe_one(args, api, model, m['id'])
|
||
if result.startswith('instrumental') or result.startswith('too long'):
|
||
state[m['id']] = {'status': 'instrumental', 'at': now}
|
||
elif result.startswith('saved') or result.startswith('skip'):
|
||
state.pop(m['id'], None)
|
||
else:
|
||
fails = entry.get('fails', 0) + 1
|
||
state[m['id']] = {'status': 'failed', 'fails': fails, 'retry_at': now + min(86400, 900 * 2 ** fails), 'error': result[:200]}
|
||
save_state(args.state, state)
|
||
print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True)
|
||
continue # straight on to the next song
|
||
next_auto = time.time() + args.watch
|
||
except Exception as e: # never die: the next cycle retries
|
||
print(f'lyrics worker: {e}', flush=True)
|
||
next_auto = time.time() + args.watch
|
||
time.sleep(min(args.watch, 10))
|
||
|
||
|
||
def process_requested(args, api, job, get_model):
|
||
"""Lease a manual request, heartbeat through model loading, return a draft."""
|
||
stopped = threading.Event()
|
||
stage = ['loading-model']
|
||
path = f'/api/lyrics-worker/jobs/{job["id"]}'
|
||
|
||
def report(extra=None):
|
||
return api.call('POST', path, {'lease': job['lease'], 'stage': stage[0], **(extra or {})})
|
||
|
||
def heartbeat():
|
||
while not stopped.wait(25):
|
||
try:
|
||
st, _ = report()
|
||
if st == 409:
|
||
stopped.set()
|
||
except Exception:
|
||
pass # transient connection failure; the durable lease handles recovery
|
||
|
||
thread = threading.Thread(target=heartbeat, daemon=True)
|
||
thread.start()
|
||
try:
|
||
model = get_model()
|
||
def progress(value):
|
||
stage[0] = value
|
||
st, _ = report()
|
||
if st == 409:
|
||
raise RuntimeError('The transcription lease expired.')
|
||
result = transcribe_one(args, api, model, job['videoId'], draft_job=job, on_stage=progress)
|
||
stopped.set()
|
||
thread.join(timeout=2)
|
||
payload = {'status': 'complete', 'result': result} if isinstance(result, dict) else {'status': 'failed', 'error': result}
|
||
st, response = report(payload)
|
||
if st == 400 and payload['status'] == 'complete':
|
||
report({'status': 'failed', 'error': response.get('error', 'The transcript was rejected.')[:300]})
|
||
if st != 200:
|
||
print(f'lyrics worker: request {job["id"]} could not finish ({st}): {response.get("error", "failed")}', flush=True)
|
||
except Exception as e:
|
||
stopped.set()
|
||
try:
|
||
report({'status': 'failed', 'error': str(e)[:300]})
|
||
except Exception:
|
||
pass # the next worker can reclaim an expired request
|
||
finally:
|
||
stopped.set()
|
||
thread.join(timeout=2)
|
||
|
||
|
||
def run_once(args, api):
|
||
if args.missing:
|
||
st, r = api.call('GET', '/api/admin/media')
|
||
if st != 200:
|
||
sys.exit(f'listing saved videos failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
|
||
todo = [m['id'] for m in r['media'] if args.overwrite or not m['lyricsLines']]
|
||
else:
|
||
todo = [x.strip() for x in args.ids.split(',') if x.strip()]
|
||
print(f'{len(todo)} video(s) to transcribe with {args.model}', flush=True)
|
||
if not todo:
|
||
return
|
||
|
||
from faster_whisper import WhisperModel # imported late: listing works without it
|
||
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
|
||
summary = [(vid, transcribe_one(args, api, model, vid)) for vid in todo]
|
||
print('\n'.join(f'{v} {s}' for v, s in summary))
|
||
|
||
|
||
def transcribe_one(args, api, model, vid, draft_job=None, on_stage=None):
|
||
"""Give one saved song lyrics. Published (often synced) lyrics from LRCLIB
|
||
beat a machine transcript, so that is tried first; transcription is the
|
||
fallback. Returns a one-line result."""
|
||
st, cur = api.call('GET', f'/api/notes/{vid}')
|
||
live = (cur or {}).get('lyrics') if st == 200 else None
|
||
if not draft_job and live and live['data']['lines'] and not args.overwrite:
|
||
return 'skip: has lyrics'
|
||
if not draft_job and not getattr(args, 'no_web', False):
|
||
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
|
||
if st == 200:
|
||
m = r.get('match') or {}
|
||
return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}"
|
||
if on_stage:
|
||
on_stage('downloading-audio')
|
||
st, audio = api.call('GET', draft_job['audioPath'] if draft_job else f'/api/media/{vid}?a=1', raw=True)
|
||
if st != 200:
|
||
return f'no cached audio ({st})'
|
||
if len(audio) > MAX_AUDIO_BYTES:
|
||
return f'too long: {len(audio) // 1048576} MB audio (cap {MAX_AUDIO_BYTES // 1048576} MB) — skipped'
|
||
with tempfile.NamedTemporaryFile(suffix='.m4a', prefix=TMP_PREFIX) as f:
|
||
f.write(audio)
|
||
f.flush()
|
||
t0 = time.time()
|
||
if on_stage:
|
||
on_stage('transcribing')
|
||
# vad_filter must stay OFF: it classifies sung music as non-speech
|
||
# and silently drops the whole song.
|
||
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
|
||
beam_size=5, condition_on_previous_text=False)
|
||
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
|
||
for s in segs for w in (s.words or []) if w.word.strip()]
|
||
took = time.time() - t0
|
||
if len(words) < args.min_words:
|
||
return f'instrumental? only {len(words)} words — skipped'
|
||
lines = segment(words)
|
||
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
|
||
head = ' / '.join(l['text'] for l in lines[:3])
|
||
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
|
||
if draft_job:
|
||
return doc
|
||
if args.dry_run:
|
||
print(json.dumps(doc, ensure_ascii=False)[:2000])
|
||
return f'dry-run {len(lines)} lines'
|
||
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
|
||
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
|
||
return f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'
|
||
|
||
|
||
if __name__ == '__main__':
|
||
main()
|