Add admin analytics, metadata collection, and grouped lyric cues

Queue reviewable Whisper drafts from the song list and lyrics editor. Preserve line breaks within one timed cue across editing, saving, reporting, and service views.

Add storage and listening analytics with a durable metadata collector, related-search depth, video limits, thumbnail storage, and a browsable metadata library.
This commit is contained in:
Jonathan Sykes
2026-10-03 07:54:11 +08:00
parent 716b61ffee
commit 2b38717c05
23 changed files with 1181 additions and 48 deletions

View File

@@ -29,6 +29,7 @@ import urllib.error
import urllib.request
import glob
import http.cookiejar
import threading
FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I)
KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ',
@@ -101,7 +102,7 @@ class Api:
data=json.dumps(body).encode() if body is not None else None)
try:
with self.opener.open(req, timeout=600) as r:
data = r.read()
data = r.read(MAX_AUDIO_BYTES + 1) if raw else r.read()
return r.status, data if raw else json.loads(data or b'{}')
except urllib.error.HTTPError as e:
data = e.read()
@@ -193,9 +194,29 @@ def watch(args):
'error': 'worker died while transcribing (likely out of memory)'}
print(f'lyrics worker: {vid} crashed the previous run — backing off', flush=True)
save_state(args.state, state)
next_auto = 0
def get_model():
nonlocal model
if model is None:
from faster_whisper import WhisperModel
print(f'lyrics worker: loading {args.model}', flush=True)
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
return model
while True:
try:
api = Api(args.base, token, password)
# Explicit admin requests take priority and always use Whisper.
# Poll every ten seconds, independently of the automatic interval.
if token:
st, request = api.call('POST', '/api/lyrics-worker/claim', {})
if st == 200 and request.get('job'):
process_requested(args, api, request['job'], get_model)
continue
if time.time() < next_auto:
time.sleep(min(args.watch, 10))
continue
st, r = api.call('GET', '/api/admin/media')
if st != 200:
raise RuntimeError(f'listing failed ({st}): {r.get("error")}')
@@ -205,10 +226,7 @@ def watch(args):
and state.get(m['id'], {}).get('status') != 'instrumental'
and state.get(m['id'], {}).get('retry_at', 0) <= now]
if todo:
if model is None:
from faster_whisper import WhisperModel
print(f'lyrics worker: loading {args.model}', flush=True)
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
get_model()
m = todo[0] # one song per cycle keeps the worker's footprint small
entry = state.get(m['id'], {})
state[m['id']] = {**entry, 'status': 'in_progress'}
@@ -224,9 +242,58 @@ def watch(args):
save_state(args.state, state)
print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True)
continue # straight on to the next song
next_auto = time.time() + args.watch
except Exception as e: # never die: the next cycle retries
print(f'lyrics worker: {e}', flush=True)
time.sleep(args.watch)
next_auto = time.time() + args.watch
time.sleep(min(args.watch, 10))
def process_requested(args, api, job, get_model):
"""Lease a manual request, heartbeat through model loading, return a draft."""
stopped = threading.Event()
stage = ['loading-model']
path = f'/api/lyrics-worker/jobs/{job["id"]}'
def report(extra=None):
return api.call('POST', path, {'lease': job['lease'], 'stage': stage[0], **(extra or {})})
def heartbeat():
while not stopped.wait(25):
try:
st, _ = report()
if st == 409:
stopped.set()
except Exception:
pass # transient connection failure; the durable lease handles recovery
thread = threading.Thread(target=heartbeat, daemon=True)
thread.start()
try:
model = get_model()
def progress(value):
stage[0] = value
st, _ = report()
if st == 409:
raise RuntimeError('The transcription lease expired.')
result = transcribe_one(args, api, model, job['videoId'], draft_job=job, on_stage=progress)
stopped.set()
thread.join(timeout=2)
payload = {'status': 'complete', 'result': result} if isinstance(result, dict) else {'status': 'failed', 'error': result}
st, response = report(payload)
if st == 400 and payload['status'] == 'complete':
report({'status': 'failed', 'error': response.get('error', 'The transcript was rejected.')[:300]})
if st != 200:
print(f'lyrics worker: request {job["id"]} could not finish ({st}): {response.get("error", "failed")}', flush=True)
except Exception as e:
stopped.set()
try:
report({'status': 'failed', 'error': str(e)[:300]})
except Exception:
pass # the next worker can reclaim an expired request
finally:
stopped.set()
thread.join(timeout=2)
def run_once(args, api):
@@ -247,20 +314,22 @@ def run_once(args, api):
print('\n'.join(f'{v} {s}' for v, s in summary))
def transcribe_one(args, api, model, vid):
def transcribe_one(args, api, model, vid, draft_job=None, on_stage=None):
"""Give one saved song lyrics. Published (often synced) lyrics from LRCLIB
beat a machine transcript, so that is tried first; transcription is the
fallback. Returns a one-line result."""
st, cur = api.call('GET', f'/api/notes/{vid}')
live = (cur or {}).get('lyrics') if st == 200 else None
if live and live['data']['lines'] and not args.overwrite:
if not draft_job and live and live['data']['lines'] and not args.overwrite:
return 'skip: has lyrics'
if not getattr(args, 'no_web', False):
if not draft_job and not getattr(args, 'no_web', False):
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
if st == 200:
m = r.get('match') or {}
return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}"
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
if on_stage:
on_stage('downloading-audio')
st, audio = api.call('GET', draft_job['audioPath'] if draft_job else f'/api/media/{vid}?a=1', raw=True)
if st != 200:
return f'no cached audio ({st})'
if len(audio) > MAX_AUDIO_BYTES:
@@ -269,6 +338,8 @@ def transcribe_one(args, api, model, vid):
f.write(audio)
f.flush()
t0 = time.time()
if on_stage:
on_stage('transcribing')
# vad_filter must stay OFF: it classifies sung music as non-speech
# and silently drops the whole song.
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
@@ -282,6 +353,8 @@ def transcribe_one(args, api, model, vid):
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
head = ' / '.join(l['text'] for l in lines[:3])
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
if draft_job:
return doc
if args.dry_run:
print(json.dumps(doc, ensure_ascii=False)[:2000])
return f'dry-run {len(lines)} lines'

View File

@@ -0,0 +1,66 @@
import importlib.util
import pathlib
import types
import unittest
spec = importlib.util.spec_from_file_location('auto_lyrics', pathlib.Path(__file__).with_name('auto_lyrics.py'))
worker = importlib.util.module_from_spec(spec)
spec.loader.exec_module(worker)
class FakeApi:
def __init__(self):
self.calls = []
def call(self, method, path, body=None, raw=False):
self.calls.append((method, path, body))
if raw:
return 200, b'fixture audio'
if path.startswith('/api/notes/'):
return 200, {'lyrics': {'rev': 5, 'data': {'lines': [{'t': 0, 'text': 'Keep the original'}]}}}
return 200, {'ok': True}
class FakeModel:
def transcribe(self, path, **options):
assert options['vad_filter'] is False
words = [types.SimpleNamespace(word=word, start=i, end=i + .8) for i, word in enumerate('Because You are God You can do anything'.split())]
return [types.SimpleNamespace(words=words)], types.SimpleNamespace(language='en', duration=10)
class WhisperRequests(unittest.TestCase):
def setUp(self):
self.args = types.SimpleNamespace(overwrite=False, no_web=False, language=None, min_words=1, dry_run=False)
self.api = FakeApi()
self.job = {'id': 'fixture-job', 'videoId': '0gfX0dFLaBc', 'lease': 'fixture-lease', 'audioPath': '/api/media/0gfX0dFLaBc?a=1'}
def test_manual_request_uses_whisper_despite_existing_lyrics_and_never_publishes(self):
result = worker.transcribe_one(self.args, self.api, FakeModel(), self.job['videoId'], draft_job=self.job)
self.assertIsInstance(result, dict)
self.assertTrue(result['lines'])
self.assertFalse(any(method == 'PUT' or path.endswith('/lyrics/web') for method, path, body in self.api.calls))
def test_requested_job_reports_stages_and_completes_a_draft(self):
worker.process_requested(self.args, self.api, self.job, lambda: FakeModel())
posts = [body for method, path, body in self.api.calls if path.startswith('/api/lyrics-worker/jobs/')]
self.assertEqual(posts[0]['stage'], 'downloading-audio')
self.assertEqual(posts[1]['stage'], 'transcribing')
self.assertEqual(posts[-1]['status'], 'complete')
self.assertEqual(posts[-1]['lease'], 'fixture-lease')
def test_rejected_draft_is_reported_failed_instead_of_retrying_forever(self):
original = self.api.call
def reject(method, path, body=None, raw=False):
if body and body.get('status') == 'complete':
return 400, {'error': 'too many lines'}
return original(method, path, body, raw)
self.api.call = reject
worker.process_requested(self.args, self.api, self.job, lambda: FakeModel())
self.assertEqual(self.api.calls[-1][2]['status'], 'failed')
self.assertEqual(self.api.calls[-1][2]['error'], 'too many lines')
def test_no_vocals_fails_without_touching_saved_lyrics(self):
self.args.min_words = 25
worker.process_requested(self.args, self.api, self.job, lambda: FakeModel())
last = self.api.calls[-1][2]
self.assertEqual(last['status'], 'failed')
self.assertFalse(any(method == 'PUT' for method, path, body in self.api.calls))
if __name__ == '__main__':
unittest.main()