Add admin analytics, metadata collection, and grouped lyric cues
Queue reviewable Whisper drafts from the song list and lyrics editor. Preserve line breaks within one timed cue across editing, saving, reporting, and service views. Add storage and listening analytics with a durable metadata collector, related-search depth, video limits, thumbnail storage, and a browsable metadata library.
This commit is contained in:
@@ -29,6 +29,7 @@ import urllib.error
|
||||
import urllib.request
|
||||
import glob
|
||||
import http.cookiejar
|
||||
import threading
|
||||
|
||||
FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I)
|
||||
KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ',
|
||||
@@ -101,7 +102,7 @@ class Api:
|
||||
data=json.dumps(body).encode() if body is not None else None)
|
||||
try:
|
||||
with self.opener.open(req, timeout=600) as r:
|
||||
data = r.read()
|
||||
data = r.read(MAX_AUDIO_BYTES + 1) if raw else r.read()
|
||||
return r.status, data if raw else json.loads(data or b'{}')
|
||||
except urllib.error.HTTPError as e:
|
||||
data = e.read()
|
||||
@@ -193,9 +194,29 @@ def watch(args):
|
||||
'error': 'worker died while transcribing (likely out of memory)'}
|
||||
print(f'lyrics worker: {vid} crashed the previous run — backing off', flush=True)
|
||||
save_state(args.state, state)
|
||||
next_auto = 0
|
||||
|
||||
def get_model():
|
||||
nonlocal model
|
||||
if model is None:
|
||||
from faster_whisper import WhisperModel
|
||||
print(f'lyrics worker: loading {args.model}', flush=True)
|
||||
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
|
||||
return model
|
||||
|
||||
while True:
|
||||
try:
|
||||
api = Api(args.base, token, password)
|
||||
# Explicit admin requests take priority and always use Whisper.
|
||||
# Poll every ten seconds, independently of the automatic interval.
|
||||
if token:
|
||||
st, request = api.call('POST', '/api/lyrics-worker/claim', {})
|
||||
if st == 200 and request.get('job'):
|
||||
process_requested(args, api, request['job'], get_model)
|
||||
continue
|
||||
if time.time() < next_auto:
|
||||
time.sleep(min(args.watch, 10))
|
||||
continue
|
||||
st, r = api.call('GET', '/api/admin/media')
|
||||
if st != 200:
|
||||
raise RuntimeError(f'listing failed ({st}): {r.get("error")}')
|
||||
@@ -205,10 +226,7 @@ def watch(args):
|
||||
and state.get(m['id'], {}).get('status') != 'instrumental'
|
||||
and state.get(m['id'], {}).get('retry_at', 0) <= now]
|
||||
if todo:
|
||||
if model is None:
|
||||
from faster_whisper import WhisperModel
|
||||
print(f'lyrics worker: loading {args.model}', flush=True)
|
||||
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
|
||||
get_model()
|
||||
m = todo[0] # one song per cycle keeps the worker's footprint small
|
||||
entry = state.get(m['id'], {})
|
||||
state[m['id']] = {**entry, 'status': 'in_progress'}
|
||||
@@ -224,9 +242,58 @@ def watch(args):
|
||||
save_state(args.state, state)
|
||||
print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True)
|
||||
continue # straight on to the next song
|
||||
next_auto = time.time() + args.watch
|
||||
except Exception as e: # never die: the next cycle retries
|
||||
print(f'lyrics worker: {e}', flush=True)
|
||||
time.sleep(args.watch)
|
||||
next_auto = time.time() + args.watch
|
||||
time.sleep(min(args.watch, 10))
|
||||
|
||||
|
||||
def process_requested(args, api, job, get_model):
|
||||
"""Lease a manual request, heartbeat through model loading, return a draft."""
|
||||
stopped = threading.Event()
|
||||
stage = ['loading-model']
|
||||
path = f'/api/lyrics-worker/jobs/{job["id"]}'
|
||||
|
||||
def report(extra=None):
|
||||
return api.call('POST', path, {'lease': job['lease'], 'stage': stage[0], **(extra or {})})
|
||||
|
||||
def heartbeat():
|
||||
while not stopped.wait(25):
|
||||
try:
|
||||
st, _ = report()
|
||||
if st == 409:
|
||||
stopped.set()
|
||||
except Exception:
|
||||
pass # transient connection failure; the durable lease handles recovery
|
||||
|
||||
thread = threading.Thread(target=heartbeat, daemon=True)
|
||||
thread.start()
|
||||
try:
|
||||
model = get_model()
|
||||
def progress(value):
|
||||
stage[0] = value
|
||||
st, _ = report()
|
||||
if st == 409:
|
||||
raise RuntimeError('The transcription lease expired.')
|
||||
result = transcribe_one(args, api, model, job['videoId'], draft_job=job, on_stage=progress)
|
||||
stopped.set()
|
||||
thread.join(timeout=2)
|
||||
payload = {'status': 'complete', 'result': result} if isinstance(result, dict) else {'status': 'failed', 'error': result}
|
||||
st, response = report(payload)
|
||||
if st == 400 and payload['status'] == 'complete':
|
||||
report({'status': 'failed', 'error': response.get('error', 'The transcript was rejected.')[:300]})
|
||||
if st != 200:
|
||||
print(f'lyrics worker: request {job["id"]} could not finish ({st}): {response.get("error", "failed")}', flush=True)
|
||||
except Exception as e:
|
||||
stopped.set()
|
||||
try:
|
||||
report({'status': 'failed', 'error': str(e)[:300]})
|
||||
except Exception:
|
||||
pass # the next worker can reclaim an expired request
|
||||
finally:
|
||||
stopped.set()
|
||||
thread.join(timeout=2)
|
||||
|
||||
|
||||
def run_once(args, api):
|
||||
@@ -247,20 +314,22 @@ def run_once(args, api):
|
||||
print('\n'.join(f'{v} {s}' for v, s in summary))
|
||||
|
||||
|
||||
def transcribe_one(args, api, model, vid):
|
||||
def transcribe_one(args, api, model, vid, draft_job=None, on_stage=None):
|
||||
"""Give one saved song lyrics. Published (often synced) lyrics from LRCLIB
|
||||
beat a machine transcript, so that is tried first; transcription is the
|
||||
fallback. Returns a one-line result."""
|
||||
st, cur = api.call('GET', f'/api/notes/{vid}')
|
||||
live = (cur or {}).get('lyrics') if st == 200 else None
|
||||
if live and live['data']['lines'] and not args.overwrite:
|
||||
if not draft_job and live and live['data']['lines'] and not args.overwrite:
|
||||
return 'skip: has lyrics'
|
||||
if not getattr(args, 'no_web', False):
|
||||
if not draft_job and not getattr(args, 'no_web', False):
|
||||
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
|
||||
if st == 200:
|
||||
m = r.get('match') or {}
|
||||
return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}"
|
||||
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
|
||||
if on_stage:
|
||||
on_stage('downloading-audio')
|
||||
st, audio = api.call('GET', draft_job['audioPath'] if draft_job else f'/api/media/{vid}?a=1', raw=True)
|
||||
if st != 200:
|
||||
return f'no cached audio ({st})'
|
||||
if len(audio) > MAX_AUDIO_BYTES:
|
||||
@@ -269,6 +338,8 @@ def transcribe_one(args, api, model, vid):
|
||||
f.write(audio)
|
||||
f.flush()
|
||||
t0 = time.time()
|
||||
if on_stage:
|
||||
on_stage('transcribing')
|
||||
# vad_filter must stay OFF: it classifies sung music as non-speech
|
||||
# and silently drops the whole song.
|
||||
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
|
||||
@@ -282,6 +353,8 @@ def transcribe_one(args, api, model, vid):
|
||||
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
|
||||
head = ' / '.join(l['text'] for l in lines[:3])
|
||||
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
|
||||
if draft_job:
|
||||
return doc
|
||||
if args.dry_run:
|
||||
print(json.dumps(doc, ensure_ascii=False)[:2000])
|
||||
return f'dry-run {len(lines)} lines'
|
||||
|
||||
66
scripts/lyrics/test_auto_lyrics.py
Normal file
66
scripts/lyrics/test_auto_lyrics.py
Normal file
@@ -0,0 +1,66 @@
|
||||
import importlib.util
|
||||
import pathlib
|
||||
import types
|
||||
import unittest
|
||||
|
||||
spec = importlib.util.spec_from_file_location('auto_lyrics', pathlib.Path(__file__).with_name('auto_lyrics.py'))
|
||||
worker = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(worker)
|
||||
|
||||
class FakeApi:
|
||||
def __init__(self):
|
||||
self.calls = []
|
||||
def call(self, method, path, body=None, raw=False):
|
||||
self.calls.append((method, path, body))
|
||||
if raw:
|
||||
return 200, b'fixture audio'
|
||||
if path.startswith('/api/notes/'):
|
||||
return 200, {'lyrics': {'rev': 5, 'data': {'lines': [{'t': 0, 'text': 'Keep the original'}]}}}
|
||||
return 200, {'ok': True}
|
||||
|
||||
class FakeModel:
|
||||
def transcribe(self, path, **options):
|
||||
assert options['vad_filter'] is False
|
||||
words = [types.SimpleNamespace(word=word, start=i, end=i + .8) for i, word in enumerate('Because You are God You can do anything'.split())]
|
||||
return [types.SimpleNamespace(words=words)], types.SimpleNamespace(language='en', duration=10)
|
||||
|
||||
class WhisperRequests(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.args = types.SimpleNamespace(overwrite=False, no_web=False, language=None, min_words=1, dry_run=False)
|
||||
self.api = FakeApi()
|
||||
self.job = {'id': 'fixture-job', 'videoId': '0gfX0dFLaBc', 'lease': 'fixture-lease', 'audioPath': '/api/media/0gfX0dFLaBc?a=1'}
|
||||
|
||||
def test_manual_request_uses_whisper_despite_existing_lyrics_and_never_publishes(self):
|
||||
result = worker.transcribe_one(self.args, self.api, FakeModel(), self.job['videoId'], draft_job=self.job)
|
||||
self.assertIsInstance(result, dict)
|
||||
self.assertTrue(result['lines'])
|
||||
self.assertFalse(any(method == 'PUT' or path.endswith('/lyrics/web') for method, path, body in self.api.calls))
|
||||
|
||||
def test_requested_job_reports_stages_and_completes_a_draft(self):
|
||||
worker.process_requested(self.args, self.api, self.job, lambda: FakeModel())
|
||||
posts = [body for method, path, body in self.api.calls if path.startswith('/api/lyrics-worker/jobs/')]
|
||||
self.assertEqual(posts[0]['stage'], 'downloading-audio')
|
||||
self.assertEqual(posts[1]['stage'], 'transcribing')
|
||||
self.assertEqual(posts[-1]['status'], 'complete')
|
||||
self.assertEqual(posts[-1]['lease'], 'fixture-lease')
|
||||
|
||||
def test_rejected_draft_is_reported_failed_instead_of_retrying_forever(self):
|
||||
original = self.api.call
|
||||
def reject(method, path, body=None, raw=False):
|
||||
if body and body.get('status') == 'complete':
|
||||
return 400, {'error': 'too many lines'}
|
||||
return original(method, path, body, raw)
|
||||
self.api.call = reject
|
||||
worker.process_requested(self.args, self.api, self.job, lambda: FakeModel())
|
||||
self.assertEqual(self.api.calls[-1][2]['status'], 'failed')
|
||||
self.assertEqual(self.api.calls[-1][2]['error'], 'too many lines')
|
||||
|
||||
def test_no_vocals_fails_without_touching_saved_lyrics(self):
|
||||
self.args.min_words = 25
|
||||
worker.process_requested(self.args, self.api, self.job, lambda: FakeModel())
|
||||
last = self.api.calls[-1][2]
|
||||
self.assertEqual(last['status'], 'failed')
|
||||
self.assertFalse(any(method == 'PUT' for method, path, body in self.api.calls))
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user