Add presenter view, stats, lyrics worker, watch party with voice chat, timestamp sharing, soundbites, notes, transcript search, PiP, EQ, sleep fade, gestures and external players; fix the crossfade end-of-song race

This commit is contained in:
Jonathan Sykes
2026-09-20 00:07:12 +08:00
parent ca07e1fdf1
commit 9c1ec4ac51
17 changed files with 2912 additions and 111 deletions

View File

@@ -0,0 +1 @@
__pycache__/

15
scripts/lyrics/Dockerfile Normal file
View File

@@ -0,0 +1,15 @@
# Lyrics worker — transcribes saved songs that have no lyrics yet, one at a
# time, with faster-whisper on CPU (no API keys). Talks to the ytplayer
# service over the compose network only; see scripts/lyrics/auto_lyrics.py.
FROM python:3.12-slim
ENV PYTHONUNBUFFERED=1 \
HF_HOME=/models \
PIP_NO_CACHE_DIR=1
WORKDIR /app
COPY requirements.txt .
RUN pip install -r requirements.txt
COPY auto_lyrics.py .
# Model weights (downloaded on first use) and the worker's memory of
# instrumentals/failures live on volumes, so rebuilds don't re-download.
VOLUME ["/models", "/data"]
CMD ["sh", "-c", "exec python auto_lyrics.py --missing --watch ${WATCH_SECONDS:-300} --state /data/state.json --model ${WHISPER_MODEL:-large-v3-turbo} --threads ${WHISPER_THREADS:-2}"]

View File

@@ -121,9 +121,83 @@ def main():
ap.add_argument('--threads', type=int, default=os.cpu_count() or 4)
ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload')
ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental')
ap.add_argument('--watch', type=int, default=0, metavar='SECONDS',
help='keep running: re-check for songs without lyrics every SECONDS (worker mode)')
ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)')
args = ap.parse_args()
if args.watch:
return watch(args)
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
run_once(args, Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')))
def load_state(path):
try:
with open(path) as f:
return json.load(f)
except (OSError, ValueError):
return {}
def save_state(path, state):
if not path:
return
tmp = path + '.tmp'
with open(tmp, 'w') as f:
json.dump(state, f)
os.replace(tmp, path)
def watch(args):
"""Worker mode: poll for saved songs without lyrics and transcribe them one
at a time, forever. Runs at low CPU priority; the state file remembers
instrumentals (never retried) and failures (retried with backoff)."""
try:
os.nice(10)
except OSError:
pass
token, password = os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')
if not (token or password):
print('lyrics worker: no YTP_TOKEN set (LYRICS_WORKER_TOKEN in the server env) — idle', flush=True)
while True:
time.sleep(3600)
args.missing, args.ids, args.overwrite, args.dry_run = True, None, False, False
model = None
while True:
try:
api = Api(args.base, token, password)
st, r = api.call('GET', '/api/admin/media')
if st != 200:
raise RuntimeError(f'listing failed ({st}): {r.get("error")}')
state = load_state(args.state)
now = time.time()
todo = [m for m in r['media'] if not m['lyricsLines']
and state.get(m['id'], {}).get('status') != 'instrumental'
and state.get(m['id'], {}).get('retry_at', 0) <= now]
if todo:
if model is None:
from faster_whisper import WhisperModel
print(f'lyrics worker: loading {args.model}', flush=True)
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
m = todo[0] # one song per cycle keeps the worker's footprint small
result = transcribe_one(args, api, model, m['id'])
entry = state.get(m['id'], {})
if result.startswith('instrumental'):
state[m['id']] = {'status': 'instrumental', 'at': now}
elif result.startswith('saved') or result.startswith('skip'):
state.pop(m['id'], None)
else:
fails = entry.get('fails', 0) + 1
state[m['id']] = {'status': 'failed', 'fails': fails, 'retry_at': now + min(86400, 900 * 2 ** fails), 'error': result[:200]}
save_state(args.state, state)
print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True)
continue # straight on to the next song
except Exception as e: # never die: the next cycle retries
print(f'lyrics worker: {e}', flush=True)
time.sleep(args.watch)
def run_once(args, api):
if args.missing:
st, r = api.call('GET', '/api/admin/media')
if st != 200:
@@ -137,44 +211,43 @@ def main():
from faster_whisper import WhisperModel # imported late: listing works without it
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
summary = []
for vid in todo:
st, cur = api.call('GET', f'/api/notes/{vid}')
live = (cur or {}).get('lyrics') if st == 200 else None
if live and live['data']['lines'] and not args.overwrite:
summary.append((vid, 'skip: has lyrics'))
continue
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
if st != 200:
summary.append((vid, f'no cached audio ({st})'))
continue
with tempfile.NamedTemporaryFile(suffix='.m4a') as f:
f.write(audio)
f.flush()
t0 = time.time()
# vad_filter must stay OFF: it classifies sung music as non-speech
# and silently drops the whole song.
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
beam_size=5, condition_on_previous_text=False)
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
for s in segs for w in (s.words or []) if w.word.strip()]
took = time.time() - t0
if len(words) < args.min_words:
summary.append((vid, f'instrumental? only {len(words)} words — skipped'))
continue
lines = segment(words)
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
head = ' / '.join(l['text'] for l in lines[:3])
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
if args.dry_run:
print(json.dumps(doc, ensure_ascii=False)[:2000])
summary.append((vid, f'dry-run {len(lines)} lines'))
continue
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
summary.append((vid, f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'))
summary = [(vid, transcribe_one(args, api, model, vid)) for vid in todo]
print('\n'.join(f'{v} {s}' for v, s in summary))
def transcribe_one(args, api, model, vid):
"""Transcribe one saved song and upload it. Returns a one-line result."""
st, cur = api.call('GET', f'/api/notes/{vid}')
live = (cur or {}).get('lyrics') if st == 200 else None
if live and live['data']['lines'] and not args.overwrite:
return 'skip: has lyrics'
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
if st != 200:
return f'no cached audio ({st})'
with tempfile.NamedTemporaryFile(suffix='.m4a') as f:
f.write(audio)
f.flush()
t0 = time.time()
# vad_filter must stay OFF: it classifies sung music as non-speech
# and silently drops the whole song.
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
beam_size=5, condition_on_previous_text=False)
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
for s in segs for w in (s.words or []) if w.word.strip()]
took = time.time() - t0
if len(words) < args.min_words:
return f'instrumental? only {len(words)} words — skipped'
lines = segment(words)
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
head = ' / '.join(l['text'] for l in lines[:3])
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
if args.dry_run:
print(json.dumps(doc, ensure_ascii=False)[:2000])
return f'dry-run {len(lines)} lines'
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
return f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'
if __name__ == '__main__':
main()