Add presenter view, stats, lyrics worker, watch party with voice chat, timestamp sharing, soundbites, notes, transcript search, PiP, EQ, sleep fade, gestures and external players; fix the crossfade end-of-song race
This commit is contained in:
1
scripts/lyrics/.dockerignore
Normal file
1
scripts/lyrics/.dockerignore
Normal file
@@ -0,0 +1 @@
|
||||
__pycache__/
|
||||
15
scripts/lyrics/Dockerfile
Normal file
15
scripts/lyrics/Dockerfile
Normal file
@@ -0,0 +1,15 @@
|
||||
# Lyrics worker — transcribes saved songs that have no lyrics yet, one at a
|
||||
# time, with faster-whisper on CPU (no API keys). Talks to the ytplayer
|
||||
# service over the compose network only; see scripts/lyrics/auto_lyrics.py.
|
||||
FROM python:3.12-slim
|
||||
ENV PYTHONUNBUFFERED=1 \
|
||||
HF_HOME=/models \
|
||||
PIP_NO_CACHE_DIR=1
|
||||
WORKDIR /app
|
||||
COPY requirements.txt .
|
||||
RUN pip install -r requirements.txt
|
||||
COPY auto_lyrics.py .
|
||||
# Model weights (downloaded on first use) and the worker's memory of
|
||||
# instrumentals/failures live on volumes, so rebuilds don't re-download.
|
||||
VOLUME ["/models", "/data"]
|
||||
CMD ["sh", "-c", "exec python auto_lyrics.py --missing --watch ${WATCH_SECONDS:-300} --state /data/state.json --model ${WHISPER_MODEL:-large-v3-turbo} --threads ${WHISPER_THREADS:-2}"]
|
||||
@@ -121,9 +121,83 @@ def main():
|
||||
ap.add_argument('--threads', type=int, default=os.cpu_count() or 4)
|
||||
ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload')
|
||||
ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental')
|
||||
ap.add_argument('--watch', type=int, default=0, metavar='SECONDS',
|
||||
help='keep running: re-check for songs without lyrics every SECONDS (worker mode)')
|
||||
ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)')
|
||||
args = ap.parse_args()
|
||||
if args.watch:
|
||||
return watch(args)
|
||||
|
||||
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
|
||||
run_once(args, Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')))
|
||||
|
||||
|
||||
def load_state(path):
|
||||
try:
|
||||
with open(path) as f:
|
||||
return json.load(f)
|
||||
except (OSError, ValueError):
|
||||
return {}
|
||||
|
||||
|
||||
def save_state(path, state):
|
||||
if not path:
|
||||
return
|
||||
tmp = path + '.tmp'
|
||||
with open(tmp, 'w') as f:
|
||||
json.dump(state, f)
|
||||
os.replace(tmp, path)
|
||||
|
||||
|
||||
def watch(args):
|
||||
"""Worker mode: poll for saved songs without lyrics and transcribe them one
|
||||
at a time, forever. Runs at low CPU priority; the state file remembers
|
||||
instrumentals (never retried) and failures (retried with backoff)."""
|
||||
try:
|
||||
os.nice(10)
|
||||
except OSError:
|
||||
pass
|
||||
token, password = os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')
|
||||
if not (token or password):
|
||||
print('lyrics worker: no YTP_TOKEN set (LYRICS_WORKER_TOKEN in the server env) — idle', flush=True)
|
||||
while True:
|
||||
time.sleep(3600)
|
||||
args.missing, args.ids, args.overwrite, args.dry_run = True, None, False, False
|
||||
model = None
|
||||
while True:
|
||||
try:
|
||||
api = Api(args.base, token, password)
|
||||
st, r = api.call('GET', '/api/admin/media')
|
||||
if st != 200:
|
||||
raise RuntimeError(f'listing failed ({st}): {r.get("error")}')
|
||||
state = load_state(args.state)
|
||||
now = time.time()
|
||||
todo = [m for m in r['media'] if not m['lyricsLines']
|
||||
and state.get(m['id'], {}).get('status') != 'instrumental'
|
||||
and state.get(m['id'], {}).get('retry_at', 0) <= now]
|
||||
if todo:
|
||||
if model is None:
|
||||
from faster_whisper import WhisperModel
|
||||
print(f'lyrics worker: loading {args.model}', flush=True)
|
||||
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
|
||||
m = todo[0] # one song per cycle keeps the worker's footprint small
|
||||
result = transcribe_one(args, api, model, m['id'])
|
||||
entry = state.get(m['id'], {})
|
||||
if result.startswith('instrumental'):
|
||||
state[m['id']] = {'status': 'instrumental', 'at': now}
|
||||
elif result.startswith('saved') or result.startswith('skip'):
|
||||
state.pop(m['id'], None)
|
||||
else:
|
||||
fails = entry.get('fails', 0) + 1
|
||||
state[m['id']] = {'status': 'failed', 'fails': fails, 'retry_at': now + min(86400, 900 * 2 ** fails), 'error': result[:200]}
|
||||
save_state(args.state, state)
|
||||
print(f'lyrics worker: {m["id"]} ({m.get("title", "")[:60]}): {result}', flush=True)
|
||||
continue # straight on to the next song
|
||||
except Exception as e: # never die: the next cycle retries
|
||||
print(f'lyrics worker: {e}', flush=True)
|
||||
time.sleep(args.watch)
|
||||
|
||||
|
||||
def run_once(args, api):
|
||||
if args.missing:
|
||||
st, r = api.call('GET', '/api/admin/media')
|
||||
if st != 200:
|
||||
@@ -137,44 +211,43 @@ def main():
|
||||
|
||||
from faster_whisper import WhisperModel # imported late: listing works without it
|
||||
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
|
||||
summary = []
|
||||
for vid in todo:
|
||||
st, cur = api.call('GET', f'/api/notes/{vid}')
|
||||
live = (cur or {}).get('lyrics') if st == 200 else None
|
||||
if live and live['data']['lines'] and not args.overwrite:
|
||||
summary.append((vid, 'skip: has lyrics'))
|
||||
continue
|
||||
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
|
||||
if st != 200:
|
||||
summary.append((vid, f'no cached audio ({st})'))
|
||||
continue
|
||||
with tempfile.NamedTemporaryFile(suffix='.m4a') as f:
|
||||
f.write(audio)
|
||||
f.flush()
|
||||
t0 = time.time()
|
||||
# vad_filter must stay OFF: it classifies sung music as non-speech
|
||||
# and silently drops the whole song.
|
||||
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
|
||||
beam_size=5, condition_on_previous_text=False)
|
||||
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
|
||||
for s in segs for w in (s.words or []) if w.word.strip()]
|
||||
took = time.time() - t0
|
||||
if len(words) < args.min_words:
|
||||
summary.append((vid, f'instrumental? only {len(words)} words — skipped'))
|
||||
continue
|
||||
lines = segment(words)
|
||||
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
|
||||
head = ' / '.join(l['text'] for l in lines[:3])
|
||||
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
|
||||
if args.dry_run:
|
||||
print(json.dumps(doc, ensure_ascii=False)[:2000])
|
||||
summary.append((vid, f'dry-run {len(lines)} lines'))
|
||||
continue
|
||||
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
|
||||
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
|
||||
summary.append((vid, f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'))
|
||||
summary = [(vid, transcribe_one(args, api, model, vid)) for vid in todo]
|
||||
print('\n'.join(f'{v} {s}' for v, s in summary))
|
||||
|
||||
|
||||
def transcribe_one(args, api, model, vid):
|
||||
"""Transcribe one saved song and upload it. Returns a one-line result."""
|
||||
st, cur = api.call('GET', f'/api/notes/{vid}')
|
||||
live = (cur or {}).get('lyrics') if st == 200 else None
|
||||
if live and live['data']['lines'] and not args.overwrite:
|
||||
return 'skip: has lyrics'
|
||||
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
|
||||
if st != 200:
|
||||
return f'no cached audio ({st})'
|
||||
with tempfile.NamedTemporaryFile(suffix='.m4a') as f:
|
||||
f.write(audio)
|
||||
f.flush()
|
||||
t0 = time.time()
|
||||
# vad_filter must stay OFF: it classifies sung music as non-speech
|
||||
# and silently drops the whole song.
|
||||
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
|
||||
beam_size=5, condition_on_previous_text=False)
|
||||
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
|
||||
for s in segs for w in (s.words or []) if w.word.strip()]
|
||||
took = time.time() - t0
|
||||
if len(words) < args.min_words:
|
||||
return f'instrumental? only {len(words)} words — skipped'
|
||||
lines = segment(words)
|
||||
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
|
||||
head = ' / '.join(l['text'] for l in lines[:3])
|
||||
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
|
||||
if args.dry_run:
|
||||
print(json.dumps(doc, ensure_ascii=False)[:2000])
|
||||
return f'dry-run {len(lines)} lines'
|
||||
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
|
||||
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
|
||||
return f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
||||
Reference in New Issue
Block a user