Add a lyrics-only service mode view with the sung line centred and auto-sized, and a free local faster-whisper script to transcribe saved songs

This commit is contained in:
Jonathan Sykes
2026-09-19 08:33:00 +08:00
parent b82f17cd46
commit 71fc51fb50
8 changed files with 457 additions and 2 deletions

View File

@@ -0,0 +1,180 @@
#!/usr/bin/env python3
"""auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics.
Free and offline: faster-whisper (CTranslate2, int8, CPU) runs on this machine —
no API key, no credits. Audio comes from the server's own cache
(/api/media/<id>?a=1), results go to /api/notes/<id>/lyrics with baseRev, so a
song someone already has lyrics for is never overwritten unless --overwrite.
Setup (once):
uv venv ~/.local/share/lyrics-asr/.venv --python 3.12
~/.local/share/lyrics-asr/.venv/bin/pip install -r scripts/lyrics/requirements.txt
Run:
YTP_TOKEN=ytp_… .venv/bin/python scripts/lyrics/auto_lyrics.py --missing
YTP_ADMIN_PASSWORD=… … --ids KcrXlpg0LKI,65vdbMHh4mo --dry-run
Measured on the 16-core WSL laptop: large-v3-turbo ≈ 0.65× real time
(6.5-min song in ~4 min); openai-whisper medium was ~2.5× real time.
Karaoke / minus-one tracks have no vocals; they are reported as "instrumental"
and skipped (their words are on screen — see the repo CLAUDE.md).
"""
import argparse
import json
import os
import re
import sys
import tempfile
import time
import urllib.error
import urllib.request
import http.cookiejar
FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I)
KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ',
'He', 'His', 'Him', 'Thee', 'Thy', 'Thou', 'Father', 'Spirit', 'Holy', 'Savior', 'Saviour', 'King', 'Lamb',
'Earth', 'Heaven', 'Panginoon', 'Diyos', 'Hesus', 'Ikaw', 'Iyo', 'Iyong'}
def segment(words, max_words=9, gap_break=1.0):
"""Word timings -> sung lines. Whisper punctuates songs sparsely, so lines
break on sentence ends, pauses, capitalised line starts and a length cap;
1-2 word fragments (a held note split a phrase) fold into the next line."""
ws = [w for w in words if w['text'].strip()]
groups, cur = [], []
for w in ws:
if cur:
gap = w['start'] - cur[-1]['end']
prev = cur[-1]['text'].strip()
word = w['text'].strip()
n = len(cur)
cap_start = word[:1].isupper() and word.strip('",.!?') not in KEEP_CAP
if (re.search(r'[.!?]$', prev) or gap >= gap_break or n >= max_words
or (re.search(r'[,;:]$', prev) and n >= 4) or (cap_start and n >= 3)
or (word in ('You', 'I') and n >= 5)):
groups.append(cur)
cur = []
cur.append(w)
if cur:
groups.append(cur)
folded, i = [], 0
while i < len(groups):
g = groups[i]
if (len(g) <= 2 and i + 1 < len(groups) and not re.search(r'[.!?]$', g[-1]['text'].strip())
and groups[i + 1][0]['start'] - g[-1]['end'] < 4):
groups[i + 1] = g + groups[i + 1]
else:
folded.append(g)
i += 1
lines = []
for g in folded:
text = re.sub(r'\s+([,.!?;:])', r'\1', ' '.join(w['text'].strip() for w in g)).strip().rstrip(',;:')
if re.sub(r'[\W_]+', '', text):
lines.append({'t': round(g[0]['start'], 2), 'end': g[-1]['end'], 'text': text[:300], 'n': len(g)})
out = []
for i, l in enumerate(lines):
nxt = lines[i + 1]['t'] if i + 1 < len(lines) else l['end'] + 99
prv = lines[i - 1]['end'] if i else -99
if l['n'] == 1 and nxt - l['end'] > 4 and l['t'] - prv > 4:
continue # isolated one-word line: intro/outro hallucination
if FILLER.match(l['text']):
continue
out.append({'t': l['t'], 'text': l['text'], 'kind': 'line'})
return out
class Api:
def __init__(self, base, token=None, password=None):
self.base = base.rstrip('/')
self.token = token
self.opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(http.cookiejar.CookieJar()))
if not token and password:
st, r = self.call('POST', '/api/admin/login', {'password': password})
if st != 200:
sys.exit(f'admin login failed: {r}')
def call(self, method, path, body=None, raw=False):
headers = {'Content-Type': 'application/json'}
if self.token:
headers['Authorization'] = f'Bearer {self.token}'
req = urllib.request.Request(self.base + path, method=method, headers=headers,
data=json.dumps(body).encode() if body is not None else None)
try:
with self.opener.open(req, timeout=600) as r:
data = r.read()
return r.status, data if raw else json.loads(data or b'{}')
except urllib.error.HTTPError as e:
data = e.read()
try:
return e.code, json.loads(data or b'{}')
except ValueError:
return e.code, {'error': data[:200].decode('utf-8', 'replace')}
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
pick = ap.add_mutually_exclusive_group(required=True)
pick.add_argument('--missing', action='store_true', help='every saved video without lyrics')
pick.add_argument('--ids', help='comma-separated video ids')
ap.add_argument('--overwrite', action='store_true', help='replace existing lyrics (kept in history)')
ap.add_argument('--model', default='large-v3-turbo')
ap.add_argument('--language', default=None, help='e.g. en, tl (default: auto-detect)')
ap.add_argument('--threads', type=int, default=os.cpu_count() or 4)
ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload')
ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental')
args = ap.parse_args()
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
if args.missing:
st, r = api.call('GET', '/api/admin/media')
if st != 200:
sys.exit(f'listing saved videos failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
todo = [m['id'] for m in r['media'] if args.overwrite or not m['lyricsLines']]
else:
todo = [x.strip() for x in args.ids.split(',') if x.strip()]
print(f'{len(todo)} video(s) to transcribe with {args.model}', flush=True)
if not todo:
return
from faster_whisper import WhisperModel # imported late: listing works without it
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
summary = []
for vid in todo:
st, cur = api.call('GET', f'/api/notes/{vid}')
live = (cur or {}).get('lyrics') if st == 200 else None
if live and live['data']['lines'] and not args.overwrite:
summary.append((vid, 'skip: has lyrics'))
continue
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
if st != 200:
summary.append((vid, f'no cached audio ({st})'))
continue
with tempfile.NamedTemporaryFile(suffix='.m4a') as f:
f.write(audio)
f.flush()
t0 = time.time()
# vad_filter must stay OFF: it classifies sung music as non-speech
# and silently drops the whole song.
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
beam_size=5, condition_on_previous_text=False)
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
for s in segs for w in (s.words or []) if w.word.strip()]
took = time.time() - t0
if len(words) < args.min_words:
summary.append((vid, f'instrumental? only {len(words)} words — skipped'))
continue
lines = segment(words)
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
head = ' / '.join(l['text'] for l in lines[:3])
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
if args.dry_run:
print(json.dumps(doc, ensure_ascii=False)[:2000])
summary.append((vid, f'dry-run {len(lines)} lines'))
continue
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
summary.append((vid, f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'))
print('\n'.join(f'{v} {s}' for v, s in summary))
if __name__ == '__main__':
main()

View File

@@ -0,0 +1 @@
faster-whisper>=1.2