Add a lyrics-only service mode view with the sung line centred and auto-sized, and a free local faster-whisper script to transcribe saved songs
This commit is contained in:
180
scripts/lyrics/auto_lyrics.py
Normal file
180
scripts/lyrics/auto_lyrics.py
Normal file
@@ -0,0 +1,180 @@
|
||||
#!/usr/bin/env python3
|
||||
"""auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics.
|
||||
|
||||
Free and offline: faster-whisper (CTranslate2, int8, CPU) runs on this machine —
|
||||
no API key, no credits. Audio comes from the server's own cache
|
||||
(/api/media/<id>?a=1), results go to /api/notes/<id>/lyrics with baseRev, so a
|
||||
song someone already has lyrics for is never overwritten unless --overwrite.
|
||||
|
||||
Setup (once):
|
||||
uv venv ~/.local/share/lyrics-asr/.venv --python 3.12
|
||||
~/.local/share/lyrics-asr/.venv/bin/pip install -r scripts/lyrics/requirements.txt
|
||||
Run:
|
||||
YTP_TOKEN=ytp_… .venv/bin/python scripts/lyrics/auto_lyrics.py --missing
|
||||
YTP_ADMIN_PASSWORD=… … --ids KcrXlpg0LKI,65vdbMHh4mo --dry-run
|
||||
Measured on the 16-core WSL laptop: large-v3-turbo ≈ 0.65× real time
|
||||
(6.5-min song in ~4 min); openai-whisper medium was ~2.5× real time.
|
||||
Karaoke / minus-one tracks have no vocals; they are reported as "instrumental"
|
||||
and skipped (their words are on screen — see the repo CLAUDE.md).
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
import http.cookiejar
|
||||
|
||||
FILLER = re.compile(r"^(?:(?:oh|ooh|ohh|oh-oh|ah|ahh|hey|yeah|mm|mm-mm|mm-mm-mm|hmm|whoa|woah|la|na|uh|come on)[\s,.!?-]*)+$", re.I)
|
||||
KEEP_CAP = {'I', "I'm", "I'll", "I've", "I'd", 'You', 'Your', "You're", 'Yours', 'Lord', 'God', 'Jesus', 'Christ',
|
||||
'He', 'His', 'Him', 'Thee', 'Thy', 'Thou', 'Father', 'Spirit', 'Holy', 'Savior', 'Saviour', 'King', 'Lamb',
|
||||
'Earth', 'Heaven', 'Panginoon', 'Diyos', 'Hesus', 'Ikaw', 'Iyo', 'Iyong'}
|
||||
|
||||
|
||||
def segment(words, max_words=9, gap_break=1.0):
|
||||
"""Word timings -> sung lines. Whisper punctuates songs sparsely, so lines
|
||||
break on sentence ends, pauses, capitalised line starts and a length cap;
|
||||
1-2 word fragments (a held note split a phrase) fold into the next line."""
|
||||
ws = [w for w in words if w['text'].strip()]
|
||||
groups, cur = [], []
|
||||
for w in ws:
|
||||
if cur:
|
||||
gap = w['start'] - cur[-1]['end']
|
||||
prev = cur[-1]['text'].strip()
|
||||
word = w['text'].strip()
|
||||
n = len(cur)
|
||||
cap_start = word[:1].isupper() and word.strip('",.!?') not in KEEP_CAP
|
||||
if (re.search(r'[.!?]$', prev) or gap >= gap_break or n >= max_words
|
||||
or (re.search(r'[,;:]$', prev) and n >= 4) or (cap_start and n >= 3)
|
||||
or (word in ('You', 'I') and n >= 5)):
|
||||
groups.append(cur)
|
||||
cur = []
|
||||
cur.append(w)
|
||||
if cur:
|
||||
groups.append(cur)
|
||||
folded, i = [], 0
|
||||
while i < len(groups):
|
||||
g = groups[i]
|
||||
if (len(g) <= 2 and i + 1 < len(groups) and not re.search(r'[.!?]$', g[-1]['text'].strip())
|
||||
and groups[i + 1][0]['start'] - g[-1]['end'] < 4):
|
||||
groups[i + 1] = g + groups[i + 1]
|
||||
else:
|
||||
folded.append(g)
|
||||
i += 1
|
||||
lines = []
|
||||
for g in folded:
|
||||
text = re.sub(r'\s+([,.!?;:])', r'\1', ' '.join(w['text'].strip() for w in g)).strip().rstrip(',;:')
|
||||
if re.sub(r'[\W_]+', '', text):
|
||||
lines.append({'t': round(g[0]['start'], 2), 'end': g[-1]['end'], 'text': text[:300], 'n': len(g)})
|
||||
out = []
|
||||
for i, l in enumerate(lines):
|
||||
nxt = lines[i + 1]['t'] if i + 1 < len(lines) else l['end'] + 99
|
||||
prv = lines[i - 1]['end'] if i else -99
|
||||
if l['n'] == 1 and nxt - l['end'] > 4 and l['t'] - prv > 4:
|
||||
continue # isolated one-word line: intro/outro hallucination
|
||||
if FILLER.match(l['text']):
|
||||
continue
|
||||
out.append({'t': l['t'], 'text': l['text'], 'kind': 'line'})
|
||||
return out
|
||||
|
||||
|
||||
class Api:
|
||||
def __init__(self, base, token=None, password=None):
|
||||
self.base = base.rstrip('/')
|
||||
self.token = token
|
||||
self.opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(http.cookiejar.CookieJar()))
|
||||
if not token and password:
|
||||
st, r = self.call('POST', '/api/admin/login', {'password': password})
|
||||
if st != 200:
|
||||
sys.exit(f'admin login failed: {r}')
|
||||
|
||||
def call(self, method, path, body=None, raw=False):
|
||||
headers = {'Content-Type': 'application/json'}
|
||||
if self.token:
|
||||
headers['Authorization'] = f'Bearer {self.token}'
|
||||
req = urllib.request.Request(self.base + path, method=method, headers=headers,
|
||||
data=json.dumps(body).encode() if body is not None else None)
|
||||
try:
|
||||
with self.opener.open(req, timeout=600) as r:
|
||||
data = r.read()
|
||||
return r.status, data if raw else json.loads(data or b'{}')
|
||||
except urllib.error.HTTPError as e:
|
||||
data = e.read()
|
||||
try:
|
||||
return e.code, json.loads(data or b'{}')
|
||||
except ValueError:
|
||||
return e.code, {'error': data[:200].decode('utf-8', 'replace')}
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
|
||||
pick = ap.add_mutually_exclusive_group(required=True)
|
||||
pick.add_argument('--missing', action='store_true', help='every saved video without lyrics')
|
||||
pick.add_argument('--ids', help='comma-separated video ids')
|
||||
ap.add_argument('--overwrite', action='store_true', help='replace existing lyrics (kept in history)')
|
||||
ap.add_argument('--model', default='large-v3-turbo')
|
||||
ap.add_argument('--language', default=None, help='e.g. en, tl (default: auto-detect)')
|
||||
ap.add_argument('--threads', type=int, default=os.cpu_count() or 4)
|
||||
ap.add_argument('--dry-run', action='store_true', help='transcribe and print, do not upload')
|
||||
ap.add_argument('--min-words', type=int, default=25, help='fewer words = treat as instrumental')
|
||||
args = ap.parse_args()
|
||||
|
||||
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
|
||||
if args.missing:
|
||||
st, r = api.call('GET', '/api/admin/media')
|
||||
if st != 200:
|
||||
sys.exit(f'listing saved videos failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
|
||||
todo = [m['id'] for m in r['media'] if args.overwrite or not m['lyricsLines']]
|
||||
else:
|
||||
todo = [x.strip() for x in args.ids.split(',') if x.strip()]
|
||||
print(f'{len(todo)} video(s) to transcribe with {args.model}', flush=True)
|
||||
if not todo:
|
||||
return
|
||||
|
||||
from faster_whisper import WhisperModel # imported late: listing works without it
|
||||
model = WhisperModel(args.model, device='cpu', compute_type='int8', cpu_threads=args.threads)
|
||||
summary = []
|
||||
for vid in todo:
|
||||
st, cur = api.call('GET', f'/api/notes/{vid}')
|
||||
live = (cur or {}).get('lyrics') if st == 200 else None
|
||||
if live and live['data']['lines'] and not args.overwrite:
|
||||
summary.append((vid, 'skip: has lyrics'))
|
||||
continue
|
||||
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
|
||||
if st != 200:
|
||||
summary.append((vid, f'no cached audio ({st})'))
|
||||
continue
|
||||
with tempfile.NamedTemporaryFile(suffix='.m4a') as f:
|
||||
f.write(audio)
|
||||
f.flush()
|
||||
t0 = time.time()
|
||||
# vad_filter must stay OFF: it classifies sung music as non-speech
|
||||
# and silently drops the whole song.
|
||||
segs, info = model.transcribe(f.name, language=args.language, word_timestamps=True, vad_filter=False,
|
||||
beam_size=5, condition_on_previous_text=False)
|
||||
words = [{'text': w.word.strip(), 'start': w.start, 'end': w.end}
|
||||
for s in segs for w in (s.words or []) if w.word.strip()]
|
||||
took = time.time() - t0
|
||||
if len(words) < args.min_words:
|
||||
summary.append((vid, f'instrumental? only {len(words)} words — skipped'))
|
||||
continue
|
||||
lines = segment(words)
|
||||
doc = {'lines': lines, 'tags': ['auto-transcribed (whisper)'], 'offset': 0}
|
||||
head = ' / '.join(l['text'] for l in lines[:3])
|
||||
print(f'{vid}: {len(lines)} lines, lang={info.language}, {took:.0f}s for {info.duration:.0f}s audio | {head[:100]}', flush=True)
|
||||
if args.dry_run:
|
||||
print(json.dumps(doc, ensure_ascii=False)[:2000])
|
||||
summary.append((vid, f'dry-run {len(lines)} lines'))
|
||||
continue
|
||||
body = {'data': doc, 'baseRev': live['rev'] if live else 0}
|
||||
st, r = api.call('PUT', f'/api/notes/{vid}/lyrics', body)
|
||||
summary.append((vid, f'saved rev {r.get("rev")}' if st == 200 else f'upload failed {st}: {r.get("error")}'))
|
||||
print('\n'.join(f'{v} {s}' for v, s in summary))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
1
scripts/lyrics/requirements.txt
Normal file
1
scripts/lyrics/requirements.txt
Normal file
@@ -0,0 +1 @@
|
||||
faster-whisper>=1.2
|
||||
Reference in New Issue
Block a user