Fetch lyrics from LRCLIB (synced when available) and serve an admin media library of uploaded video and audio with cover art and embedded lyrics

This commit is contained in:
Jonathan Sykes
2026-09-20 05:59:50 +08:00
parent 9c1ec4ac51
commit b983c6ca3c
14 changed files with 963 additions and 25 deletions

View File

@@ -1,8 +1,9 @@
#!/usr/bin/env python3
"""auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics.
Free and offline: faster-whisper (CTranslate2, int8, CPU) runs on this machine —
no API key, no credits. Audio comes from the server's own cache
Lyrics come from LRCLIB first (free, key-less, often SYNCED) and otherwise from
faster-whisper (CTranslate2, int8, CPU) running on this machine — no API key, no
credits either way. Audio comes from the server's own cache
(/api/media/<id>?a=1), results go to /api/notes/<id>/lyrics with baseRev, so a
song someone already has lyrics for is never overwritten unless --overwrite.
@@ -124,6 +125,7 @@ def main():
ap.add_argument('--watch', type=int, default=0, metavar='SECONDS',
help='keep running: re-check for songs without lyrics every SECONDS (worker mode)')
ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)')
ap.add_argument('--no-web', action='store_true', help='skip the LRCLIB lookup and always transcribe')
args = ap.parse_args()
if args.watch:
return watch(args)
@@ -216,11 +218,18 @@ def run_once(args, api):
def transcribe_one(args, api, model, vid):
"""Transcribe one saved song and upload it. Returns a one-line result."""
"""Give one saved song lyrics. Published (often synced) lyrics from LRCLIB
beat a machine transcript, so that is tried first; transcription is the
fallback. Returns a one-line result."""
st, cur = api.call('GET', f'/api/notes/{vid}')
live = (cur or {}).get('lyrics') if st == 200 else None
if live and live['data']['lines'] and not args.overwrite:
return 'skip: has lyrics'
if not getattr(args, 'no_web', False):
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
if st == 200:
m = r.get('match') or {}
return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}"
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
if st != 200:
return f'no cached audio ({st})'

View File

@@ -0,0 +1,113 @@
#!/usr/bin/env python3
"""web_lyrics.py — fill in lyrics for saved songs from the web.
Two sources, in order:
1. LRCLIB (server side, free, no key) — often SYNCED lyrics. The server does
this itself: POST /api/notes/<id>/lyrics/web.
2. agy (the Antigravity CLI, flat-rate) — a web search for songs LRCLIB
doesn't have; the result is plain text, so those lines land UNTIMED and
can be timed later with Tap-sync in the app.
Only songs with no lyrics are touched (unless --overwrite). Lyrics fetched
from the web are third-party text: fine for a private library, not for
redistribution.
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/web_lyrics.py --missing --agy
YTP_TOKEN=ytp_… python3 scripts/lyrics/web_lyrics.py --ids ID1,ID2
"""
import argparse
import json
import os
import re
import subprocess
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from auto_lyrics import Api # noqa: E402 (same tiny HTTP helper)
AGY_TOOL = 'agy-bridge__agy_research'
def ask_agy(title, artist, timeout=900):
"""Ask agy to find the lyrics on the web. Returns a list of lines."""
topic = (
f'Find the full song lyrics for "{title}"' + (f' by {artist}' if artist else '') + '. '
'Search the web (AZLyrics, Genius, Musixmatch, hymnary, the artist\'s own site…) and return ONLY the lyrics '
'as plain text: one sung line per line, blank line between sections, no chords, no commentary, no timestamps, '
'no section labels unless they are sung. If you cannot find the exact song with confidence, reply exactly: NOT FOUND'
)
out = subprocess.run(
['mcpjungle', 'invoke', AGY_TOOL, '--input', json.dumps({'topic': topic, 'depth': 'quick'})],
capture_output=True, text=True, timeout=timeout,
).stdout
m = re.search(r'report saved to (\S+)', out)
text = ''
if m and os.path.exists(m.group(1)):
text = open(m.group(1), encoding='utf-8').read()
else:
text = out
if 'NOT FOUND' in text.upper():
return []
# Keep plain sung lines: drop markdown, headings, links and section labels.
lines = []
for raw in text.splitlines():
t = raw.strip().strip('*_`')
if not t or t.startswith(('#', '>', '|', '-', '=', 'http')):
continue
if re.fullmatch(r'\[?\(?(verse|chorus|bridge|intro|outro|pre-chorus|refrain|tag|repeat)[^\]\)]*\)?\]?', t, re.I):
continue
if len(t) > 200:
continue
lines.append(t)
return lines[:400]
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
pick = ap.add_mutually_exclusive_group(required=True)
pick.add_argument('--missing', action='store_true', help='every saved song without lyrics')
pick.add_argument('--ids', help='comma-separated video ids')
ap.add_argument('--overwrite', action='store_true')
ap.add_argument('--agy', action='store_true', help='fall back to an agy web search when LRCLIB has nothing')
ap.add_argument('--dry-run', action='store_true')
args = ap.parse_args()
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
if args.missing:
st, r = api.call('GET', '/api/admin/media')
if st != 200:
sys.exit(f'listing failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
todo = [(m['id'], m.get('title', ''), m.get('channel', '')) for m in r['media'] if args.overwrite or not m['lyricsLines']]
else:
todo = [(x.strip(), '', '') for x in args.ids.split(',') if x.strip()]
print(f'{len(todo)} song(s) without lyrics', flush=True)
for vid, title, artist in todo:
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
if st == 200:
m = r.get('match') or {}
print(f'{vid} LRCLIB {"synced" if r.get("synced") else "plain"} · {r.get("lines")} lines · {m.get("artist", "")} – {m.get("track", "")}', flush=True)
continue
if st == 409:
print(f'{vid} skip: has lyrics', flush=True)
continue
if not args.agy:
print(f'{vid} no LRCLIB match ({r.get("error", "")[:70]})', flush=True)
continue
lines = ask_agy(title or vid, artist)
if not lines:
print(f'{vid} agy: not found', flush=True)
continue
doc = {'lines': [{'t': None, 'text': l, 'kind': 'line'} for l in lines], 'tags': ['from the web (untimed) — check and Tap-sync'], 'offset': 0}
if args.dry_run:
print(f'{vid} agy: {len(lines)} lines (dry run)', flush=True)
continue
st, cur = api.call('GET', f'/api/notes/{vid}')
live = (cur or {}).get('lyrics') if st == 200 else None
st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0})
print(f'{vid} agy: {len(lines)} untimed lines → {"rev " + str(rr.get("rev")) if st == 200 else rr.get("error")}', flush=True)
if __name__ == '__main__':
main()