Fetch lyrics from LRCLIB (synced when available) and serve an admin media library of uploaded video and audio with cover art and embedded lyrics
This commit is contained in:
@@ -1,8 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
"""auto_lyrics.py — transcribe saved songs locally and inject them as shared lyrics.
|
||||
|
||||
Free and offline: faster-whisper (CTranslate2, int8, CPU) runs on this machine —
|
||||
no API key, no credits. Audio comes from the server's own cache
|
||||
Lyrics come from LRCLIB first (free, key-less, often SYNCED) and otherwise from
|
||||
faster-whisper (CTranslate2, int8, CPU) running on this machine — no API key, no
|
||||
credits either way. Audio comes from the server's own cache
|
||||
(/api/media/<id>?a=1), results go to /api/notes/<id>/lyrics with baseRev, so a
|
||||
song someone already has lyrics for is never overwritten unless --overwrite.
|
||||
|
||||
@@ -124,6 +125,7 @@ def main():
|
||||
ap.add_argument('--watch', type=int, default=0, metavar='SECONDS',
|
||||
help='keep running: re-check for songs without lyrics every SECONDS (worker mode)')
|
||||
ap.add_argument('--state', default='', help='JSON file remembering instrumentals/failures (worker mode)')
|
||||
ap.add_argument('--no-web', action='store_true', help='skip the LRCLIB lookup and always transcribe')
|
||||
args = ap.parse_args()
|
||||
if args.watch:
|
||||
return watch(args)
|
||||
@@ -216,11 +218,18 @@ def run_once(args, api):
|
||||
|
||||
|
||||
def transcribe_one(args, api, model, vid):
|
||||
"""Transcribe one saved song and upload it. Returns a one-line result."""
|
||||
"""Give one saved song lyrics. Published (often synced) lyrics from LRCLIB
|
||||
beat a machine transcript, so that is tried first; transcription is the
|
||||
fallback. Returns a one-line result."""
|
||||
st, cur = api.call('GET', f'/api/notes/{vid}')
|
||||
live = (cur or {}).get('lyrics') if st == 200 else None
|
||||
if live and live['data']['lines'] and not args.overwrite:
|
||||
return 'skip: has lyrics'
|
||||
if not getattr(args, 'no_web', False):
|
||||
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
|
||||
if st == 200:
|
||||
m = r.get('match') or {}
|
||||
return f"saved rev {r.get('rev')} — LRCLIB {'synced' if r.get('synced') else 'plain'}: {m.get('artist', '')} – {m.get('track', '')}"
|
||||
st, audio = api.call('GET', f'/api/media/{vid}?a=1', raw=True)
|
||||
if st != 200:
|
||||
return f'no cached audio ({st})'
|
||||
|
||||
113
scripts/lyrics/web_lyrics.py
Normal file
113
scripts/lyrics/web_lyrics.py
Normal file
@@ -0,0 +1,113 @@
|
||||
#!/usr/bin/env python3
|
||||
"""web_lyrics.py — fill in lyrics for saved songs from the web.
|
||||
|
||||
Two sources, in order:
|
||||
1. LRCLIB (server side, free, no key) — often SYNCED lyrics. The server does
|
||||
this itself: POST /api/notes/<id>/lyrics/web.
|
||||
2. agy (the Antigravity CLI, flat-rate) — a web search for songs LRCLIB
|
||||
doesn't have; the result is plain text, so those lines land UNTIMED and
|
||||
can be timed later with Tap-sync in the app.
|
||||
|
||||
Only songs with no lyrics are touched (unless --overwrite). Lyrics fetched
|
||||
from the web are third-party text: fine for a private library, not for
|
||||
redistribution.
|
||||
|
||||
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/web_lyrics.py --missing --agy
|
||||
YTP_TOKEN=ytp_… python3 scripts/lyrics/web_lyrics.py --ids ID1,ID2
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from auto_lyrics import Api # noqa: E402 (same tiny HTTP helper)
|
||||
|
||||
AGY_TOOL = 'agy-bridge__agy_research'
|
||||
|
||||
|
||||
def ask_agy(title, artist, timeout=900):
|
||||
"""Ask agy to find the lyrics on the web. Returns a list of lines."""
|
||||
topic = (
|
||||
f'Find the full song lyrics for "{title}"' + (f' by {artist}' if artist else '') + '. '
|
||||
'Search the web (AZLyrics, Genius, Musixmatch, hymnary, the artist\'s own site…) and return ONLY the lyrics '
|
||||
'as plain text: one sung line per line, blank line between sections, no chords, no commentary, no timestamps, '
|
||||
'no section labels unless they are sung. If you cannot find the exact song with confidence, reply exactly: NOT FOUND'
|
||||
)
|
||||
out = subprocess.run(
|
||||
['mcpjungle', 'invoke', AGY_TOOL, '--input', json.dumps({'topic': topic, 'depth': 'quick'})],
|
||||
capture_output=True, text=True, timeout=timeout,
|
||||
).stdout
|
||||
m = re.search(r'report saved to (\S+)', out)
|
||||
text = ''
|
||||
if m and os.path.exists(m.group(1)):
|
||||
text = open(m.group(1), encoding='utf-8').read()
|
||||
else:
|
||||
text = out
|
||||
if 'NOT FOUND' in text.upper():
|
||||
return []
|
||||
# Keep plain sung lines: drop markdown, headings, links and section labels.
|
||||
lines = []
|
||||
for raw in text.splitlines():
|
||||
t = raw.strip().strip('*_`')
|
||||
if not t or t.startswith(('#', '>', '|', '-', '=', 'http')):
|
||||
continue
|
||||
if re.fullmatch(r'\[?\(?(verse|chorus|bridge|intro|outro|pre-chorus|refrain|tag|repeat)[^\]\)]*\)?\]?', t, re.I):
|
||||
continue
|
||||
if len(t) > 200:
|
||||
continue
|
||||
lines.append(t)
|
||||
return lines[:400]
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
|
||||
pick = ap.add_mutually_exclusive_group(required=True)
|
||||
pick.add_argument('--missing', action='store_true', help='every saved song without lyrics')
|
||||
pick.add_argument('--ids', help='comma-separated video ids')
|
||||
ap.add_argument('--overwrite', action='store_true')
|
||||
ap.add_argument('--agy', action='store_true', help='fall back to an agy web search when LRCLIB has nothing')
|
||||
ap.add_argument('--dry-run', action='store_true')
|
||||
args = ap.parse_args()
|
||||
|
||||
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
|
||||
if args.missing:
|
||||
st, r = api.call('GET', '/api/admin/media')
|
||||
if st != 200:
|
||||
sys.exit(f'listing failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
|
||||
todo = [(m['id'], m.get('title', ''), m.get('channel', '')) for m in r['media'] if args.overwrite or not m['lyricsLines']]
|
||||
else:
|
||||
todo = [(x.strip(), '', '') for x in args.ids.split(',') if x.strip()]
|
||||
print(f'{len(todo)} song(s) without lyrics', flush=True)
|
||||
|
||||
for vid, title, artist in todo:
|
||||
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
|
||||
if st == 200:
|
||||
m = r.get('match') or {}
|
||||
print(f'{vid} LRCLIB {"synced" if r.get("synced") else "plain"} · {r.get("lines")} lines · {m.get("artist", "")} – {m.get("track", "")}', flush=True)
|
||||
continue
|
||||
if st == 409:
|
||||
print(f'{vid} skip: has lyrics', flush=True)
|
||||
continue
|
||||
if not args.agy:
|
||||
print(f'{vid} no LRCLIB match ({r.get("error", "")[:70]})', flush=True)
|
||||
continue
|
||||
lines = ask_agy(title or vid, artist)
|
||||
if not lines:
|
||||
print(f'{vid} agy: not found', flush=True)
|
||||
continue
|
||||
doc = {'lines': [{'t': None, 'text': l, 'kind': 'line'} for l in lines], 'tags': ['from the web (untimed) — check and Tap-sync'], 'offset': 0}
|
||||
if args.dry_run:
|
||||
print(f'{vid} agy: {len(lines)} lines (dry run)', flush=True)
|
||||
continue
|
||||
st, cur = api.call('GET', f'/api/notes/{vid}')
|
||||
live = (cur or {}).get('lyrics') if st == 200 else None
|
||||
st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0})
|
||||
print(f'{vid} agy: {len(lines)} untimed lines → {"rev " + str(rr.get("rev")) if st == 200 else rr.get("error")}', flush=True)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in New Issue
Block a user