#!/usr/bin/env python3 """web_lyrics.py — fill in lyrics for saved songs from the web. Two sources, in order: 1. LRCLIB (server side, free, no key) — often SYNCED lyrics. The server does this itself: POST /api/notes//lyrics/web. 2. agy (the Antigravity CLI, flat-rate) — a web search for songs LRCLIB doesn't have; the result is plain text, so those lines land UNTIMED and can be timed later with Tap-sync in the app. Only songs with no lyrics are touched (unless --overwrite). Lyrics fetched from the web are third-party text: fine for a private library, not for redistribution. YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/web_lyrics.py --missing --agy YTP_TOKEN=ytp_… python3 scripts/lyrics/web_lyrics.py --ids ID1,ID2 """ import argparse import json import os import re import subprocess import sys sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from auto_lyrics import Api # noqa: E402 (same tiny HTTP helper) AGY_TOOL = 'agy-bridge__agy_research' def ask_agy(title, artist, timeout=900): """Ask agy to find the lyrics on the web. Returns a list of lines.""" topic = ( f'Find the full song lyrics for "{title}"' + (f' by {artist}' if artist else '') + '. ' 'Search the web (AZLyrics, Genius, Musixmatch, hymnary, the artist\'s own site…) and return ONLY the lyrics ' 'as plain text: one sung line per line, blank line between sections, no chords, no commentary, no timestamps, ' 'no section labels unless they are sung. If you cannot find the exact song with confidence, reply exactly: NOT FOUND' ) out = subprocess.run( ['mcpjungle', 'invoke', AGY_TOOL, '--input', json.dumps({'topic': topic, 'depth': 'quick'})], capture_output=True, text=True, timeout=timeout, ).stdout m = re.search(r'report saved to (\S+)', out) text = '' if m and os.path.exists(m.group(1)): text = open(m.group(1), encoding='utf-8').read() else: text = out if 'NOT FOUND' in text.upper(): return [] # Keep plain sung lines: drop markdown, headings, links and section labels. lines = [] for raw in text.splitlines(): t = raw.strip().strip('*_`') if not t or t.startswith(('#', '>', '|', '-', '=', 'http')): continue if re.fullmatch(r'\[?\(?(verse|chorus|bridge|intro|outro|pre-chorus|refrain|tag|repeat)[^\]\)]*\)?\]?', t, re.I): continue if len(t) > 200: continue lines.append(t) return lines[:400] def main(): ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs')) pick = ap.add_mutually_exclusive_group(required=True) pick.add_argument('--missing', action='store_true', help='every saved song without lyrics') pick.add_argument('--ids', help='comma-separated video ids') ap.add_argument('--overwrite', action='store_true') ap.add_argument('--agy', action='store_true', help='fall back to an agy web search when LRCLIB has nothing') ap.add_argument('--dry-run', action='store_true') args = ap.parse_args() api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')) if args.missing: st, r = api.call('GET', '/api/admin/media') if st != 200: sys.exit(f'listing failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD') todo = [(m['id'], m.get('title', ''), m.get('channel', '')) for m in r['media'] if args.overwrite or not m['lyricsLines']] else: todo = [(x.strip(), '', '') for x in args.ids.split(',') if x.strip()] print(f'{len(todo)} song(s) without lyrics', flush=True) for vid, title, artist in todo: st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)}) if st == 200: m = r.get('match') or {} print(f'{vid} LRCLIB {"synced" if r.get("synced") else "plain"} · {r.get("lines")} lines · {m.get("artist", "")} – {m.get("track", "")}', flush=True) continue if st == 409: print(f'{vid} skip: has lyrics', flush=True) continue if not args.agy: print(f'{vid} no LRCLIB match ({r.get("error", "")[:70]})', flush=True) continue lines = ask_agy(title or vid, artist) if not lines: print(f'{vid} agy: not found', flush=True) continue doc = {'lines': [{'t': None, 'text': l, 'kind': 'line'} for l in lines], 'tags': ['from the web (untimed) — check and Tap-sync'], 'offset': 0} if args.dry_run: print(f'{vid} agy: {len(lines)} lines (dry run)', flush=True) continue st, cur = api.call('GET', f'/api/notes/{vid}') live = (cur or {}).get('lyrics') if st == 200 else None st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0}) print(f'{vid} agy: {len(lines)} untimed lines → {"rev " + str(rr.get("rev")) if st == 200 else rr.get("error")}', flush=True) if __name__ == '__main__': main()