114 lines
5.1 KiB
Python
114 lines
5.1 KiB
Python
#!/usr/bin/env python3
|
||
"""web_lyrics.py — fill in lyrics for saved songs from the web.
|
||
|
||
Two sources, in order:
|
||
1. LRCLIB (server side, free, no key) — often SYNCED lyrics. The server does
|
||
this itself: POST /api/notes/<id>/lyrics/web.
|
||
2. agy (the Antigravity CLI, flat-rate) — a web search for songs LRCLIB
|
||
doesn't have; the result is plain text, so those lines land UNTIMED and
|
||
can be timed later with Tap-sync in the app.
|
||
|
||
Only songs with no lyrics are touched (unless --overwrite). Lyrics fetched
|
||
from the web are third-party text: fine for a private library, not for
|
||
redistribution.
|
||
|
||
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/web_lyrics.py --missing --agy
|
||
YTP_TOKEN=ytp_… python3 scripts/lyrics/web_lyrics.py --ids ID1,ID2
|
||
"""
|
||
import argparse
|
||
import json
|
||
import os
|
||
import re
|
||
import subprocess
|
||
import sys
|
||
|
||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||
from auto_lyrics import Api # noqa: E402 (same tiny HTTP helper)
|
||
|
||
AGY_TOOL = 'agy-bridge__agy_research'
|
||
|
||
|
||
def ask_agy(title, artist, timeout=900):
|
||
"""Ask agy to find the lyrics on the web. Returns a list of lines."""
|
||
topic = (
|
||
f'Find the full song lyrics for "{title}"' + (f' by {artist}' if artist else '') + '. '
|
||
'Search the web (AZLyrics, Genius, Musixmatch, hymnary, the artist\'s own site…) and return ONLY the lyrics '
|
||
'as plain text: one sung line per line, blank line between sections, no chords, no commentary, no timestamps, '
|
||
'no section labels unless they are sung. If you cannot find the exact song with confidence, reply exactly: NOT FOUND'
|
||
)
|
||
out = subprocess.run(
|
||
['mcpjungle', 'invoke', AGY_TOOL, '--input', json.dumps({'topic': topic, 'depth': 'quick'})],
|
||
capture_output=True, text=True, timeout=timeout,
|
||
).stdout
|
||
m = re.search(r'report saved to (\S+)', out)
|
||
text = ''
|
||
if m and os.path.exists(m.group(1)):
|
||
text = open(m.group(1), encoding='utf-8').read()
|
||
else:
|
||
text = out
|
||
if 'NOT FOUND' in text.upper():
|
||
return []
|
||
# Keep plain sung lines: drop markdown, headings, links and section labels.
|
||
lines = []
|
||
for raw in text.splitlines():
|
||
t = raw.strip().strip('*_`')
|
||
if not t or t.startswith(('#', '>', '|', '-', '=', 'http')):
|
||
continue
|
||
if re.fullmatch(r'\[?\(?(verse|chorus|bridge|intro|outro|pre-chorus|refrain|tag|repeat)[^\]\)]*\)?\]?', t, re.I):
|
||
continue
|
||
if len(t) > 200:
|
||
continue
|
||
lines.append(t)
|
||
return lines[:400]
|
||
|
||
|
||
def main():
|
||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
|
||
pick = ap.add_mutually_exclusive_group(required=True)
|
||
pick.add_argument('--missing', action='store_true', help='every saved song without lyrics')
|
||
pick.add_argument('--ids', help='comma-separated video ids')
|
||
ap.add_argument('--overwrite', action='store_true')
|
||
ap.add_argument('--agy', action='store_true', help='fall back to an agy web search when LRCLIB has nothing')
|
||
ap.add_argument('--dry-run', action='store_true')
|
||
args = ap.parse_args()
|
||
|
||
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
|
||
if args.missing:
|
||
st, r = api.call('GET', '/api/admin/media')
|
||
if st != 200:
|
||
sys.exit(f'listing failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
|
||
todo = [(m['id'], m.get('title', ''), m.get('channel', '')) for m in r['media'] if args.overwrite or not m['lyricsLines']]
|
||
else:
|
||
todo = [(x.strip(), '', '') for x in args.ids.split(',') if x.strip()]
|
||
print(f'{len(todo)} song(s) without lyrics', flush=True)
|
||
|
||
for vid, title, artist in todo:
|
||
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
|
||
if st == 200:
|
||
m = r.get('match') or {}
|
||
print(f'{vid} LRCLIB {"synced" if r.get("synced") else "plain"} · {r.get("lines")} lines · {m.get("artist", "")} – {m.get("track", "")}', flush=True)
|
||
continue
|
||
if st == 409:
|
||
print(f'{vid} skip: has lyrics', flush=True)
|
||
continue
|
||
if not args.agy:
|
||
print(f'{vid} no LRCLIB match ({r.get("error", "")[:70]})', flush=True)
|
||
continue
|
||
lines = ask_agy(title or vid, artist)
|
||
if not lines:
|
||
print(f'{vid} agy: not found', flush=True)
|
||
continue
|
||
doc = {'lines': [{'t': None, 'text': l, 'kind': 'line'} for l in lines], 'tags': ['from the web (untimed) — check and Tap-sync'], 'offset': 0}
|
||
if args.dry_run:
|
||
print(f'{vid} agy: {len(lines)} lines (dry run)', flush=True)
|
||
continue
|
||
st, cur = api.call('GET', f'/api/notes/{vid}')
|
||
live = (cur or {}).get('lyrics') if st == 200 else None
|
||
st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0})
|
||
print(f'{vid} agy: {len(lines)} untimed lines → {"rev " + str(rr.get("rev")) if st == 200 else rr.get("error")}', flush=True)
|
||
|
||
|
||
if __name__ == '__main__':
|
||
main()
|