Files
ytplayer/scripts/lyrics/web_lyrics.py

114 lines
5.1 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""web_lyrics.py — fill in lyrics for saved songs from the web.
Two sources, in order:
1. LRCLIB (server side, free, no key) — often SYNCED lyrics. The server does
this itself: POST /api/notes/<id>/lyrics/web.
2. agy (the Antigravity CLI, flat-rate) — a web search for songs LRCLIB
doesn't have; the result is plain text, so those lines land UNTIMED and
can be timed later with Tap-sync in the app.
Only songs with no lyrics are touched (unless --overwrite). Lyrics fetched
from the web are third-party text: fine for a private library, not for
redistribution.
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/web_lyrics.py --missing --agy
YTP_TOKEN=ytp_… python3 scripts/lyrics/web_lyrics.py --ids ID1,ID2
"""
import argparse
import json
import os
import re
import subprocess
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from auto_lyrics import Api # noqa: E402 (same tiny HTTP helper)
AGY_TOOL = 'agy-bridge__agy_research'
def ask_agy(title, artist, timeout=900):
"""Ask agy to find the lyrics on the web. Returns a list of lines."""
topic = (
f'Find the full song lyrics for "{title}"' + (f' by {artist}' if artist else '') + '. '
'Search the web (AZLyrics, Genius, Musixmatch, hymnary, the artist\'s own site…) and return ONLY the lyrics '
'as plain text: one sung line per line, blank line between sections, no chords, no commentary, no timestamps, '
'no section labels unless they are sung. If you cannot find the exact song with confidence, reply exactly: NOT FOUND'
)
out = subprocess.run(
['mcpjungle', 'invoke', AGY_TOOL, '--input', json.dumps({'topic': topic, 'depth': 'quick'})],
capture_output=True, text=True, timeout=timeout,
).stdout
m = re.search(r'report saved to (\S+)', out)
text = ''
if m and os.path.exists(m.group(1)):
text = open(m.group(1), encoding='utf-8').read()
else:
text = out
if 'NOT FOUND' in text.upper():
return []
# Keep plain sung lines: drop markdown, headings, links and section labels.
lines = []
for raw in text.splitlines():
t = raw.strip().strip('*_`')
if not t or t.startswith(('#', '>', '|', '-', '=', 'http')):
continue
if re.fullmatch(r'\[?\(?(verse|chorus|bridge|intro|outro|pre-chorus|refrain|tag|repeat)[^\]\)]*\)?\]?', t, re.I):
continue
if len(t) > 200:
continue
lines.append(t)
return lines[:400]
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
pick = ap.add_mutually_exclusive_group(required=True)
pick.add_argument('--missing', action='store_true', help='every saved song without lyrics')
pick.add_argument('--ids', help='comma-separated video ids')
ap.add_argument('--overwrite', action='store_true')
ap.add_argument('--agy', action='store_true', help='fall back to an agy web search when LRCLIB has nothing')
ap.add_argument('--dry-run', action='store_true')
args = ap.parse_args()
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
if args.missing:
st, r = api.call('GET', '/api/admin/media')
if st != 200:
sys.exit(f'listing failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
todo = [(m['id'], m.get('title', ''), m.get('channel', '')) for m in r['media'] if args.overwrite or not m['lyricsLines']]
else:
todo = [(x.strip(), '', '') for x in args.ids.split(',') if x.strip()]
print(f'{len(todo)} song(s) without lyrics', flush=True)
for vid, title, artist in todo:
st, r = api.call('POST', f'/api/notes/{vid}/lyrics/web', {'overwrite': bool(args.overwrite)})
if st == 200:
m = r.get('match') or {}
print(f'{vid} LRCLIB {"synced" if r.get("synced") else "plain"} · {r.get("lines")} lines · {m.get("artist", "")} – {m.get("track", "")}', flush=True)
continue
if st == 409:
print(f'{vid} skip: has lyrics', flush=True)
continue
if not args.agy:
print(f'{vid} no LRCLIB match ({r.get("error", "")[:70]})', flush=True)
continue
lines = ask_agy(title or vid, artist)
if not lines:
print(f'{vid} agy: not found', flush=True)
continue
doc = {'lines': [{'t': None, 'text': l, 'kind': 'line'} for l in lines], 'tags': ['from the web (untimed) — check and Tap-sync'], 'offset': 0}
if args.dry_run:
print(f'{vid} agy: {len(lines)} lines (dry run)', flush=True)
continue
st, cur = api.call('GET', f'/api/notes/{vid}')
live = (cur or {}).get('lyrics') if st == 200 else None
st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0})
print(f'{vid} agy: {len(lines)} untimed lines → {"rev " + str(rr.get("rev")) if st == 200 else rr.get("error")}', flush=True)
if __name__ == '__main__':
main()