Files
ytplayer/scripts/lyrics/agy_lyrics.py

230 lines
12 KiB
Python

#!/usr/bin/env python3
"""agy_lyrics.py — find lyrics for saved songs with agy (the Antigravity CLI).
agy is flat-rate, so this costs nothing per song. It is the LAST resort:
LRCLIB first (free and usually SYNCED), then this. agy searches the open web
(Genius, AZLyrics, hymnary, the artist's own site, YouTube transcripts) and is
asked for LRC-format timings when it can find or derive them; most of the time
it comes back with plain text, which lands UNTIMED and can be timed afterwards
in the admin lyric editor's Tap mode.
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/agy_lyrics.py --missing
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/agy_lyrics.py --ids ZHl6EwSwjv0 --apply
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/agy_lyrics.py --ids ID --from-file words.txt --apply
Nothing is written without --apply, and the current lyrics of every song the run
touches are backed up to a JSON file first (the server also keeps every previous
version as a revision).
"""
import argparse
import datetime as dt
import json
import os
import re
import subprocess
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from auto_lyrics import Api # noqa: E402
from lrclib_regen import clean_artist, clean_title, parse_lrc # noqa: E402
ASK_TOOL = 'agy-bridge__agy_ask'
FETCH_TOOL = 'agy-bridge__fetch_output'
# agy prepends STATUS/SUMMARY and appends its own metadata; neither is lyrics.
META = re.compile(r'^(STATUS:|SUMMARY:|MODEL:|INSTANCE:|AGY-META:|WOULD RUN:|NEXT:|WHY:|ERROR:|\$ )', re.I)
# agy can fail per instance (quota) and still exit 0 through the bridge; its
# error prose would otherwise be parsed as lyrics — one such line even carried
# a [05:37.76] stamp and looked like a perfectly good synced lyric.
FAILED = re.compile(r'(STATUS:\s*error|agy-ask failed|quota exhausted|rc=[1-9])', re.I)
# agy streams its own progress chatter into the answer ("Waiting for task
# execution...", "Background task <uuid> completed"). None of it is lyrics.
CHATTER = re.compile(
r'(background task|waiting for task|task execution|completed with|'
r'[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}|'
r'^\.\.\.|^\s*task-\d+|tool call|searching the web|let me |i (will|.ll) )', re.I)
SECTION = re.compile(r'^\[?\(?(verse|chorus|bridge|intro|outro|pre-?chorus|refrain|tag|vamp|instrumental|repeat|x\d)[^\]\)]*\)?\]?:?$', re.I)
def mcp(tool, payload, timeout=900):
"""Call an MCP tool through the mcpjungle CLI and return `result` text."""
proc = subprocess.run(
['mcpjungle', 'invoke', tool, '--input', json.dumps(payload)],
capture_output=True, text=True, timeout=timeout,
)
# mcpjungle prints the tool's answer on STDERR, not stdout — read both.
out = (proc.stderr or '') + (proc.stdout or '')
# The CLI prints a human block, then "** Structured Content **" + JSON.
m = re.search(r'\*\*\s*Structured Content\s*\*\*\s*(\{.*)', out, re.S)
if m:
try:
return str(json.loads(m.group(1)).get('result', ''))
except json.JSONDecodeError:
pass
return out
def ask_agy(title, artist, duration=0, workdir='.', effort='medium'):
"""Ask agy for one song's lyrics. Returns the raw text ('' when not found)."""
dur = f' (about {int(duration) // 60} minutes {int(duration) % 60} seconds long)' if duration else ''
prompt = (
f'Find the full lyrics for the song "{title}"' + (f' by {artist}' if artist else '') + dur + '. '
'Search the web (Genius, AZLyrics, hymnary, worshiptogether, the artist\'s own site, YouTube transcripts). '
'Return ONLY the lyrics. If you can find or derive a timing for each line, return them in LRC format: '
'every line prefixed with [mm:ss.xx] and one sung line per line. If no timings exist anywhere, return the '
'plain lyrics, one sung line per line, with NO timestamps and NO invented ones. '
'No commentary, no chords, no section labels unless they are actually sung. '
'If you cannot find this exact song with confidence, reply exactly: NOT FOUND'
)
# The bridge rotates across the authenticated agy accounts (agy…agy8) by
# itself and falls back when one is out of quota, so there is no instance
# to choose here; `effort` is the knob that matters — a deeper search is
# what turns up a caption track with real timings.
text = mcp(ASK_TOOL, {'dir': os.path.abspath(workdir), 'prompt': prompt, 'effort': effort})
# Long answers come back truncated with a keep_id — fetch the rest.
keep = re.search(r"fetch_output\(keep_id='([^']+)'", text)
if keep:
full = mcp(FETCH_TOOL, {'keep_id': keep.group(1)})
if len(full) > len(text):
text = full
if FAILED.search(text):
raise RuntimeError('agy did not answer: ' + ' '.join(text.split())[:160])
return '' if 'NOT FOUND' in text.upper()[:400] else text
def clean_lines(text):
"""agy's answer -> lyric lines. Keeps LRC stamps when it supplied them."""
body = []
for raw in str(text or '').splitlines():
t = raw.strip().strip('*_`')
if not t or META.match(t) or SECTION.match(t) or CHATTER.search(t):
continue
if t.startswith(('#', '>', '|', 'http')) or re.fullmatch(r'[-=_*]{3,}', t):
continue
if len(t) > 200: # a paragraph of commentary, not a sung line
continue
body.append(t)
lines = parse_lrc('\n'.join(body))
# Trim a repeated tail ("In Jesus' name" x4 is real; 20 identical lines is not)
return lines[:400]
def score(lines):
"""Rank two agy answers: timed beats untimed, then longer beats shorter."""
return (sum(1 for l in lines if l['t'] is not None) > 0, len(lines))
def summarise(lines):
timed = sum(1 for l in lines if l['t'] is not None)
return f'{len(lines)} lines ({timed} timed)' if timed else f'{len(lines)} lines (untimed)'
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
pick = ap.add_mutually_exclusive_group(required=True)
pick.add_argument('--missing', action='store_true', help='every saved song with no lyrics')
pick.add_argument('--ids', help='comma-separated video ids')
ap.add_argument('--from-file', help='use this file\'s text instead of asking agy (LRC or plain; one song, needs --ids)')
ap.add_argument('--overwrite', action='store_true', help='also replace lyrics a song already has')
ap.add_argument('--apply', action='store_true', help='actually save (default: dry run)')
ap.add_argument('--min-lines', type=int, default=6, help='reject an answer shorter than this')
ap.add_argument('--tries', type=int, default=3, help='ask agy up to N times and keep the best answer')
# 'high' does search harder, but measured at >10 minutes for a single song
# — long enough that batches never finish. 'medium' answers in ~2 minutes
# and found timed lyrics for every song that had any.
ap.add_argument('--effort', default='medium', choices=['low', 'medium', 'high'],
help='how hard agy searches (high is much slower, rarely better)')
ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', ''))
args = ap.parse_args()
if args.from_file and not args.ids:
sys.exit('--from-file needs --ids (it is one song\'s words)')
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
if args.missing:
st, r = api.call('GET', '/api/admin/media')
if st != 200:
sys.exit(f'listing failed ({st}): {r.get("error")} — set YTP_TOKEN or YTP_ADMIN_PASSWORD')
todo = [m['id'] for m in r['media'] if args.overwrite or not m.get('lyricsLines')]
else:
todo = [x.strip() for x in args.ids.split(',') if x.strip()]
if not todo:
sys.exit('nothing to do — every song already has lyrics')
# Back up whatever these songs have now, before anything is replaced.
backup = {}
for vid in todo:
st, n = api.call('GET', f'/api/notes/{vid}')
cur = (n or {}).get('lyrics') if st == 200 else None
if cur:
backup[vid] = {'savedAt': cur['updatedAt'], 'rev': cur['rev'], 'data': cur['data']}
if backup:
out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~'))
path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{dt.datetime.now():%Y%m%d-%H%M%S}.json')
with open(path, 'w', encoding='utf-8') as f:
json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base, 'songs': backup}, f, ensure_ascii=False, indent=1)
print(f'backup: {len(backup)} songs → {path}\n')
saved = skipped = 0
for vid in todo:
st, sres = api.call('GET', f'/api/streams?v={vid}')
meta = ((sres or {}).get('data') or {}).get('meta') or {}
title, artist = clean_title(meta.get('title') or vid), clean_artist(meta.get('channel'))
st, n = api.call('GET', f'/api/notes/{vid}')
cur = (n or {}).get('lyrics') if st == 200 else None
if cur and not args.overwrite:
print(f'{vid} skip: already has {len(cur["data"]["lines"])} lines (use --overwrite)')
skipped += 1
continue
if args.from_file:
lines = clean_lines(open(args.from_file, encoding='utf-8').read())
source = f'from a file ({os.path.basename(args.from_file)})'
else:
print(f'{vid} asking agy about "{title}"' + (f' by {artist}' if artist else '') + ' …', flush=True)
# agy is not deterministic: the same question can come back synced,
# plain, or empty. Ask a few times and keep the best answer — most
# timed lines wins, then most lines.
best, why = [], ''
for attempt in range(max(1, args.tries)):
try:
got = clean_lines(ask_agy(title, artist, float(meta.get('duration') or 0), effort=args.effort))
except RuntimeError as err:
why = str(err)
continue
if score(got) > score(best):
best = got
if any(l['t'] is not None for l in best) and len(best) >= args.min_lines:
break # a timed answer is as good as it gets
if attempt + 1 < max(1, args.tries):
print(f' try {attempt + 1}: {summarise(got) if got else "nothing"} — asking again')
lines = best
if not lines and why:
print(f' {why}')
source = 'from the web via agy'
if len(lines) < args.min_lines:
print(f'{vid} no usable answer ({len(lines)} lines) — left alone')
skipped += 1
continue
synced = any(l['t'] is not None for l in lines)
tag = f'{source} ({"synced" if synced else "untimed"})' + ('' if synced else ' — check and Tap-sync')
print(f'{vid} {summarise(lines)} | {title[:40]}')
if not args.apply:
saved += 1 # "would save" — the dry-run summary says so
print(' ' + ' / '.join(l['text'] for l in lines[:3])[:110])
continue
doc = {'lines': lines, 'tags': [tag], 'offset': 0}
st, rr = api.call('PUT', f'/api/notes/{vid}/lyrics', {'data': doc, 'baseRev': cur['rev'] if cur else 0})
if st == 200:
saved += 1
print(f' saved as rev {rr.get("rev")}')
else:
print(f' write failed {st}: {rr.get("error")}')
print(f'\n{saved} {"saved" if args.apply else "would be saved"}, {skipped} left alone'
+ ('' if args.apply else ' (dry run — add --apply)'))
if __name__ == '__main__':
main()