From 0aa6c95494a03adb74b335602674f807361c7fdd Mon Sep 17 00:00:00 2001 From: Jonathan Sykes Date: Mon, 21 Sep 2026 20:16:02 +0800 Subject: [PATCH] Time correct lyrics from a machine transcript by aligning the two word streams --- .agents/skills/lyrics-agy/SKILL.md | 35 ++++++ scripts/lyrics/retime_lyrics.py | 183 +++++++++++++++++++++++++++++ 2 files changed, 218 insertions(+) create mode 100644 scripts/lyrics/retime_lyrics.py diff --git a/.agents/skills/lyrics-agy/SKILL.md b/.agents/skills/lyrics-agy/SKILL.md index 8cf0c48..54aa0c8 100644 --- a/.agents/skills/lyrics-agy/SKILL.md +++ b/.agents/skills/lyrics-agy/SKILL.md @@ -114,5 +114,40 @@ Written into `data.tags` so a later run can tell where lyrics came from: | `from the web via agy (untimed) — check and Tap-sync` | agy, words only | | `from a file () (synced\|untimed)` | `--from-file` | +## Right words + real timings: `retime_lyrics.py` + +The two sources fail in opposite ways — the web (LRCLIB plain, agy) has the +right words and line breaks but no timings; Whisper has a time for every line +but mishears words and breaks lines mid-phrase ("To show for the / years"). +Rather than re-rolling agy for a timed answer that may never come, align them: + +```bash +# correct words in a file (LRCLIB plain text, agy output, or pasted lyrics) +python3 scripts/lyrics/retime_lyrics.py --id MU5dlCRTLY8 --words correct.txt +python3 scripts/lyrics/retime_lyrics.py --id MU5dlCRTLY8 --words correct.txt --apply +# take timings from a backup rather than what is live +… --id ID --words w.txt --times-from ytplayer-lyrics-backup-….json +``` + +It matches on a **word stream, not line to line**, precisely because the line +breaks disagree: each transcript line's words are spread across the gap to the +next line, `difflib` aligns the two word sequences, and a correct line takes the +time of the earliest transcript word inside it. Times are forced monotonic, and +lines that matched nothing are interpolated between their neighbours. On "Take +Me to the End": 46 correct lines, 44 timed directly, 2 interpolated — and the +anchors came out identical to Whisper's own line times. + +Check the report before `--apply`: it warns when the last line lands past the +end of the song, which means the alignment slipped. + +## When a song has NO lyrics, check LRCLIB again first + +`lyrics-regenerate` only revisits songs that already have lyrics, and the +lyrics-worker transcribes anything with none — so a song can end up with a +Whisper transcript even though LRCLIB had the real words all along. That is +exactly what happened to "Take Me to the End". Before reaching for agy on a +freshly transcribed song, search LRCLIB by hand; if it has the words, the +`retime_lyrics.py` route above beats everything else. + Related: `lyrics-lookup` (LRCLIB first, then this), `lyrics-regenerate` (replace wrong lyrics from LRCLIB), `deploy-prod`. diff --git a/scripts/lyrics/retime_lyrics.py b/scripts/lyrics/retime_lyrics.py new file mode 100644 index 0000000..f6852a2 --- /dev/null +++ b/scripts/lyrics/retime_lyrics.py @@ -0,0 +1,183 @@ +#!/usr/bin/env python3 +"""retime_lyrics.py — give correct words the timings of a machine transcript. + +The two sources we can get lyrics from fail in opposite ways: + + * the web (LRCLIB plain text, agy) has the RIGHT WORDS and right line breaks, + but usually no timings; + * Whisper has TIMINGS for every line, but mishears words and breaks lines + mid-phrase ("To show for the / years"). + +This aligns the two: each correct line is matched to the point in the transcript +where it is actually sung, so the result has the right words AND real timings — +no tapping, no agy re-roll. + +Matching is done on a WORD stream, not line to line, precisely because the line +breaks disagree. Whisper gives a time per line, so each line's words are spread +across the gap to the next line; a correct line then takes the time of the +transcript word its first words align to. + + # current (timed, wrong) lyrics on the server + correct words from a file + YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/retime_lyrics.py \\ + --id MU5dlCRTLY8 --words correct.txt + + # …and write it + … --id MU5dlCRTLY8 --words correct.txt --apply + # take the timings from a backup instead of what is live now + … --id ID --words w.txt --times-from ytplayer-lyrics-backup-….json +""" +import argparse +import datetime as dt +import difflib +import json +import os +import re +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from auto_lyrics import Api # noqa: E402 +from lrclib_regen import parse_lrc # noqa: E402 + +WORD = re.compile(r"[a-z0-9']+") + + +def words_of(text): + return WORD.findall(str(text or '').lower()) + + +def timed_word_stream(lines, total=0): + """Timed lines -> [(word, time)], each line's words spread over its span.""" + out = [] + timed = [l for l in lines if l.get('t') is not None] + for i, line in enumerate(timed): + ws = words_of(line['text']) + if not ws: + continue + start = float(line['t']) + nxt = float(timed[i + 1]['t']) if i + 1 < len(timed) else max(start + 3.0, total or start + 3.0) + span = max(0.25, nxt - start) + step = span / len(ws) + for j, w in enumerate(ws): + out.append((w, round(start + j * step, 2))) + return out + + +def align(correct_lines, stream): + """Time each correct line from where its words appear in the stream. + + difflib gives the matching blocks between the two word sequences; a correct + line takes the time of the earliest transcript word inside it. Lines whose + words were misheard badly enough to match nothing are interpolated between + their neighbours, so no line is left without a time. + """ + src = [w for w, _ in stream] + tgt, owner = [], [] # every correct word + which line it belongs to + for i, line in enumerate(correct_lines): + for w in words_of(line['text']): + tgt.append(w) + owner.append(i) + + times = [None] * len(correct_lines) + sm = difflib.SequenceMatcher(a=src, b=tgt, autojunk=False) + for a0, b0, size in sm.get_matching_blocks(): + for k in range(size): + line_i = owner[b0 + k] + t = stream[a0 + k][1] + if times[line_i] is None or t < times[line_i]: + times[line_i] = t + + # Monotonic: a later line can never start before an earlier one. + best = 0.0 + for i, t in enumerate(times): + if t is None: + continue + times[i] = max(t, best) + best = times[i] + + # Fill gaps by spreading evenly between the timed neighbours. + known = [i for i, t in enumerate(times) if t is not None] + if not known: + return times, 0 + for i in range(len(times)): + if times[i] is not None: + continue + prev = max((k for k in known if k < i), default=None) + nxt = min((k for k in known if k > i), default=None) + if prev is None: + times[i] = max(0.0, times[nxt] - 2.0) + elif nxt is None: + times[i] = times[prev] + 2.5 + else: + step = (times[nxt] - times[prev]) / (nxt - prev) + times[i] = round(times[prev] + step * (i - prev), 2) + return times, len(known) + + +def main(): + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs')) + ap.add_argument('--id', required=True, help='video id') + ap.add_argument('--words', required=True, help='file with the correct lyrics (plain or LRC)') + ap.add_argument('--times-from', help='backup JSON to take the timings from (default: what is live)') + ap.add_argument('--apply', action='store_true') + ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', '')) + args = ap.parse_args() + + api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')) + st, n = api.call('GET', f'/api/notes/{args.id}') + live = (n or {}).get('lyrics') if st == 200 else None + + if args.times_from: + with open(args.times_from, encoding='utf-8') as f: + song = json.load(f)['songs'][args.id] + timed_lines = song['data']['lines'] + else: + if not live: + sys.exit('this song has no lyrics to take timings from — pass --times-from') + timed_lines = live['data']['lines'] + if not any(l.get('t') is not None for l in timed_lines): + sys.exit('the timing source has no timed lines') + + correct = parse_lrc(open(args.words, encoding='utf-8').read()) + correct = [l for l in correct if l['text'].strip()] + if not correct: + sys.exit('no lyrics found in --words') + + st, sres = api.call('GET', f'/api/streams?v={args.id}') + total = float((((sres or {}).get('data') or {}).get('meta') or {}).get('duration') or 0) + + stream = timed_word_stream(timed_lines, total) + times, matched = align(correct, stream) + for line, t in zip(correct, times): + line['t'] = round(float(t), 2) + + print(f'{args.id}: {len(correct)} correct lines timed from {len(timed_lines)} transcript lines ' + f'({matched} matched directly, {len(correct) - matched} interpolated)\n') + for line in correct[:10]: + print(f' [{line["t"]:7.2f}] {line["text"][:62]}') + if len(correct) > 12: + print(' …') + for line in correct[-2:]: + print(f' [{line["t"]:7.2f}] {line["text"][:62]}') + if total and correct[-1]['t'] > total + 5: + print(f'\n!! last line lands at {correct[-1]["t"]:.0f}s but the song is {total:.0f}s — check before applying') + + if not args.apply: + print('\n(dry run — add --apply)') + return + + if live: + out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~')) + path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{dt.datetime.now():%Y%m%d-%H%M%S}.json') + with open(path, 'w', encoding='utf-8') as f: + json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base, + 'songs': {args.id: {'savedAt': live['updatedAt'], 'rev': live['rev'], 'data': live['data']}}}, f, ensure_ascii=False, indent=1) + print(f'backup → {path}') + + doc = {'lines': correct, 'tags': ['correct words, timed from the transcript'], 'offset': 0} + st, rr = api.call('PUT', f'/api/notes/{args.id}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0}) + print(f'saved as rev {rr.get("rev")}' if st == 200 else f'write failed {st}: {rr.get("error")}') + + +if __name__ == '__main__': + main()