#!/usr/bin/env python3 """retime_lyrics.py — give correct words the timings of a machine transcript. The two sources we can get lyrics from fail in opposite ways: * the web (LRCLIB plain text, agy) has the RIGHT WORDS and right line breaks, but usually no timings; * Whisper has TIMINGS for every line, but mishears words and breaks lines mid-phrase ("To show for the / years"). This aligns the two: each correct line is matched to the point in the transcript where it is actually sung, so the result has the right words AND real timings — no tapping, no agy re-roll. Matching is done on a WORD stream, not line to line, precisely because the line breaks disagree. Whisper gives a time per line, so each line's words are spread across the gap to the next line; a correct line then takes the time of the transcript word its first words align to. # current (timed, wrong) lyrics on the server + correct words from a file YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/retime_lyrics.py \\ --id MU5dlCRTLY8 --words correct.txt # …and write it … --id MU5dlCRTLY8 --words correct.txt --apply # take the timings from a backup instead of what is live now … --id ID --words w.txt --times-from ytplayer-lyrics-backup-….json """ import argparse import datetime as dt import difflib import json import os import re import sys sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from auto_lyrics import Api # noqa: E402 from lrclib_regen import parse_lrc # noqa: E402 WORD = re.compile(r"[a-z0-9']+") def words_of(text): return WORD.findall(str(text or '').lower()) def timed_word_stream(lines, total=0): """Timed lines -> [(word, time)], each line's words spread over its span.""" out = [] timed = [l for l in lines if l.get('t') is not None] for i, line in enumerate(timed): ws = words_of(line['text']) if not ws: continue start = float(line['t']) nxt = float(timed[i + 1]['t']) if i + 1 < len(timed) else max(start + 3.0, total or start + 3.0) span = max(0.25, nxt - start) step = span / len(ws) for j, w in enumerate(ws): out.append((w, round(start + j * step, 2))) return out def align(correct_lines, stream): """Time each correct line from where its words appear in the stream. difflib gives the matching blocks between the two word sequences; a correct line takes the time of the earliest transcript word inside it. Lines whose words were misheard badly enough to match nothing are interpolated between their neighbours, so no line is left without a time. """ src = [w for w, _ in stream] tgt, owner = [], [] # every correct word + which line it belongs to for i, line in enumerate(correct_lines): for w in words_of(line['text']): tgt.append(w) owner.append(i) times = [None] * len(correct_lines) sm = difflib.SequenceMatcher(a=src, b=tgt, autojunk=False) for a0, b0, size in sm.get_matching_blocks(): for k in range(size): line_i = owner[b0 + k] t = stream[a0 + k][1] if times[line_i] is None or t < times[line_i]: times[line_i] = t # Monotonic: a later line can never start before an earlier one. best = 0.0 for i, t in enumerate(times): if t is None: continue times[i] = max(t, best) best = times[i] # Fill gaps by spreading evenly between the timed neighbours. known = [i for i, t in enumerate(times) if t is not None] if not known: return times, 0 for i in range(len(times)): if times[i] is not None: continue prev = max((k for k in known if k < i), default=None) nxt = min((k for k in known if k > i), default=None) if prev is None: times[i] = max(0.0, times[nxt] - 2.0) elif nxt is None: times[i] = times[prev] + 2.5 else: step = (times[nxt] - times[prev]) / (nxt - prev) times[i] = round(times[prev] + step * (i - prev), 2) return times, len(known) def main(): ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs')) ap.add_argument('--id', required=True, help='video id') ap.add_argument('--words', required=True, help='file with the correct lyrics (plain or LRC)') ap.add_argument('--times-from', help='backup JSON to take the timings from (default: what is live)') ap.add_argument('--apply', action='store_true') ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', '')) args = ap.parse_args() api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD')) st, n = api.call('GET', f'/api/notes/{args.id}') live = (n or {}).get('lyrics') if st == 200 else None if args.times_from: with open(args.times_from, encoding='utf-8') as f: song = json.load(f)['songs'][args.id] timed_lines = song['data']['lines'] else: if not live: sys.exit('this song has no lyrics to take timings from — pass --times-from') timed_lines = live['data']['lines'] if not any(l.get('t') is not None for l in timed_lines): sys.exit('the timing source has no timed lines') correct = parse_lrc(open(args.words, encoding='utf-8').read()) correct = [l for l in correct if l['text'].strip()] if not correct: sys.exit('no lyrics found in --words') st, sres = api.call('GET', f'/api/streams?v={args.id}') total = float((((sres or {}).get('data') or {}).get('meta') or {}).get('duration') or 0) stream = timed_word_stream(timed_lines, total) times, matched = align(correct, stream) for line, t in zip(correct, times): line['t'] = round(float(t), 2) print(f'{args.id}: {len(correct)} correct lines timed from {len(timed_lines)} transcript lines ' f'({matched} matched directly, {len(correct) - matched} interpolated)\n') for line in correct[:10]: print(f' [{line["t"]:7.2f}] {line["text"][:62]}') if len(correct) > 12: print(' …') for line in correct[-2:]: print(f' [{line["t"]:7.2f}] {line["text"][:62]}') if total and correct[-1]['t'] > total + 5: print(f'\n!! last line lands at {correct[-1]["t"]:.0f}s but the song is {total:.0f}s — check before applying') if not args.apply: print('\n(dry run — add --apply)') return if live: out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~')) path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{dt.datetime.now():%Y%m%d-%H%M%S}.json') with open(path, 'w', encoding='utf-8') as f: json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base, 'songs': {args.id: {'savedAt': live['updatedAt'], 'rev': live['rev'], 'data': live['data']}}}, f, ensure_ascii=False, indent=1) print(f'backup → {path}') doc = {'lines': correct, 'tags': ['correct words, timed from the transcript'], 'offset': 0} st, rr = api.call('PUT', f'/api/notes/{args.id}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0}) print(f'saved as rev {rr.get("rev")}' if st == 200 else f'write failed {st}: {rr.get("error")}') if __name__ == '__main__': main()