Files
ytplayer/scripts/lyrics/retime_lyrics.py

184 lines
7.4 KiB
Python

#!/usr/bin/env python3
"""retime_lyrics.py — give correct words the timings of a machine transcript.
The two sources we can get lyrics from fail in opposite ways:
* the web (LRCLIB plain text, agy) has the RIGHT WORDS and right line breaks,
but usually no timings;
* Whisper has TIMINGS for every line, but mishears words and breaks lines
mid-phrase ("To show for the / years").
This aligns the two: each correct line is matched to the point in the transcript
where it is actually sung, so the result has the right words AND real timings —
no tapping, no agy re-roll.
Matching is done on a WORD stream, not line to line, precisely because the line
breaks disagree. Whisper gives a time per line, so each line's words are spread
across the gap to the next line; a correct line then takes the time of the
transcript word its first words align to.
# current (timed, wrong) lyrics on the server + correct words from a file
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/retime_lyrics.py \\
--id MU5dlCRTLY8 --words correct.txt
# …and write it
… --id MU5dlCRTLY8 --words correct.txt --apply
# take the timings from a backup instead of what is live now
… --id ID --words w.txt --times-from ytplayer-lyrics-backup-….json
"""
import argparse
import datetime as dt
import difflib
import json
import os
import re
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from auto_lyrics import Api # noqa: E402
from lrclib_regen import parse_lrc # noqa: E402
WORD = re.compile(r"[a-z0-9']+")
def words_of(text):
return WORD.findall(str(text or '').lower())
def timed_word_stream(lines, total=0):
"""Timed lines -> [(word, time)], each line's words spread over its span."""
out = []
timed = [l for l in lines if l.get('t') is not None]
for i, line in enumerate(timed):
ws = words_of(line['text'])
if not ws:
continue
start = float(line['t'])
nxt = float(timed[i + 1]['t']) if i + 1 < len(timed) else max(start + 3.0, total or start + 3.0)
span = max(0.25, nxt - start)
step = span / len(ws)
for j, w in enumerate(ws):
out.append((w, round(start + j * step, 2)))
return out
def align(correct_lines, stream):
"""Time each correct line from where its words appear in the stream.
difflib gives the matching blocks between the two word sequences; a correct
line takes the time of the earliest transcript word inside it. Lines whose
words were misheard badly enough to match nothing are interpolated between
their neighbours, so no line is left without a time.
"""
src = [w for w, _ in stream]
tgt, owner = [], [] # every correct word + which line it belongs to
for i, line in enumerate(correct_lines):
for w in words_of(line['text']):
tgt.append(w)
owner.append(i)
times = [None] * len(correct_lines)
sm = difflib.SequenceMatcher(a=src, b=tgt, autojunk=False)
for a0, b0, size in sm.get_matching_blocks():
for k in range(size):
line_i = owner[b0 + k]
t = stream[a0 + k][1]
if times[line_i] is None or t < times[line_i]:
times[line_i] = t
# Monotonic: a later line can never start before an earlier one.
best = 0.0
for i, t in enumerate(times):
if t is None:
continue
times[i] = max(t, best)
best = times[i]
# Fill gaps by spreading evenly between the timed neighbours.
known = [i for i, t in enumerate(times) if t is not None]
if not known:
return times, 0
for i in range(len(times)):
if times[i] is not None:
continue
prev = max((k for k in known if k < i), default=None)
nxt = min((k for k in known if k > i), default=None)
if prev is None:
times[i] = max(0.0, times[nxt] - 2.0)
elif nxt is None:
times[i] = times[prev] + 2.5
else:
step = (times[nxt] - times[prev]) / (nxt - prev)
times[i] = round(times[prev] + step * (i - prev), 2)
return times, len(known)
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
ap.add_argument('--id', required=True, help='video id')
ap.add_argument('--words', required=True, help='file with the correct lyrics (plain or LRC)')
ap.add_argument('--times-from', help='backup JSON to take the timings from (default: what is live)')
ap.add_argument('--apply', action='store_true')
ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', ''))
args = ap.parse_args()
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
st, n = api.call('GET', f'/api/notes/{args.id}')
live = (n or {}).get('lyrics') if st == 200 else None
if args.times_from:
with open(args.times_from, encoding='utf-8') as f:
song = json.load(f)['songs'][args.id]
timed_lines = song['data']['lines']
else:
if not live:
sys.exit('this song has no lyrics to take timings from — pass --times-from')
timed_lines = live['data']['lines']
if not any(l.get('t') is not None for l in timed_lines):
sys.exit('the timing source has no timed lines')
correct = parse_lrc(open(args.words, encoding='utf-8').read())
correct = [l for l in correct if l['text'].strip()]
if not correct:
sys.exit('no lyrics found in --words')
st, sres = api.call('GET', f'/api/streams?v={args.id}')
total = float((((sres or {}).get('data') or {}).get('meta') or {}).get('duration') or 0)
stream = timed_word_stream(timed_lines, total)
times, matched = align(correct, stream)
for line, t in zip(correct, times):
line['t'] = round(float(t), 2)
print(f'{args.id}: {len(correct)} correct lines timed from {len(timed_lines)} transcript lines '
f'({matched} matched directly, {len(correct) - matched} interpolated)\n')
for line in correct[:10]:
print(f' [{line["t"]:7.2f}] {line["text"][:62]}')
if len(correct) > 12:
print(' …')
for line in correct[-2:]:
print(f' [{line["t"]:7.2f}] {line["text"][:62]}')
if total and correct[-1]['t'] > total + 5:
print(f'\n!! last line lands at {correct[-1]["t"]:.0f}s but the song is {total:.0f}s — check before applying')
if not args.apply:
print('\n(dry run — add --apply)')
return
if live:
out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~'))
path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{dt.datetime.now():%Y%m%d-%H%M%S}.json')
with open(path, 'w', encoding='utf-8') as f:
json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base,
'songs': {args.id: {'savedAt': live['updatedAt'], 'rev': live['rev'], 'data': live['data']}}}, f, ensure_ascii=False, indent=1)
print(f'backup → {path}')
doc = {'lines': correct, 'tags': ['correct words, timed from the transcript'], 'offset': 0}
st, rr = api.call('PUT', f'/api/notes/{args.id}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0})
print(f'saved as rev {rr.get("rev")}' if st == 200 else f'write failed {st}: {rr.get("error")}')
if __name__ == '__main__':
main()