Time correct lyrics from a machine transcript by aligning the two word streams
This commit is contained in:
@@ -114,5 +114,40 @@ Written into `data.tags` so a later run can tell where lyrics came from:
|
||||
| `from the web via agy (untimed) — check and Tap-sync` | agy, words only |
|
||||
| `from a file (<name>) (synced\|untimed)` | `--from-file` |
|
||||
|
||||
## Right words + real timings: `retime_lyrics.py`
|
||||
|
||||
The two sources fail in opposite ways — the web (LRCLIB plain, agy) has the
|
||||
right words and line breaks but no timings; Whisper has a time for every line
|
||||
but mishears words and breaks lines mid-phrase ("To show for the / years").
|
||||
Rather than re-rolling agy for a timed answer that may never come, align them:
|
||||
|
||||
```bash
|
||||
# correct words in a file (LRCLIB plain text, agy output, or pasted lyrics)
|
||||
python3 scripts/lyrics/retime_lyrics.py --id MU5dlCRTLY8 --words correct.txt
|
||||
python3 scripts/lyrics/retime_lyrics.py --id MU5dlCRTLY8 --words correct.txt --apply
|
||||
# take timings from a backup rather than what is live
|
||||
… --id ID --words w.txt --times-from ytplayer-lyrics-backup-….json
|
||||
```
|
||||
|
||||
It matches on a **word stream, not line to line**, precisely because the line
|
||||
breaks disagree: each transcript line's words are spread across the gap to the
|
||||
next line, `difflib` aligns the two word sequences, and a correct line takes the
|
||||
time of the earliest transcript word inside it. Times are forced monotonic, and
|
||||
lines that matched nothing are interpolated between their neighbours. On "Take
|
||||
Me to the End": 46 correct lines, 44 timed directly, 2 interpolated — and the
|
||||
anchors came out identical to Whisper's own line times.
|
||||
|
||||
Check the report before `--apply`: it warns when the last line lands past the
|
||||
end of the song, which means the alignment slipped.
|
||||
|
||||
## When a song has NO lyrics, check LRCLIB again first
|
||||
|
||||
`lyrics-regenerate` only revisits songs that already have lyrics, and the
|
||||
lyrics-worker transcribes anything with none — so a song can end up with a
|
||||
Whisper transcript even though LRCLIB had the real words all along. That is
|
||||
exactly what happened to "Take Me to the End". Before reaching for agy on a
|
||||
freshly transcribed song, search LRCLIB by hand; if it has the words, the
|
||||
`retime_lyrics.py` route above beats everything else.
|
||||
|
||||
Related: `lyrics-lookup` (LRCLIB first, then this), `lyrics-regenerate`
|
||||
(replace wrong lyrics from LRCLIB), `deploy-prod`.
|
||||
|
||||
183
scripts/lyrics/retime_lyrics.py
Normal file
183
scripts/lyrics/retime_lyrics.py
Normal file
@@ -0,0 +1,183 @@
|
||||
#!/usr/bin/env python3
|
||||
"""retime_lyrics.py — give correct words the timings of a machine transcript.
|
||||
|
||||
The two sources we can get lyrics from fail in opposite ways:
|
||||
|
||||
* the web (LRCLIB plain text, agy) has the RIGHT WORDS and right line breaks,
|
||||
but usually no timings;
|
||||
* Whisper has TIMINGS for every line, but mishears words and breaks lines
|
||||
mid-phrase ("To show for the / years").
|
||||
|
||||
This aligns the two: each correct line is matched to the point in the transcript
|
||||
where it is actually sung, so the result has the right words AND real timings —
|
||||
no tapping, no agy re-roll.
|
||||
|
||||
Matching is done on a WORD stream, not line to line, precisely because the line
|
||||
breaks disagree. Whisper gives a time per line, so each line's words are spread
|
||||
across the gap to the next line; a correct line then takes the time of the
|
||||
transcript word its first words align to.
|
||||
|
||||
# current (timed, wrong) lyrics on the server + correct words from a file
|
||||
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/retime_lyrics.py \\
|
||||
--id MU5dlCRTLY8 --words correct.txt
|
||||
|
||||
# …and write it
|
||||
… --id MU5dlCRTLY8 --words correct.txt --apply
|
||||
# take the timings from a backup instead of what is live now
|
||||
… --id ID --words w.txt --times-from ytplayer-lyrics-backup-….json
|
||||
"""
|
||||
import argparse
|
||||
import datetime as dt
|
||||
import difflib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from auto_lyrics import Api # noqa: E402
|
||||
from lrclib_regen import parse_lrc # noqa: E402
|
||||
|
||||
WORD = re.compile(r"[a-z0-9']+")
|
||||
|
||||
|
||||
def words_of(text):
|
||||
return WORD.findall(str(text or '').lower())
|
||||
|
||||
|
||||
def timed_word_stream(lines, total=0):
|
||||
"""Timed lines -> [(word, time)], each line's words spread over its span."""
|
||||
out = []
|
||||
timed = [l for l in lines if l.get('t') is not None]
|
||||
for i, line in enumerate(timed):
|
||||
ws = words_of(line['text'])
|
||||
if not ws:
|
||||
continue
|
||||
start = float(line['t'])
|
||||
nxt = float(timed[i + 1]['t']) if i + 1 < len(timed) else max(start + 3.0, total or start + 3.0)
|
||||
span = max(0.25, nxt - start)
|
||||
step = span / len(ws)
|
||||
for j, w in enumerate(ws):
|
||||
out.append((w, round(start + j * step, 2)))
|
||||
return out
|
||||
|
||||
|
||||
def align(correct_lines, stream):
|
||||
"""Time each correct line from where its words appear in the stream.
|
||||
|
||||
difflib gives the matching blocks between the two word sequences; a correct
|
||||
line takes the time of the earliest transcript word inside it. Lines whose
|
||||
words were misheard badly enough to match nothing are interpolated between
|
||||
their neighbours, so no line is left without a time.
|
||||
"""
|
||||
src = [w for w, _ in stream]
|
||||
tgt, owner = [], [] # every correct word + which line it belongs to
|
||||
for i, line in enumerate(correct_lines):
|
||||
for w in words_of(line['text']):
|
||||
tgt.append(w)
|
||||
owner.append(i)
|
||||
|
||||
times = [None] * len(correct_lines)
|
||||
sm = difflib.SequenceMatcher(a=src, b=tgt, autojunk=False)
|
||||
for a0, b0, size in sm.get_matching_blocks():
|
||||
for k in range(size):
|
||||
line_i = owner[b0 + k]
|
||||
t = stream[a0 + k][1]
|
||||
if times[line_i] is None or t < times[line_i]:
|
||||
times[line_i] = t
|
||||
|
||||
# Monotonic: a later line can never start before an earlier one.
|
||||
best = 0.0
|
||||
for i, t in enumerate(times):
|
||||
if t is None:
|
||||
continue
|
||||
times[i] = max(t, best)
|
||||
best = times[i]
|
||||
|
||||
# Fill gaps by spreading evenly between the timed neighbours.
|
||||
known = [i for i, t in enumerate(times) if t is not None]
|
||||
if not known:
|
||||
return times, 0
|
||||
for i in range(len(times)):
|
||||
if times[i] is not None:
|
||||
continue
|
||||
prev = max((k for k in known if k < i), default=None)
|
||||
nxt = min((k for k in known if k > i), default=None)
|
||||
if prev is None:
|
||||
times[i] = max(0.0, times[nxt] - 2.0)
|
||||
elif nxt is None:
|
||||
times[i] = times[prev] + 2.5
|
||||
else:
|
||||
step = (times[nxt] - times[prev]) / (nxt - prev)
|
||||
times[i] = round(times[prev] + step * (i - prev), 2)
|
||||
return times, len(known)
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
|
||||
ap.add_argument('--id', required=True, help='video id')
|
||||
ap.add_argument('--words', required=True, help='file with the correct lyrics (plain or LRC)')
|
||||
ap.add_argument('--times-from', help='backup JSON to take the timings from (default: what is live)')
|
||||
ap.add_argument('--apply', action='store_true')
|
||||
ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', ''))
|
||||
args = ap.parse_args()
|
||||
|
||||
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
|
||||
st, n = api.call('GET', f'/api/notes/{args.id}')
|
||||
live = (n or {}).get('lyrics') if st == 200 else None
|
||||
|
||||
if args.times_from:
|
||||
with open(args.times_from, encoding='utf-8') as f:
|
||||
song = json.load(f)['songs'][args.id]
|
||||
timed_lines = song['data']['lines']
|
||||
else:
|
||||
if not live:
|
||||
sys.exit('this song has no lyrics to take timings from — pass --times-from')
|
||||
timed_lines = live['data']['lines']
|
||||
if not any(l.get('t') is not None for l in timed_lines):
|
||||
sys.exit('the timing source has no timed lines')
|
||||
|
||||
correct = parse_lrc(open(args.words, encoding='utf-8').read())
|
||||
correct = [l for l in correct if l['text'].strip()]
|
||||
if not correct:
|
||||
sys.exit('no lyrics found in --words')
|
||||
|
||||
st, sres = api.call('GET', f'/api/streams?v={args.id}')
|
||||
total = float((((sres or {}).get('data') or {}).get('meta') or {}).get('duration') or 0)
|
||||
|
||||
stream = timed_word_stream(timed_lines, total)
|
||||
times, matched = align(correct, stream)
|
||||
for line, t in zip(correct, times):
|
||||
line['t'] = round(float(t), 2)
|
||||
|
||||
print(f'{args.id}: {len(correct)} correct lines timed from {len(timed_lines)} transcript lines '
|
||||
f'({matched} matched directly, {len(correct) - matched} interpolated)\n')
|
||||
for line in correct[:10]:
|
||||
print(f' [{line["t"]:7.2f}] {line["text"][:62]}')
|
||||
if len(correct) > 12:
|
||||
print(' …')
|
||||
for line in correct[-2:]:
|
||||
print(f' [{line["t"]:7.2f}] {line["text"][:62]}')
|
||||
if total and correct[-1]['t'] > total + 5:
|
||||
print(f'\n!! last line lands at {correct[-1]["t"]:.0f}s but the song is {total:.0f}s — check before applying')
|
||||
|
||||
if not args.apply:
|
||||
print('\n(dry run — add --apply)')
|
||||
return
|
||||
|
||||
if live:
|
||||
out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~'))
|
||||
path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{dt.datetime.now():%Y%m%d-%H%M%S}.json')
|
||||
with open(path, 'w', encoding='utf-8') as f:
|
||||
json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base,
|
||||
'songs': {args.id: {'savedAt': live['updatedAt'], 'rev': live['rev'], 'data': live['data']}}}, f, ensure_ascii=False, indent=1)
|
||||
print(f'backup → {path}')
|
||||
|
||||
doc = {'lines': correct, 'tags': ['correct words, timed from the transcript'], 'offset': 0}
|
||||
st, rr = api.call('PUT', f'/api/notes/{args.id}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0})
|
||||
print(f'saved as rev {rr.get("rev")}' if st == 200 else f'write failed {st}: {rr.get("error")}')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in New Issue
Block a user