Time correct lyrics from a machine transcript by aligning the two word streams
This commit is contained in:
@@ -114,5 +114,40 @@ Written into `data.tags` so a later run can tell where lyrics came from:
|
|||||||
| `from the web via agy (untimed) — check and Tap-sync` | agy, words only |
|
| `from the web via agy (untimed) — check and Tap-sync` | agy, words only |
|
||||||
| `from a file (<name>) (synced\|untimed)` | `--from-file` |
|
| `from a file (<name>) (synced\|untimed)` | `--from-file` |
|
||||||
|
|
||||||
|
## Right words + real timings: `retime_lyrics.py`
|
||||||
|
|
||||||
|
The two sources fail in opposite ways — the web (LRCLIB plain, agy) has the
|
||||||
|
right words and line breaks but no timings; Whisper has a time for every line
|
||||||
|
but mishears words and breaks lines mid-phrase ("To show for the / years").
|
||||||
|
Rather than re-rolling agy for a timed answer that may never come, align them:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# correct words in a file (LRCLIB plain text, agy output, or pasted lyrics)
|
||||||
|
python3 scripts/lyrics/retime_lyrics.py --id MU5dlCRTLY8 --words correct.txt
|
||||||
|
python3 scripts/lyrics/retime_lyrics.py --id MU5dlCRTLY8 --words correct.txt --apply
|
||||||
|
# take timings from a backup rather than what is live
|
||||||
|
… --id ID --words w.txt --times-from ytplayer-lyrics-backup-….json
|
||||||
|
```
|
||||||
|
|
||||||
|
It matches on a **word stream, not line to line**, precisely because the line
|
||||||
|
breaks disagree: each transcript line's words are spread across the gap to the
|
||||||
|
next line, `difflib` aligns the two word sequences, and a correct line takes the
|
||||||
|
time of the earliest transcript word inside it. Times are forced monotonic, and
|
||||||
|
lines that matched nothing are interpolated between their neighbours. On "Take
|
||||||
|
Me to the End": 46 correct lines, 44 timed directly, 2 interpolated — and the
|
||||||
|
anchors came out identical to Whisper's own line times.
|
||||||
|
|
||||||
|
Check the report before `--apply`: it warns when the last line lands past the
|
||||||
|
end of the song, which means the alignment slipped.
|
||||||
|
|
||||||
|
## When a song has NO lyrics, check LRCLIB again first
|
||||||
|
|
||||||
|
`lyrics-regenerate` only revisits songs that already have lyrics, and the
|
||||||
|
lyrics-worker transcribes anything with none — so a song can end up with a
|
||||||
|
Whisper transcript even though LRCLIB had the real words all along. That is
|
||||||
|
exactly what happened to "Take Me to the End". Before reaching for agy on a
|
||||||
|
freshly transcribed song, search LRCLIB by hand; if it has the words, the
|
||||||
|
`retime_lyrics.py` route above beats everything else.
|
||||||
|
|
||||||
Related: `lyrics-lookup` (LRCLIB first, then this), `lyrics-regenerate`
|
Related: `lyrics-lookup` (LRCLIB first, then this), `lyrics-regenerate`
|
||||||
(replace wrong lyrics from LRCLIB), `deploy-prod`.
|
(replace wrong lyrics from LRCLIB), `deploy-prod`.
|
||||||
|
|||||||
183
scripts/lyrics/retime_lyrics.py
Normal file
183
scripts/lyrics/retime_lyrics.py
Normal file
@@ -0,0 +1,183 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""retime_lyrics.py — give correct words the timings of a machine transcript.
|
||||||
|
|
||||||
|
The two sources we can get lyrics from fail in opposite ways:
|
||||||
|
|
||||||
|
* the web (LRCLIB plain text, agy) has the RIGHT WORDS and right line breaks,
|
||||||
|
but usually no timings;
|
||||||
|
* Whisper has TIMINGS for every line, but mishears words and breaks lines
|
||||||
|
mid-phrase ("To show for the / years").
|
||||||
|
|
||||||
|
This aligns the two: each correct line is matched to the point in the transcript
|
||||||
|
where it is actually sung, so the result has the right words AND real timings —
|
||||||
|
no tapping, no agy re-roll.
|
||||||
|
|
||||||
|
Matching is done on a WORD stream, not line to line, precisely because the line
|
||||||
|
breaks disagree. Whisper gives a time per line, so each line's words are spread
|
||||||
|
across the gap to the next line; a correct line then takes the time of the
|
||||||
|
transcript word its first words align to.
|
||||||
|
|
||||||
|
# current (timed, wrong) lyrics on the server + correct words from a file
|
||||||
|
YTP_ADMIN_PASSWORD=… python3 scripts/lyrics/retime_lyrics.py \\
|
||||||
|
--id MU5dlCRTLY8 --words correct.txt
|
||||||
|
|
||||||
|
# …and write it
|
||||||
|
… --id MU5dlCRTLY8 --words correct.txt --apply
|
||||||
|
# take the timings from a backup instead of what is live now
|
||||||
|
… --id ID --words w.txt --times-from ytplayer-lyrics-backup-….json
|
||||||
|
"""
|
||||||
|
import argparse
|
||||||
|
import datetime as dt
|
||||||
|
import difflib
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||||
|
from auto_lyrics import Api # noqa: E402
|
||||||
|
from lrclib_regen import parse_lrc # noqa: E402
|
||||||
|
|
||||||
|
WORD = re.compile(r"[a-z0-9']+")
|
||||||
|
|
||||||
|
|
||||||
|
def words_of(text):
|
||||||
|
return WORD.findall(str(text or '').lower())
|
||||||
|
|
||||||
|
|
||||||
|
def timed_word_stream(lines, total=0):
|
||||||
|
"""Timed lines -> [(word, time)], each line's words spread over its span."""
|
||||||
|
out = []
|
||||||
|
timed = [l for l in lines if l.get('t') is not None]
|
||||||
|
for i, line in enumerate(timed):
|
||||||
|
ws = words_of(line['text'])
|
||||||
|
if not ws:
|
||||||
|
continue
|
||||||
|
start = float(line['t'])
|
||||||
|
nxt = float(timed[i + 1]['t']) if i + 1 < len(timed) else max(start + 3.0, total or start + 3.0)
|
||||||
|
span = max(0.25, nxt - start)
|
||||||
|
step = span / len(ws)
|
||||||
|
for j, w in enumerate(ws):
|
||||||
|
out.append((w, round(start + j * step, 2)))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def align(correct_lines, stream):
|
||||||
|
"""Time each correct line from where its words appear in the stream.
|
||||||
|
|
||||||
|
difflib gives the matching blocks between the two word sequences; a correct
|
||||||
|
line takes the time of the earliest transcript word inside it. Lines whose
|
||||||
|
words were misheard badly enough to match nothing are interpolated between
|
||||||
|
their neighbours, so no line is left without a time.
|
||||||
|
"""
|
||||||
|
src = [w for w, _ in stream]
|
||||||
|
tgt, owner = [], [] # every correct word + which line it belongs to
|
||||||
|
for i, line in enumerate(correct_lines):
|
||||||
|
for w in words_of(line['text']):
|
||||||
|
tgt.append(w)
|
||||||
|
owner.append(i)
|
||||||
|
|
||||||
|
times = [None] * len(correct_lines)
|
||||||
|
sm = difflib.SequenceMatcher(a=src, b=tgt, autojunk=False)
|
||||||
|
for a0, b0, size in sm.get_matching_blocks():
|
||||||
|
for k in range(size):
|
||||||
|
line_i = owner[b0 + k]
|
||||||
|
t = stream[a0 + k][1]
|
||||||
|
if times[line_i] is None or t < times[line_i]:
|
||||||
|
times[line_i] = t
|
||||||
|
|
||||||
|
# Monotonic: a later line can never start before an earlier one.
|
||||||
|
best = 0.0
|
||||||
|
for i, t in enumerate(times):
|
||||||
|
if t is None:
|
||||||
|
continue
|
||||||
|
times[i] = max(t, best)
|
||||||
|
best = times[i]
|
||||||
|
|
||||||
|
# Fill gaps by spreading evenly between the timed neighbours.
|
||||||
|
known = [i for i, t in enumerate(times) if t is not None]
|
||||||
|
if not known:
|
||||||
|
return times, 0
|
||||||
|
for i in range(len(times)):
|
||||||
|
if times[i] is not None:
|
||||||
|
continue
|
||||||
|
prev = max((k for k in known if k < i), default=None)
|
||||||
|
nxt = min((k for k in known if k > i), default=None)
|
||||||
|
if prev is None:
|
||||||
|
times[i] = max(0.0, times[nxt] - 2.0)
|
||||||
|
elif nxt is None:
|
||||||
|
times[i] = times[prev] + 2.5
|
||||||
|
else:
|
||||||
|
step = (times[nxt] - times[prev]) / (nxt - prev)
|
||||||
|
times[i] = round(times[prev] + step * (i - prev), 2)
|
||||||
|
return times, len(known)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||||
|
ap.add_argument('--base', default=os.environ.get('YTP_BASE', 'https://worship.hesed.sbs'))
|
||||||
|
ap.add_argument('--id', required=True, help='video id')
|
||||||
|
ap.add_argument('--words', required=True, help='file with the correct lyrics (plain or LRC)')
|
||||||
|
ap.add_argument('--times-from', help='backup JSON to take the timings from (default: what is live)')
|
||||||
|
ap.add_argument('--apply', action='store_true')
|
||||||
|
ap.add_argument('--backup-dir', default=os.environ.get('YTP_BACKUP_DIR', ''))
|
||||||
|
args = ap.parse_args()
|
||||||
|
|
||||||
|
api = Api(args.base, os.environ.get('YTP_TOKEN'), os.environ.get('YTP_ADMIN_PASSWORD'))
|
||||||
|
st, n = api.call('GET', f'/api/notes/{args.id}')
|
||||||
|
live = (n or {}).get('lyrics') if st == 200 else None
|
||||||
|
|
||||||
|
if args.times_from:
|
||||||
|
with open(args.times_from, encoding='utf-8') as f:
|
||||||
|
song = json.load(f)['songs'][args.id]
|
||||||
|
timed_lines = song['data']['lines']
|
||||||
|
else:
|
||||||
|
if not live:
|
||||||
|
sys.exit('this song has no lyrics to take timings from — pass --times-from')
|
||||||
|
timed_lines = live['data']['lines']
|
||||||
|
if not any(l.get('t') is not None for l in timed_lines):
|
||||||
|
sys.exit('the timing source has no timed lines')
|
||||||
|
|
||||||
|
correct = parse_lrc(open(args.words, encoding='utf-8').read())
|
||||||
|
correct = [l for l in correct if l['text'].strip()]
|
||||||
|
if not correct:
|
||||||
|
sys.exit('no lyrics found in --words')
|
||||||
|
|
||||||
|
st, sres = api.call('GET', f'/api/streams?v={args.id}')
|
||||||
|
total = float((((sres or {}).get('data') or {}).get('meta') or {}).get('duration') or 0)
|
||||||
|
|
||||||
|
stream = timed_word_stream(timed_lines, total)
|
||||||
|
times, matched = align(correct, stream)
|
||||||
|
for line, t in zip(correct, times):
|
||||||
|
line['t'] = round(float(t), 2)
|
||||||
|
|
||||||
|
print(f'{args.id}: {len(correct)} correct lines timed from {len(timed_lines)} transcript lines '
|
||||||
|
f'({matched} matched directly, {len(correct) - matched} interpolated)\n')
|
||||||
|
for line in correct[:10]:
|
||||||
|
print(f' [{line["t"]:7.2f}] {line["text"][:62]}')
|
||||||
|
if len(correct) > 12:
|
||||||
|
print(' …')
|
||||||
|
for line in correct[-2:]:
|
||||||
|
print(f' [{line["t"]:7.2f}] {line["text"][:62]}')
|
||||||
|
if total and correct[-1]['t'] > total + 5:
|
||||||
|
print(f'\n!! last line lands at {correct[-1]["t"]:.0f}s but the song is {total:.0f}s — check before applying')
|
||||||
|
|
||||||
|
if not args.apply:
|
||||||
|
print('\n(dry run — add --apply)')
|
||||||
|
return
|
||||||
|
|
||||||
|
if live:
|
||||||
|
out_dir = args.backup_dir or ('/mnt/c/Users/josh/Documents' if os.path.isdir('/mnt/c/Users/josh/Documents') else os.path.expanduser('~'))
|
||||||
|
path = os.path.join(out_dir, f'ytplayer-lyrics-backup-{dt.datetime.now():%Y%m%d-%H%M%S}.json')
|
||||||
|
with open(path, 'w', encoding='utf-8') as f:
|
||||||
|
json.dump({'exportedAt': dt.datetime.now().isoformat(), 'base': args.base,
|
||||||
|
'songs': {args.id: {'savedAt': live['updatedAt'], 'rev': live['rev'], 'data': live['data']}}}, f, ensure_ascii=False, indent=1)
|
||||||
|
print(f'backup → {path}')
|
||||||
|
|
||||||
|
doc = {'lines': correct, 'tags': ['correct words, timed from the transcript'], 'offset': 0}
|
||||||
|
st, rr = api.call('PUT', f'/api/notes/{args.id}/lyrics', {'data': doc, 'baseRev': live['rev'] if live else 0})
|
||||||
|
print(f'saved as rev {rr.get("rev")}' if st == 200 else f'write failed {st}: {rr.get("error")}')
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user