Backfill device song metadata and queue guarded server caching

This commit is contained in:
Jonathan Sykes
2026-10-03 17:55:49 +08:00
parent 9e960cf70f
commit 856ddda6a6
8 changed files with 187 additions and 7 deletions

View File

@@ -0,0 +1,11 @@
# Backfill songs played from device storage
A web player's first actual playback of a YouTube video in a session reports its known title, artist/channel, duration and thumbnail to `POST /api/media/:id/meta`. Local OPFS playback therefore no longer depends on `/api/streams` being called. Offline plays are held in memory until the browser returns online. Uploads and edited copies are excluded. Playback never waits for the report.
The server accepts only 11-character YouTube IDs, bounded strings, numeric duration and approved HTTPS YouTube artwork (local catalog artwork maps to the canonical thumbnail). Bodies are capped at 8 KB. Reports are limited to 20 per IP per minute and 100 globally, with a maximum of 32 outstanding backfill fetches. A response contains `{ok, known, cache}`; `known` describes whether a media row existed before the report, and `cache` is ready, queued, downloading, validating, unavailable or busy. Metadata can be registered even when the cache volume is offline or the queue is busy.
Device hints fill missing media fields with an atomic SQLite merge; existing titles, artists, durations, art and richer metadata are retained. Catalog ingestion also fills gaps and queues the existing bounded thumbnail collector. Admin's song list includes metadata-only entries, so its lyric tools have a title and artist before downloading completes. Inclusion in that list does not imply cached media is ready; `cacheStatus` reports the actual state.
Missing copies use the normal low-priority automatic `ensureCached` job, including its deduplication, single fetch lane, yt-dlp path, validation, duration/backoff limits, LRU byte budget, disk guard and USB volume marker. No parallel downloader or cache directory is introduced. Existing copies and in-progress jobs are skipped. Download failure keeps song metadata; the usual cache retry policy applies on later sessions. Offline-volume and busy responses do not force a fetch or bypass safety limits.
Verification: server validation/merge/route tests, admin metadata-only listing test, frontend once-per-session/offline reporter tests, all frontend unit tests, app syntax and server build. On a real phone, play a previously unknown OPFS song, inspect `/api/admin/media`, then verify a normal media-cache job and lyrics lookup. Repeat the play to confirm no second report; also test with the server's cache volume unavailable.

31
server/media-meta-core.js Normal file
View File

@@ -0,0 +1,31 @@
// Device hints only fill gaps; extractor metadata stays authoritative.
export function mergeMissing(existing = {}, incoming = {}) {
const out = { ...existing };
for (const [key, value] of Object.entries(incoming)) {
const old = out[key];
if (old == null || old === '' || old === 0 || (key === 'title' && (old === '(untitled)' || old === out.id))) out[key] = value;
}
if (!existing.channel && (existing.uploader || existing.artist)) out.channel = existing.uploader || existing.artist;
return out;
}
export function validateMeta(id, body) {
if (!/^[A-Za-z0-9_-]{11}$/.test(id) || !body || typeof body !== 'object' || Array.isArray(body) || (body.id != null && body.id !== id) || body.custom || body.upload) throw new Error('Invalid YouTube video');
const clean = (value, required = false) => {
if (value == null && !required) return '';
if (typeof value !== 'string' || value.length > 300 || /[\x00-\x08\x0b\x0c\x0e-\x1f]/.test(value)) throw new Error('Invalid title or artist');
const text = value.trim(); if (required && (!text || text === '(untitled)' || text === id)) throw new Error('A video title is required'); return text;
};
const title = clean(body.title, true), channel = clean(body.channel || body.artist);
const duration = body.duration ?? 0;
if (typeof duration !== 'number' || !Number.isFinite(duration) || duration < 0 || duration > 86400) throw new Error('Invalid duration');
let thumbnail = `https://i.ytimg.com/vi/${id}/hqdefault.jpg`;
if (body.thumbnail) {
if (typeof body.thumbnail !== 'string' || body.thumbnail.length > 2048) throw new Error('Invalid thumbnail');
if (body.thumbnail !== `/api/catalog/${id}/thumbnail`) {
let url; try { url = new URL(body.thumbnail); } catch { throw new Error('Invalid thumbnail'); }
if (url.protocol !== 'https:' || !/^(?:i|i\d)\.ytimg\.com$/.test(url.hostname) || url.port || url.username || url.password) throw new Error('Invalid thumbnail');
thumbnail = url.href;
}
}
return { id, title, channel, duration, thumbnail };
}

66
server/media-meta.js Normal file
View File

@@ -0,0 +1,66 @@
import { validateMeta, mergeMissing } from './media-meta-core.js';
async function readBody(request) {
if (Number(request.headers.get('content-length')) > 8192) throw new Error('Metadata is too large');
const reader = request.body?.getReader(); if (!reader) throw new Error('Metadata is required');
const chunks = []; let size = 0;
try {
for (;;) {
const { done, value } = await reader.read(); if (done) break;
size += value.byteLength; if (size > 8192) throw new Error('Metadata is too large'); chunks.push(value);
}
} finally { await reader.cancel().catch(() => {}); }
const bytes = new Uint8Array(size); let offset = 0;
for (const chunk of chunks) { bytes.set(chunk, offset); offset += chunk.byteLength; }
return JSON.parse(new TextDecoder().decode(bytes));
}
export function createLimiter({ now = Date.now, perClient = 20, global = 100 } = {}) {
const clients = new Map(); let windowStart = now(), count = 0;
return key => {
const time = now();
if (time - windowStart >= 60000) { clients.clear(); count = 0; windowStart = time; }
if (count >= global || (clients.get(key) || 0) >= perClient || (!clients.has(key) && clients.size >= 2000)) return false;
clients.set(key, (clients.get(key) || 0) + 1); count++; return true;
};
}
export function registerMediaMetaRoutes(app, { db, catalog, media, now = Date.now, limiter = createLimiter({ now }), maxPending = 32, log = console }) {
const pending = new Map();
app.post('/api/media/:id/meta', async c => {
const who = (c.req.header('x-forwarded-for') || c.req.header('x-real-ip') || 'local').split(',')[0].trim();
if (!limiter(who)) return c.json({ ok: false, error: 'Too many metadata reports' }, 429, { 'Retry-After': '60' });
let card;
try { card = validateMeta(c.req.param('id'), await readBody(c.req.raw)); }
catch (error) { return c.json({ ok: false, error: error.message }, 400); }
try {
const old = await db.getMedia(card.id);
const catalogRow = (await db.db.execute({ sql: 'SELECT card FROM video_meta WHERE id=?', args: [card.id] })).rows[0];
let richer = {}; try { richer = JSON.parse(catalogRow?.card || '{}'); } catch { /* absent */ }
const hint = mergeMissing(richer, card);
// SQL merges against the row at write time, so a concurrent extractor
// update cannot be overwritten by a stale device read.
await db.db.execute({ sql: `INSERT INTO media_cache (video_id,meta,duration,status,auto,created_at,updated_at,last_access)
VALUES (?,?,?,'failed',1,?,?,?) ON CONFLICT(video_id) DO UPDATE SET
meta=json_patch(media_cache.meta, (SELECT COALESCE(json_group_object(key,value),'{}') FROM json_each(excluded.meta)
WHERE (json_extract(media_cache.meta,'$.'||key) IS NULL OR json_extract(media_cache.meta,'$.'||key) IN ('',0)
OR (key='title' AND json_extract(media_cache.meta,'$.title') IN ('(untitled)',media_cache.video_id)))
AND NOT (key='channel' AND COALESCE(NULLIF(json_extract(media_cache.meta,'$.uploader'),''),NULLIF(json_extract(media_cache.meta,'$.artist'),''),'')!=''))),
duration=CASE WHEN media_cache.duration<=0 THEN excluded.duration ELSE media_cache.duration END`,
args: [card.id, JSON.stringify(Object.fromEntries(Object.keys(card).map(key => [key, hint[key]]))), hint.duration || 0, now(), now(), now()] });
// Keep the recommendation/thumbnail catalog in step with the media row.
const row = await db.getMedia(card.id), saved = JSON.parse(row.meta);
await catalog.ingest([{ ...saved, channel: saved.channel || saved.uploader || saved.artist || '', id: card.id }], 'device-play', { fillMissing: true });
let cache = 'queued';
if (await media.getReady(card.id)) cache = 'ready';
else if (!media.available()) cache = 'unavailable';
else if (!pending.has(card.id)) {
if (pending.size >= maxPending) cache = 'busy';
else {
const work = media.ensureCached(card.id, { priority: 1, auto: true });
pending.set(card.id, work);
work.catch(error => log.warn?.(`[media-meta] ${card.id}: ${error.message}`)).finally(() => pending.delete(card.id));
}
}
return c.json({ ok: true, known: !!old, cache }, cache === 'ready' ? 200 : 202, { 'Cache-Control': 'no-store' });
} catch (error) { log.warn?.('[media-meta]', error.message); return c.json({ ok: false, error: 'Could not register video metadata' }, 500); }
});
}

58
server/media-meta.test.js Normal file
View File

@@ -0,0 +1,58 @@
import { test, expect, beforeAll, afterAll } from 'bun:test';
import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { Hono } from 'hono';
import { validateMeta, mergeMissing } from './media-meta-core.js';
import { registerMediaMetaRoutes, createLimiter } from './media-meta.js';
const root = mkdtempSync(join(tmpdir(),'ytp-device-meta-'));
process.env.DB_PATH = join(root,'test.db');
const db = await import('./db.js'), catalog = await import('./video-catalog.js');
const id='0gfX0dFLaBc', card={id,title:'Because You are God',artist:'Cathedral of Praise Worship',duration:240};
let calls=0, online=true, ready=false;
const app=new Hono();
registerMediaMetaRoutes(app,{db,catalog,media:{getReady:async()=>ready,available:()=>online,status:async()=>({status:'failed'}),ensureCached:()=>{calls++;return new Promise(()=>{});}},log:{warn(){}}});
const post=(video=id,body=card)=>app.request(`/api/media/${video}/meta`,{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify(body)});
beforeAll(db.initDb);afterAll(()=>{db.db.close();rmSync(root,{recursive:true,force:true});});
test('validation rejects non-YouTube ids, mismatches, malformed fields and unsafe thumbnail URLs',()=>{
for(const body of [{...card,id:'bbbbbbbbbbb'},{...card,title:{}},{...card,duration:Infinity},{...card,duration:-1},{...card,thumbnail:'http://127.0.0.1/art'},{...card,thumbnail:'https://i.ytimg.com:1234/art'}])expect(()=>validateMeta(id,body)).toThrow();
expect(()=>validateMeta('upl_123',card)).toThrow();
expect(validateMeta(id,card).channel).toBe(card.artist);
expect(validateMeta(id,{...card,thumbnail:`/api/catalog/${id}/thumbnail`}).thumbnail).toContain('https://i.ytimg.com/');
expect(()=>validateMeta(id,{...card,title:'a'.repeat(301)})).toThrow();
});
test('merge only fills gaps and retains richer titles, durations, art and artist aliases',()=>{
const rich={title:'Full server title',duration:250,uploader:'Server artist',description:'Rich description',thumbnail:'rich-art'};
expect(mergeMissing(rich,validateMeta(id,card))).toMatchObject({...rich,channel:'Server artist'});
expect(mergeMissing({id,title:id,duration:0},validateMeta(id,card))).toMatchObject({title:card.title,duration:240});
});
test('unknown phone song gets a media row, catalog entry and one background cache job',async()=>{
const response=await post();expect(response.status).toBe(202);expect((await response.json()).known).toBe(false);
const row=await db.getMedia(id);expect(JSON.parse(row.meta).title).toBe(card.title);expect(JSON.parse(row.meta).channel).toBe(card.artist);expect(row.duration).toBe(240);
expect((await db.db.execute({sql:'SELECT card FROM video_meta WHERE id=?',args:[id]})).rows).toHaveLength(1);
await post();expect(calls).toBe(1);
});
test('atomic missing-field SQL and catalog merge retain existing server metadata',async()=>{
const video='bbbbbbbbbbb';
await db.upsertMedia(video,{meta:JSON.stringify({title:'Server title',uploader:'Server artist',duration:300,thumbnail:'https://i.ytimg.com/vi/bbbbbbbbbbb/maxresdefault.jpg',tags:['rich']}),duration:300,status:'ready'});
await catalog.ingest([{id:video,title:'Catalog title',channel:'Catalog artist',duration:310,tags:['catalog']}]);
ready=true;const before=calls;const response=await post(video,{...card,id:video});expect(response.status).toBe(200);expect(calls).toBe(before);ready=false;
const meta=JSON.parse((await db.getMedia(video)).meta);expect(meta.title).toBe('Server title');expect(meta.uploader).toBe('Server artist');expect(meta.channel).toBeUndefined();expect(meta.tags).toEqual(['rich']);expect((await db.getMedia(video)).duration).toBe(300);
const c=JSON.parse((await db.db.execute({sql:'SELECT card FROM video_meta WHERE id=?',args:[video]})).rows[0].card);expect(c.title).toBe('Catalog title');expect(c.channel).toBe('Catalog artist');
});
test('offline cache volume retains labels without queuing a fetch',async()=>{
online=false;const before=calls;const response=await post('ccccccccccc',{...card,id:'ccccccccccc'});expect((await response.json()).cache).toBe('unavailable');expect(calls).toBe(before);expect(JSON.parse((await db.getMedia('ccccccccccc')).meta).title).toBe(card.title);online=true;
});
test('bounded body and rate limits reject expensive or repeated reports',async()=>{
expect((await post(id,{...card,title:'x'.repeat(10000)})).status).toBe(400);
let time=0;const limited=createLimiter({now:()=>time,perClient:2,global:3});expect(limited('a')).toBe(true);expect(limited('a')).toBe(true);expect(limited('a')).toBe(false);expect(limited('b')).toBe(true);expect(limited('c')).toBe(false);time=60000;expect(limited('a')).toBe(true);
const throttled=new Hono();registerMediaMetaRoutes(throttled,{db,catalog,media:{},limiter:()=>false});expect((await throttled.request(`/api/media/${id}/meta`,{method:'POST',body:'{}'})).status).toBe(429);
});
test('a full backfill queue keeps song metadata without starting another download',async()=>{
let fetched=0;const bounded=new Hono();
registerMediaMetaRoutes(bounded,{db,catalog,maxPending:1,media:{getReady:async()=>false,available:()=>true,ensureCached:()=>{fetched++;return new Promise(()=>{});}},log:{warn(){}}});
const report=video=>bounded.request(`/api/media/${video}/meta`,{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({...card,id:video})});
expect((await (await report('ddddddddddd')).json()).cache).toBe('queued');
expect((await (await report('eeeeeeeeeee')).json()).cache).toBe('busy');expect(fetched).toBe(1);
expect(JSON.parse((await db.getMedia('eeeeeeeeeee')).meta).title).toBe(card.title);
});

View File

@@ -687,14 +687,14 @@ export function registerNoteRoutes(app, deps) {
// Saved (server-cached) videos with whether each already has lyrics — the // Saved (server-cached) videos with whether each already has lyrics — the
// work list for batch lyric injection. // work list for batch lyric injection.
app.get('/api/admin/media', requireAdminOrToken, async (c) => { app.get('/api/admin/media', requireAdminOrToken, async (c) => {
const rows = (await db.listMedia()).filter((r) => r.status === 'ready'); const rows = await db.listMedia();
const out = []; const out = [];
for (const r of rows) { for (const r of rows) {
let meta = {}; let meta = {};
try { meta = JSON.parse(r.meta || '{}'); } catch { /* corrupt meta */ } try { meta = JSON.parse(r.meta || '{}'); } catch { /* corrupt meta */ }
const n = await db.getNotes(r.video_id); const n = await db.getNotes(r.video_id);
out.push({ out.push({
id: r.video_id, title: meta.title || '', channel: meta.channel || meta.uploader || '', id: r.video_id, cacheStatus: r.status, title: meta.title || '', channel: meta.channel || meta.uploader || '',
duration: Number(r.duration) || 0, lastAccess: Number(r.last_access) || 0, duration: Number(r.duration) || 0, lastAccess: Number(r.last_access) || 0,
lyricsRev: n.lyrics ? n.lyrics.rev : 0, lyricsLines: n.lyrics ? n.lyrics.data.lines.length : 0, lyricsRev: n.lyrics ? n.lyrics.rev : 0, lyricsLines: n.lyrics ? n.lyrics.data.lines.length : 0,
}); });

View File

@@ -271,3 +271,12 @@ test('secondary and phonetic lyric fields survive validation without changing ol
const result = N.sanitizeLyrics({ lines:[{t:1,text:'主',kind:'line',secondary:' Your grace\n is enough ',phonetic:' zhǔ\u0007 '},{t:2,text:'Legacy',kind:'line'}] }); const result = N.sanitizeLyrics({ lines:[{t:1,text:'主',kind:'line',secondary:' Your grace\n is enough ',phonetic:' zhǔ\u0007 '},{t:2,text:'Legacy',kind:'line'}] });
expect(result.lines[0].secondary).toBe('Your grace\nis enough'); expect(result.lines[0].phonetic).toBe('zhǔ'); expect(result.lines[1]).toEqual({t:2,text:'Legacy',kind:'line'}); expect(result.lines[0].secondary).toBe('Your grace\nis enough'); expect(result.lines[0].phonetic).toBe('zhǔ'); expect(result.lines[1]).toEqual({t:2,text:'Legacy',kind:'line'});
}); });
test('admin song list includes metadata-only phone songs for lyrics lookup', async () => {
const id = 'phoneSong01';
await dbmod.upsertMedia(id, { status: 'failed', duration: 240, meta: JSON.stringify({ title: 'Phone song', channel: 'Phone artist' }) });
const cookie = await adminCookie();
const res = await app.request('/api/admin/media', { headers: { Cookie: cookie } });
const song = (await res.json()).media.find(x => x.id === id);
expect(song).toMatchObject({ title: 'Phone song', channel: 'Phone artist', duration: 240, cacheStatus: 'failed' });
});

View File

@@ -47,6 +47,8 @@ import * as notesDb from './db.js';
import { registerNoteRoutes, parseLrc, sanitizeLyrics } from './notes.js'; import { registerNoteRoutes, parseLrc, sanitizeLyrics } from './notes.js';
import { createRemoteHub } from './remote.js'; import { createRemoteHub } from './remote.js';
import { createPartyHub } from './party.js'; import { createPartyHub } from './party.js';
import { registerMediaMetaRoutes } from './media-meta.js';
import * as videoCatalog from './video-catalog.js';
import { registerPianoRoutes } from './piano.js'; import { registerPianoRoutes } from './piano.js';
import { registerUploadRoutes } from './uploads.js'; import { registerUploadRoutes } from './uploads.js';
import { admitFile } from './p2p-admit.js'; import { admitFile } from './p2p-admit.js';
@@ -1619,6 +1621,8 @@ app.get('/api/media/:id/status', async (c) => {
} }
}); });
registerMediaMetaRoutes(app, { db: notesDb, catalog: videoCatalog, media });
// POST /api/media/:id/redownload — the client's "Broken" button. Idempotent: // POST /api/media/:id/redownload — the client's "Broken" button. Idempotent:
// a job already queued/running for this id is just reported back. // a job already queued/running for this id is just reported back.
app.post('/api/media/:id/redownload', async (c) => { app.post('/api/media/:id/redownload', async (c) => {

View File

@@ -1,8 +1,9 @@
// One catalog for every discovery source. Media-file caching is independent. // One catalog for every discovery source. Media-file caching is independent.
import { db } from './db.js'; import { db } from './db.js';
import { mergeMissing } from './media-meta-core.js';
const VIDEO_ID = /^[\w-]{11}$/; const VIDEO_ID = /^[\w-]{11}$/;
const SOURCES = new Set(['search', 'search-cache', 'client-search', 'channel', 'streams', 'playlist', 'profile', 'sync', 'related', 'backfill', 'collector']); const SOURCES = new Set(['search', 'search-cache', 'client-search', 'channel', 'streams', 'playlist', 'profile', 'sync', 'related', 'backfill', 'collector', 'device-play']);
const text = (v, n = 300) => typeof v === 'string' ? v.trim().slice(0, n) : ''; const text = (v, n = 300) => typeof v === 'string' ? v.trim().slice(0, n) : '';
const positive = v => Number.isFinite(Number(v)) && Number(v) > 0 ? Number(v) : 0; const positive = v => Number.isFinite(Number(v)) && Number(v) > 0 ? Number(v) : 0;
const canonicalThumb = id => `https://i.ytimg.com/vi/${id}/hqdefault.jpg`; const canonicalThumb = id => `https://i.ytimg.com/vi/${id}/hqdefault.jpg`;
@@ -48,13 +49,13 @@ export function extractCards(value, limit = 5000) {
let ingestion = Promise.resolve(); let ingestion = Promise.resolve();
let writes = 0; let writes = 0;
export function ingest(cards, source = 'search') { export function ingest(cards, source = 'search', { fillMissing = false } = {}) {
const task = ingestion.then(() => ingestBatch(cards, source)); const task = ingestion.then(() => ingestBatch(cards, source, fillMissing));
ingestion = task.catch(() => {}); ingestion = task.catch(() => {});
return task; return task;
} }
async function ingestBatch(cards, source) { async function ingestBatch(cards, source, fillMissing) {
source = SOURCES.has(source) ? source : 'playlist'; source = SOURCES.has(source) ? source : 'playlist';
const unique = extractCards(cards); const unique = extractCards(cards);
const now = Date.now(); const now = Date.now();
@@ -65,7 +66,7 @@ async function ingestBatch(cards, source) {
const old = new Map(existing.rows.map(r => [r.id, JSON.parse(r.card)])); const old = new Map(existing.rows.map(r => [r.id, JSON.parse(r.card)]));
await db.batch(batch.flatMap(c => { await db.batch(batch.flatMap(c => {
// Sparse playlist cards must not erase richer extractor metadata. // Sparse playlist cards must not erase richer extractor metadata.
const merged = { ...(old.get(c.id) || {}), ...Object.fromEntries(Object.entries(c).filter(([, v]) => v !== '' && v !== 0 && (!Array.isArray(v) || v.length))) }; const merged = fillMissing ? mergeMissing(old.get(c.id) || {}, c) : { ...(old.get(c.id) || {}), ...Object.fromEntries(Object.entries(c).filter(([, v]) => v !== '' && v !== 0 && (!Array.isArray(v) || v.length))) };
return [{ sql: `INSERT INTO video_meta (id,card,hay,seen,updated_at) VALUES (?,?,?,1,?) return [{ sql: `INSERT INTO video_meta (id,card,hay,seen,updated_at) VALUES (?,?,?,1,?)
ON CONFLICT(id) DO UPDATE SET card=excluded.card,hay=excluded.hay,seen=seen+1,updated_at=excluded.updated_at`, ON CONFLICT(id) DO UPDATE SET card=excluded.card,hay=excluded.hay,seen=seen+1,updated_at=excluded.updated_at`,
args: [c.id, JSON.stringify(merged), `${merged.title} ${merged.channel || ''} ${(merged.tags || []).join(' ')}`.toLowerCase(), now] }, args: [c.id, JSON.stringify(merged), `${merged.title} ${merged.channel || ''} ${(merged.tags || []).join(' ')}`.toLowerCase(), now] },