Collect discovered video metadata and recommend videos from listening history
This commit is contained in:
191
server/recommendations.test.js
Normal file
191
server/recommendations.test.js
Normal file
@@ -0,0 +1,191 @@
|
||||
import { test, expect, beforeAll, afterAll } from 'bun:test';
|
||||
import { mkdtempSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { Hono } from 'hono';
|
||||
|
||||
const root = mkdtempSync(join(tmpdir(), 'ytp-recommendations-'));
|
||||
process.env.DB_PATH = join(root, 'catalog.db');
|
||||
const { initDb, db } = await import('./db.js');
|
||||
const catalog = await import('./video-catalog.js');
|
||||
const { recommend, rankRecommendations, registerCatalogRoutes } = await import('./recommendations.js');
|
||||
const cache = await import('./search-cache.js');
|
||||
const a = { id: 'aaaaaaaaaaa', title: 'Quiet piano morning', channel: 'Piano studio', duration: 240, tags: ['piano', 'acoustic'] };
|
||||
const b = { id: 'bbbbbbbbbbb', title: 'Evening at the piano', channel: 'Piano studio', duration: 320 };
|
||||
const c = { id: 'ccccccccccc', title: 'Ocean documentary', channel: 'Nature films', duration: 900 };
|
||||
const day = new Date().toISOString().slice(0, 10);
|
||||
beforeAll(initDb);
|
||||
afterAll(() => { db.close(); rmSync(root, { recursive: true, force: true }); });
|
||||
const json = body => ({ method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify(body) });
|
||||
|
||||
test('normalizes metadata, excludes custom files, and refuses arbitrary thumbnail hosts', () => {
|
||||
expect(catalog.normalizeCard({ ...a, thumbnail: 'http://127.0.0.1/secrets' }).thumbnail).toBe('https://i.ytimg.com/vi/aaaaaaaaaaa/hqdefault.jpg');
|
||||
expect(catalog.normalizeCard({ ...a, thumbnail: 'https://i.ytimg.com:1234/private' }).thumbnail).toContain('/vi/aaaaaaaaaaa/');
|
||||
expect(catalog.normalizeCard({ ...a, id: '../anything' })).toBeNull();
|
||||
expect(catalog.normalizeCard({ ...a, custom: true })).toBeNull();
|
||||
expect(catalog.extractCards({ id: 'playlist1', title: 'My playlist', videos: [a, a, b] })).toHaveLength(2);
|
||||
});
|
||||
|
||||
test('all discovery sources preserve richer metadata on sparse updates', async () => {
|
||||
await catalog.ingest([a, b, c], 'search');
|
||||
await catalog.ingest([{ id: a.id, title: a.title }], 'playlist');
|
||||
const row = (await db.execute({ sql: 'SELECT card FROM video_meta WHERE id=?', args: [a.id] })).rows[0];
|
||||
expect(JSON.parse(row.card).tags).toEqual(a.tags);
|
||||
expect(JSON.parse(row.card).channel).toBe(a.channel);
|
||||
expect((await db.execute({ sql: 'SELECT source FROM video_meta_sources WHERE video_id=?', args: [a.id] })).rows.map(r => r.source).sort()).toEqual(['playlist', 'search']);
|
||||
});
|
||||
|
||||
test('daily listening snapshots are idempotent and monotonic, with malformed dates/counts ignored', async () => {
|
||||
const stats = { days: { [day]: { songs: { [a.id]: 8, [c.id]: 2, invalid: 999 } }, tomorrow: { songs: { [a.id]: 999 } } }, meta: {} };
|
||||
await catalog.syncListening('listener-a', stats);
|
||||
await catalog.syncListening('listener-a', stats);
|
||||
await catalog.syncListening('listener-a', { days: { [day]: { songs: { [a.id]: 3 } } } });
|
||||
const rows = (await db.execute('SELECT * FROM listening_daily')).rows;
|
||||
expect(rows).toHaveLength(2);
|
||||
expect(Number(rows.find(r => r.video_id === a.id).plays)).toBe(8);
|
||||
});
|
||||
|
||||
test('server ranks personal favorites and unplayed channel matches without a media cache', async () => {
|
||||
const cards = await recommend('listener-a');
|
||||
expect(cards[0].id).toBe(a.id);
|
||||
expect(cards.find(v => v.id === b.id).reason).toBe('More from Piano studio');
|
||||
expect((await db.execute('SELECT COUNT(*) AS n FROM media_cache')).rows[0].n).toBe(0);
|
||||
expect((await recommend())[0].id).toBe(a.id);
|
||||
});
|
||||
|
||||
test('ranking diversifies channels and mixes discoveries with most played', () => {
|
||||
const candidates = Array.from({ length: 20 }, (_, i) => ({ card: { ...a, id: String(i).padStart(11, '0'), channel: 'Channel ' + Math.floor(i / 5) }, personal: i < 10 ? 50 : 0, plays: 50 - i, recent: 5 }));
|
||||
const cards = rankRecommendations(candidates, { limit: 12 });
|
||||
expect(cards.filter(v => v.reason === 'One of your most played').length).toBeLessThanOrEqual(6);
|
||||
for (const channel of new Set(cards.map(v => v.channel))) expect(cards.filter(v => v.channel === channel).length).toBeLessThanOrEqual(3);
|
||||
expect(cards.some(v => v.reason !== 'One of your most played')).toBe(true);
|
||||
});
|
||||
|
||||
test('metadata and recommendations survive search-result cache eviction', async () => {
|
||||
await cache.put('piano', [a, b]);
|
||||
await cache.trim({ maxRows: 0 });
|
||||
expect(await cache.get('piano')).toBeNull();
|
||||
expect((await cache.searchVideos('piano')).length).toBe(2);
|
||||
expect((await recommend('listener-a')).some(v => v.id === b.id)).toBe(true);
|
||||
});
|
||||
|
||||
test('thumbnail bytes are persisted and served locally without an upstream redirect', async () => {
|
||||
const calls = [];
|
||||
await catalog.drainThumbnails({ fetchImage: async (url, options) => { calls.push({ url, options }); return new Response(new Uint8Array([255, 216, 255, 217]), { headers: { 'Content-Type': 'image/jpeg' } }); } });
|
||||
expect(calls).toHaveLength(3);
|
||||
expect(calls.every(c => c.options.redirect === 'error')).toBe(true);
|
||||
const app = new Hono(); registerCatalogRoutes(app);
|
||||
const res = await app.request('/api/catalog/' + a.id + '/thumbnail');
|
||||
expect(res.status).toBe(200);
|
||||
expect(res.headers.get('content-type')).toBe('image/jpeg');
|
||||
expect(new Uint8Array(await res.arrayBuffer())).toEqual(new Uint8Array([255, 216, 255, 217]));
|
||||
expect((await recommend('listener-a'))[0].thumbnail).toBe('/api/catalog/' + a.id + '/thumbnail');
|
||||
});
|
||||
|
||||
test('failed image jobs remain durable, retry later, and respect the thumbnail byte budget', async () => {
|
||||
const v = { ...a, id: 'ddddddddddd' }; await catalog.ingest([v], 'related');
|
||||
await catalog.drainThumbnails({ fetchImage: async () => { throw new Error('offline'); } });
|
||||
const row = (await db.execute({ sql: 'SELECT * FROM video_thumbnails WHERE video_id=?', args: [v.id] })).rows[0];
|
||||
expect(row.data).toBeNull(); expect(Number(row.attempts)).toBe(1); expect(Number(row.retry_at)).toBeGreaterThan(Date.now());
|
||||
let called = false;
|
||||
await catalog.drainThumbnails({ fetchImage: async () => { called = true; throw new Error('unexpected'); }, budget: 4 });
|
||||
expect(called).toBe(false);
|
||||
expect(Number((await db.execute('SELECT SUM(size) AS n FROM video_thumbnails')).rows[0].n)).toBeLessThanOrEqual(4);
|
||||
expect((await db.execute('SELECT COUNT(*) AS n FROM video_meta')).rows[0].n).toBe(4);
|
||||
});
|
||||
|
||||
test('successful cached searches, channels, streams, and playlists all feed the catalog', async () => {
|
||||
const app = new Hono(); registerCatalogRoutes(app);
|
||||
const paths = ['/api/search', '/api/channel', '/api/streams', '/api/playlist/shared', '/api/playlist/expand', '/api/profile/load', '/api/p2p/available'];
|
||||
for (let i = 0; i < paths.length; i++) {
|
||||
const card = { ...b, id: 'surface' + String(i).padStart(4, '0') };
|
||||
app.get(paths[i], c => c.json({ ok: true, data: { playlists: [{ videos: [card] }] } }));
|
||||
}
|
||||
for (let i = 0; i < paths.length; i++) {
|
||||
const card = { ...b, id: 'surface' + String(i).padStart(4, '0') };
|
||||
expect((await app.request(paths[i])).status).toBe(200);
|
||||
expect((await db.execute({ sql: 'SELECT id FROM video_meta WHERE id=?', args: [card.id] })).rows).toHaveLength(1);
|
||||
}
|
||||
});
|
||||
|
||||
test('profile and playlist POST bodies are collected only after successful writes', async () => {
|
||||
const app = new Hono(); registerCatalogRoutes(app);
|
||||
app.post('/api/profile/save', c => c.json({ ok: true }));
|
||||
app.post('/api/playlist/share', c => c.json({ ok: false }, 403));
|
||||
const card = { ...b, id: 'postprofile' };
|
||||
await app.request('/api/profile/save', json({ data: { playlists: [{ videos: [card] }] } }));
|
||||
expect((await db.execute({ sql: 'SELECT id FROM video_meta WHERE id=?', args: [card.id] })).rows).toHaveLength(1);
|
||||
const rejected = { ...b, id: 'deniedvideo' };
|
||||
await app.request('/api/playlist/share', json({ playlist: { videos: [rejected] } }));
|
||||
expect((await db.execute({ sql: 'SELECT id FROM video_meta WHERE id=?', args: [rejected.id] })).rows).toHaveLength(0);
|
||||
});
|
||||
|
||||
test('client cached search collection validates input and stores new metadata', async () => {
|
||||
const app = new Hono(); registerCatalogRoutes(app);
|
||||
const card = { ...b, id: 'cachedvideo' };
|
||||
const res = await app.request('/api/catalog/collect', json({ cards: [card] }));
|
||||
expect(await res.json()).toEqual({ ok: true, collected: 1 });
|
||||
expect((await app.request('/api/catalog/collect', json({ cards: Array(501).fill(card) }))).status).toBe(400);
|
||||
const picks = await (await app.request('/api/recommendations?fp=listener-a&limit=4')).json();
|
||||
expect(picks.results).toHaveLength(4);
|
||||
});
|
||||
|
||||
test('legacy backfill progresses with a durable cursor and finds unplayed search metadata', async () => {
|
||||
const card = { ...b, id: 'legacyvideo' };
|
||||
await cache.put('legacy', [card]);
|
||||
await db.execute({ sql: 'INSERT INTO media_cache (video_id,meta) VALUES (?,?)', args: ['legacymedia', JSON.stringify({ title: 'Older cached piano', channel: 'Piano studio', duration: 60 })] });
|
||||
for (let i = 0; i < 25; i++) await catalog.backfillStep(2);
|
||||
expect((await db.execute({ sql: 'SELECT id FROM video_meta WHERE id=?', args: [card.id] })).rows).toHaveLength(1);
|
||||
expect((await db.execute({ sql: 'SELECT id FROM video_meta WHERE id=?', args: ['legacymedia'] })).rows).toHaveLength(1);
|
||||
const state = (await db.execute("SELECT * FROM catalog_backfill WHERE source='search_cache'")).rows[0];
|
||||
expect(Number(state.cursor)).toBe(Number(state.ceiling));
|
||||
const before = (await db.execute('SELECT COUNT(*) AS n FROM video_meta_sources')).rows[0].n;
|
||||
await catalog.backfillStep(2);
|
||||
expect((await db.execute('SELECT COUNT(*) AS n FROM video_meta_sources')).rows[0].n).toBe(before);
|
||||
});
|
||||
|
||||
test('linking devices to one profile merges daily counts without multiplying synced history', async () => {
|
||||
const stats = { days: { [day]: { songs: { [a.id]: 20 } } } };
|
||||
await catalog.syncListening('device:one', stats); await catalog.syncListening('device:two', stats);
|
||||
await catalog.linkListening('device:one', 'profile:linked'); await catalog.linkListening('device:two', 'profile:linked');
|
||||
await catalog.syncListening('profile:linked', stats);
|
||||
const rows = (await db.execute("SELECT * FROM listening_daily WHERE fingerprint IN ('device:one','device:two','profile:linked')")).rows;
|
||||
expect(rows).toHaveLength(1); expect(Number(rows[0].plays)).toBe(20);
|
||||
});
|
||||
|
||||
test('profile recommendations use the profile gate and separate profile identities', async () => {
|
||||
const app = new Hono();
|
||||
registerCatalogRoutes(app, { resolveListener: async (c, name, fp) => {
|
||||
if (!name) return fp;
|
||||
if (c.req.header('x-profile-secret') !== 'test-secret') return c.json({ ok: false }, 401);
|
||||
return 'profile:' + name;
|
||||
} });
|
||||
expect((await app.request('/api/recommendations?profile=linked')).status).toBe(401);
|
||||
const linked = await (await app.request('/api/recommendations?profile=linked', { headers: { 'X-Profile-Secret': 'test-secret' } })).json();
|
||||
expect(linked.results.find(c => c.id === a.id).reason).toBe('One of your most played');
|
||||
const other = await (await app.request('/api/recommendations?profile=other', { headers: { 'X-Profile-Secret': 'test-secret' } })).json();
|
||||
expect(other.results.find(c => c.id === a.id).reason).not.toBe('One of your most played');
|
||||
});
|
||||
|
||||
test('concurrent sparse discovery does not erase metadata; corrupt thumbnail jobs cannot fetch private hosts', async () => {
|
||||
const card = { ...a, id: 'concurrent1' };
|
||||
await Promise.all([catalog.ingest([card]), catalog.ingest([{ id: card.id, title: card.title }], 'playlist')]);
|
||||
const row = (await db.execute({ sql: 'SELECT card FROM video_meta WHERE id=?', args: [card.id] })).rows[0];
|
||||
expect(JSON.parse(row.card).tags).toEqual(a.tags);
|
||||
await db.execute({ sql: 'UPDATE video_thumbnails SET url=?,retry_at=0,data=NULL WHERE video_id=?', args: ['http://127.0.0.1/private', card.id] });
|
||||
const urls = [];
|
||||
await catalog.drainThumbnails({ fetchImage: async url => { urls.push(url); return new Response(new Uint8Array([1]), { headers: { 'Content-Type': 'image/jpeg' } }); } });
|
||||
expect(urls.some(url => url.includes('127.0.0.1'))).toBe(false);
|
||||
});
|
||||
|
||||
test('indexed channel candidates include older unplayed matches beyond the recent discovery window', async () => {
|
||||
const seed = { id: 'oldchannel1', title: 'Favorite strings', channel: 'Rare ensemble' };
|
||||
const match = { id: 'oldchannel2', title: 'Archival strings', channel: 'Rare ensemble' };
|
||||
await catalog.ingest([seed, match]);
|
||||
await catalog.syncListening('rare-listener', { days: { [day]: { songs: { [seed.id]: 10 } } } });
|
||||
await catalog.ingest(Array.from({ length: 650 }, (_, i) => ({ id: 'archive' + String(i).padStart(4, '0'), title: 'Archive entry ' + i, channel: 'Other channel ' + i })), 'search');
|
||||
await db.execute({ sql: 'UPDATE video_meta SET updated_at=0 WHERE id=?', args: [match.id] });
|
||||
expect((await recommend('rare-listener')).some(c => c.id === match.id)).toBe(true);
|
||||
const plan = (await db.execute({ sql: 'EXPLAIN QUERY PLAN SELECT video_id FROM video_channels WHERE channel=? ORDER BY updated_at DESC LIMIT 30', args: ['piano studio'] })).rows;
|
||||
expect(plan.some(r => String(r.detail).includes('idx_video_channel'))).toBe(true);
|
||||
});
|
||||
Reference in New Issue
Block a user