From 08e95814e5ae55ec6d7c34c245b3c54e54d25313 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 18:48:37 +0000 Subject: [PATCH] Support hour-long uploads and videos: raise body limit, stream uploads to disk, cache up to 3 h, add deferred-ideas note --- CLAUDE.md | 12 ++- docs/deferred-ideas.md | 34 +++++++++ frontend/admin.html | 36 ++++++--- server/db.js | 4 + server/media-cache.js | 2 +- server/server.js | 6 +- server/uploads.js | 170 ++++++++++++++++++++++++++++------------- server/uploads.test.js | 34 +++++++++ 8 files changed, 232 insertions(+), 66 deletions(-) create mode 100644 docs/deferred-ideas.md diff --git a/CLAUDE.md b/CLAUDE.md index 93c50da..f2e6916 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -28,8 +28,18 @@ API endpoints (notes/admin routes are listed in the header of `server/notes.js`) JSON shapes mirror the Tauri Rust bridge exactly — don't change one side alone (`/api/streams`' `data.serverCached` is additive and web-only). +## Long videos (1 h+) +- Bun's default request-body cap is 128 MB — `Bun.serve` sets `maxRequestBodySize` to 5 GiB; + routes enforce their own limits (uploads 4 GB, intake `P2P_INTAKE_MAX_BYTES` 3 GiB). Without it + every large upload/intake died with 413. +- Admin uploads use `PUT /api/admin/uploads/stream?name=…` (raw body → disk in chunks, browser + sends the `File` itself); cover images go to `PUT /api/admin/uploads/:id/art`. The multipart + `POST /api/admin/uploads` parses the whole body in memory — scripts with small files only. +- Videos over `transcode.maxSeconds` are never re-encoded (lane would be tied up for hours). +- Deferred ideas (Cloudflare Worker fetch, Android device download): `docs/deferred-ideas.md`. + ## Server media cache (`server/media-cache.js`) -- Every played (`/api/streams`, LOW priority, ≤ `MEDIA_AUTO_MAX_SECONDS`) or saved +- Every played (`/api/streams`, LOW priority, ≤ `MEDIA_AUTO_MAX_SECONDS`, default 3 h) or saved (`/api/download`, HIGH, ≤ 3 h) video gets ONE copy: `$MEDIA_DIR/..mp4` (≤720p H.264 8-bit + AAC, faststart) + `..m4a` audio sidecar for audio-only mode. `MEDIA_DIR` defaults to `./data/media` (the `ytplayer-data` volume in prod). diff --git a/docs/deferred-ideas.md b/docs/deferred-ideas.md new file mode 100644 index 0000000..08cdd8c --- /dev/null +++ b/docs/deferred-ideas.md @@ -0,0 +1,34 @@ +# Deferred ideas + +Parked on purpose; nothing here is built. + +## Cloudflare Worker as a YouTube fetch fallback (deferred 2026-10-01) + +Idea: a free Cloudflare Worker (e.g. https://gist.github.com/hizkifw/ae229eb0c5ff809fc2a4a88735bfd604) +that fetches the watch page, decodes the signature cipher and streams a chosen +`itag`, so downloads come from somewhere other than the server's single IP. + +Why it was not adopted yet (read from the gist, never deployed or tested): +- The fetch still happens from Cloudflare's datacenter egress, which YouTube + often challenges or blocks; the IP is shared with every other Worker. +- It uses the old watch-page + cipher approach (ytdl-core). No PO tokens, no + throttle handling, adaptive formats only, so it breaks as YouTube changes. +- Free tier CPU (~10 ms/request, check current limits) may not cover decoding + the player script, and proxying large media may breach Cloudflare's terms. +- Stream URLs are bound to the fetching IP, so the Worker has to relay the + bytes itself. + +If revisited: deploy with `wrangler`, test ~10 real videos, and if it works add +it as the last tier behind an env var (`YT_WORKER_URL`): server cache → server +yt-dlp → Worker, with the result going through `validateMedia`. A newer Worker +could use the Android/iOS client endpoints (no cipher) but hits the same IP and +token blocks. + +## Device-side download (Android app) (deferred 2026-10-01) + +A web page cannot fetch googlevideo.com (CORS, PO tokens), and a yt-dlp WASM +build does not change that. The workable route is an Android app that runs +yt-dlp (or an equivalent extractor) on the phone's own IP, checks the server +cache first, then uploads the finished file in the background through the +device intake (`POST /api/p2p/intake`, `server/p2p-intake.js`). The server's +yt-dlp stays as the fallback when the device fails. diff --git a/frontend/admin.html b/frontend/admin.html index 29775eb..b4fde2b 100644 --- a/frontend/admin.html +++ b/frontend/admin.html @@ -636,25 +636,37 @@ $('upForm').addEventListener('submit', (e) => { e.preventDefault(); const f = $('upFile').files[0]; if (!f) return; - const fd = new FormData(); - fd.append('file', f); - if ($('upArt').files[0]) fd.append('art', $('upArt').files[0]); - if ($('upTitle').value.trim()) fd.append('title', $('upTitle').value.trim()); - if ($('upArtist').value.trim()) fd.append('artist', $('upArtist').value.trim()); + const qs = new URLSearchParams({ name: f.name }); + if ($('upTitle').value.trim()) qs.set('title', $('upTitle').value.trim()); + if ($('upArtist').value.trim()) qs.set('artist', $('upArtist').value.trim()); + const cover = $('upArt').files[0]; const msg = $('upProgress'); msg.className = 'msg'; msg.textContent = `Uploading ${f.name}…`; + // The raw file is the request body (streamed from disk by the browser, to + // disk by the server) so hour-long, multi-GB videos need no memory. const xhr = new XMLHttpRequest(); - xhr.open('POST', '/api/admin/uploads'); xhr.withCredentials = true; - xhr.upload.onprogress = (ev) => { if (ev.lengthComputable) msg.textContent = `Uploading ${f.name}… ${Math.round((ev.loaded / ev.total) * 100)}%`; }; - xhr.onload = () => { + xhr.open('PUT', '/api/admin/uploads/stream?' + qs); xhr.withCredentials = true; + xhr.setRequestHeader('Content-Type', 'application/octet-stream'); + const t0 = Date.now(); + xhr.upload.onprogress = (ev) => { + if (!ev.lengthComputable) return; + const pct = Math.round((ev.loaded / ev.total) * 100), mbps = ev.loaded / 1048576 / Math.max(1, (Date.now() - t0) / 1000); + msg.textContent = `Uploading ${f.name}… ${pct}% · ${(ev.loaded / 1048576).toFixed(0)} of ${(ev.total / 1048576).toFixed(0)} MB · ${mbps.toFixed(1)} MB/s${pct >= 100 ? ' — processing…' : ''}`; + }; + xhr.onload = async () => { let j = {}; try { j = JSON.parse(xhr.responseText); } catch { /* non-JSON */ } if (xhr.status === 200 && j.ok) { + let artNote = j.upload.art ? ', cover art' : ''; + if (cover) { + const ar = await fetch(`/api/admin/uploads/${j.upload.id}/art`, { method: 'PUT', credentials: 'same-origin', body: cover }).then((r) => r.json()).catch(() => ({})); + artNote = ar.ok ? ', cover art' : ', cover image unusable'; + } msg.className = 'msg ok'; - msg.textContent = `Added “${j.upload.title}” (${j.upload.kind}${j.upload.art ? ', cover art' : ''}${j.lyricLines ? `, ${j.lyricLines} lyric lines` : ''}).`; - $('upForm').reset(); loadUploads(); loadRecent(); - } else { msg.className = 'msg err'; msg.textContent = j.error || `Upload failed (HTTP ${xhr.status})`; } + msg.textContent = `Added “${j.upload.title}” (${j.upload.kind}${artNote}${j.lyricLines ? `, ${j.lyricLines} lyric lines` : ''}).`; + $('upForm').reset(); loadUploads(); loadRecent(); loadSongs(); + } else { msg.className = 'msg err'; msg.textContent = j.error || (xhr.status === 413 ? 'File is too large (limit 4 GB)' : `Upload failed (HTTP ${xhr.status})`); } }; xhr.onerror = () => { msg.className = 'msg err'; msg.textContent = 'Upload failed — network error'; }; - xhr.send(fd); + xhr.send(f); }); $('upTable').addEventListener('click', async (e) => { const b = e.target.closest('button'); if (!b) return; diff --git a/server/db.js b/server/db.js index 0d89eb6..88637a4 100644 --- a/server/db.js +++ b/server/db.js @@ -713,6 +713,10 @@ export async function createUpload(u) { }); } +export async function setUploadArt(id, art) { + await db.execute({ sql: 'UPDATE uploads SET art = ? WHERE id = ?', args: [art, id] }); +} + export async function getUpload(id) { const r = await db.execute({ sql: 'SELECT * FROM uploads WHERE id = ?', args: [id] }); return r.rows[0] ? uploadRow(r.rows[0]) : null; diff --git a/server/media-cache.js b/server/media-cache.js index 651ec78..fc04a28 100644 --- a/server/media-cache.js +++ b/server/media-cache.js @@ -162,7 +162,7 @@ export function createMediaCache({ ffmpeg = 'ffmpeg', ffprobe = 'ffprobe', nice = 'nice', maxBytes = 10 * 1024 ** 3, minFreeBytes = 5 * 1024 ** 3, - autoMaxSeconds = 3600, + autoMaxSeconds = 3 * 3600, saveMaxSeconds = 3 * 3600, transcode = { enabled: true, codec: 'hevc', crf: 28, preset: 'medium', threads: 2, maxSeconds: 3600 }, freeBytes = () => { const s = statfsSync(dir); return s.bavail * s.bsize; }, diff --git a/server/server.js b/server/server.js index a7bc571..1bf8802 100644 --- a/server/server.js +++ b/server/server.js @@ -1105,7 +1105,7 @@ const media = createMediaCache({ ffprobe: process.env.FFPROBE_PATH || 'ffprobe', maxBytes: envNum('MEDIA_CACHE_MAX_BYTES', 10 * 1024 ** 3), minFreeBytes: envNum('MEDIA_MIN_FREE_BYTES', 5 * 1024 ** 3), - autoMaxSeconds: envNum('MEDIA_AUTO_MAX_SECONDS', 3600), + autoMaxSeconds: envNum('MEDIA_AUTO_MAX_SECONDS', 3 * 3600), saveMaxSeconds: MAX_SAVE_SECONDS, transcode: { enabled: process.env.MEDIA_TRANSCODE !== '0', @@ -2156,6 +2156,10 @@ async function main() { // 0 disables the idle timeout entirely (verified with a 300s request); // runaway downloads are bounded by MAX_SAVE_SECONDS + kill-on-disconnect. idleTimeout: 0, + // Bun's default request-body cap is 128 MB, which silently rejected every + // upload / device intake of a long video with a 413. Streaming routes + // enforce their own per-route limits (uploads 4 GB, intake P2P_INTAKE_MAX_BYTES). + maxRequestBodySize: 5 * 1024 ** 3, }); console.log(`[ytplayer] Listening → http://localhost:${PORT}`); diff --git a/server/uploads.js b/server/uploads.js index 147ba63..80a3d02 100644 --- a/server/uploads.js +++ b/server/uploads.js @@ -23,7 +23,7 @@ * ========================================================================== */ import { spawn } from 'node:child_process'; -import { existsSync, mkdirSync, unlinkSync } from 'node:fs'; +import { createWriteStream, existsSync, mkdirSync, unlinkSync } from 'node:fs'; import { join } from 'node:path'; import { randomBytes } from 'node:crypto'; @@ -92,6 +92,63 @@ export function registerUploadRoutes(app, deps) { return JSON.parse(out); } + // Probe a file already on disk, extract cover/lyrics and register the row. + // Shared by the multipart route (small files) and the streaming route. + async function ingest({ id, ext, target, size, filename, fields = {}, artFile = null }) { + const probe = await probeFile(target); + const info = describeProbe(probe, filename); + if (!info.hasAudio && info.kind === 'audio') throw new Error('no audio or video streams found in this file'); + + // Cover art: an uploaded image wins, else the embedded picture. + let art = null; + if (artFile && typeof artFile !== 'string' && artFile.size) { + const tmp = join(uploadDir, `${id}.art.src`); + await Bun.write(tmp, artFile); + try { + await run(ffmpeg, ['-v', 'error', '-y', '-i', tmp, '-frames:v', '1', '-vf', "scale='min(800,iw)':-2", artPath(id)]); + art = 'file'; + } catch { /* unusable image — ignore */ } + try { unlinkSync(tmp); } catch { /* gone */ } + } + if (!art && info.coverIndex !== null) { + try { + await run(ffmpeg, ['-v', 'error', '-y', '-i', target, '-map', `0:${info.coverIndex}`, '-frames:v', '1', '-vf', "scale='min(800,iw)':-2", artPath(id)]); + art = 'embedded'; + } catch { /* cover we can't decode */ } + } + + const row = { + id, kind: info.kind, ext, mime: MIME[ext] || (info.kind === 'video' ? 'video/mp4' : 'audio/mpeg'), + size, art, + title: String(fields.title || info.title || baseName(filename)).slice(0, 300), + artist: String(fields.artist || info.artist || '').slice(0, 200), + album: String(fields.album || info.album || '').slice(0, 200), + duration: info.duration, + }; + await db.createUpload(row); + + // Embedded lyrics → the song's shared lyrics (synced if they are LRC). + let lyricLines = 0; + if (info.lyrics) { + const lines = parseLrc(info.lyrics); + if (lines.length) { + const synced = lines.some((l) => l.t !== null); + const data = sanitizeLyrics({ lines, tags: [synced ? 'from the file (synced)' : 'from the file'], offset: 0 }); + await db.saveNote({ videoId: id, kind: 'lyrics', data, source: 'auto', updatedBy: 'upload', force: true }); + lyricLines = data.lines.length; + } + } + return { upload: await db.getUpload(id), lyricLines }; + } + + const extError = (ext) => `unsupported file type ".${ext}" — video: ${[...VIDEO_EXT].join(', ')}; audio: ${[...AUDIO_EXT].join(', ')}`; + const cleanup = (id, target) => { + try { unlinkSync(target); } catch { /* not written */ } + try { unlinkSync(artPath(id)); } catch { /* none */ } + }; + + // Multipart upload — fine for small files and scripts, but the whole body is + // parsed in memory, so the admin page uses the streaming route below. app.post('/api/admin/uploads', requireAdminOrToken, async (c) => { let body; try { body = await c.req.parseBody(); } catch { return c.json({ ok: false, error: 'send the file as multipart/form-data' }, 400); } @@ -99,65 +156,76 @@ export function registerUploadRoutes(app, deps) { if (!file || typeof file === 'string' || !file.name) return c.json({ ok: false, error: 'no file' }, 400); if (file.size > MAX_BYTES) return c.json({ ok: false, error: 'file is larger than 4 GB' }, 413); const ext = extOf(file.name); - if (!VIDEO_EXT.has(ext) && !AUDIO_EXT.has(ext)) { - return c.json({ ok: false, error: `unsupported file type ".${ext}" — video: ${[...VIDEO_EXT].join(', ')}; audio: ${[...AUDIO_EXT].join(', ')}` }, 415); - } + if (!VIDEO_EXT.has(ext) && !AUDIO_EXT.has(ext)) return c.json({ ok: false, error: extError(ext) }, 415); const id = 'upl_' + randomBytes(6).toString('hex'); const target = join(uploadDir, `${id}.${ext}`); try { await Bun.write(target, file); - const probe = await probeFile(target); - const info = describeProbe(probe, file.name); - if (!info.hasAudio && info.kind === 'audio') throw new Error('no audio or video streams found in this file'); - - // Cover art: an uploaded image wins, else the embedded picture. - let art = null; - const artFile = body.art; - if (artFile && typeof artFile !== 'string' && artFile.size) { - const tmp = join(uploadDir, `${id}.art.src`); - await Bun.write(tmp, artFile); - try { - await run(ffmpeg, ['-v', 'error', '-y', '-i', tmp, '-frames:v', '1', '-vf', "scale='min(800,iw)':-2", artPath(id)]); - art = 'file'; - } catch { /* unusable image — ignore */ } - try { unlinkSync(tmp); } catch { /* gone */ } - } - if (!art && info.coverIndex !== null) { - try { - await run(ffmpeg, ['-v', 'error', '-y', '-i', target, '-map', `0:${info.coverIndex}`, '-frames:v', '1', '-vf', "scale='min(800,iw)':-2", artPath(id)]); - art = 'embedded'; - } catch { /* cover we can't decode */ } - } - - const row = { - id, kind: info.kind, ext, mime: MIME[ext] || (info.kind === 'video' ? 'video/mp4' : 'audio/mpeg'), - size: file.size, art, - title: String(body.title || info.title || baseName(file.name)).slice(0, 300), - artist: String(body.artist || info.artist || '').slice(0, 200), - album: String(body.album || info.album || '').slice(0, 200), - duration: info.duration, - }; - await db.createUpload(row); - - // Embedded lyrics → the song's shared lyrics (synced if they are LRC). - let lyricLines = 0; - if (info.lyrics) { - const lines = parseLrc(info.lyrics); - if (lines.length) { - const synced = lines.some((l) => l.t !== null); - const data = sanitizeLyrics({ lines, tags: [synced ? 'from the file (synced)' : 'from the file'], offset: 0 }); - await db.saveNote({ videoId: id, kind: 'lyrics', data, source: 'auto', updatedBy: 'upload', force: true }); - lyricLines = data.lines.length; - } - } - return c.json({ ok: true, upload: await db.getUpload(id), lyricLines }); + const r = await ingest({ id, ext, target, size: file.size, filename: file.name, fields: body, artFile: body.art }); + return c.json({ ok: true, ...r }); } catch (err) { - try { unlinkSync(target); } catch { /* not written */ } - try { unlinkSync(artPath(id)); } catch { /* none */ } + cleanup(id, target); return c.json({ ok: false, error: err.message }, 500); } }); + // Streaming upload: the raw file is the request body, copied to disk chunk by + // chunk so a multi-GB (hour-long) video never sits in memory. + // PUT /api/admin/uploads/stream?name=talk.mp4&title=&artist=&album= + app.put('/api/admin/uploads/stream', requireAdminOrToken, async (c) => { + const name = String(c.req.query('name') || ''); + const ext = extOf(name); + if (!VIDEO_EXT.has(ext) && !AUDIO_EXT.has(ext)) return c.json({ ok: false, error: extError(ext) }, 415); + const declared = Number(c.req.header('content-length')); + if (declared > MAX_BYTES) return c.json({ ok: false, error: 'file is larger than 4 GB' }, 413); + const reader = c.req.raw.body ? c.req.raw.body.getReader() : null; + if (!reader) return c.json({ ok: false, error: 'empty body' }, 400); + const id = 'upl_' + randomBytes(6).toString('hex'); + const target = join(uploadDir, `${id}.${ext}`); + try { + let got = 0; + const out = createWriteStream(target); + try { + for (;;) { + const { done, value } = await reader.read(); + if (done) break; + got += value.byteLength; + if (got > MAX_BYTES) throw Object.assign(new Error('file is larger than 4 GB'), { status: 413 }); + if (!out.write(value)) await new Promise((r) => out.once('drain', r)); + } + } finally { + await new Promise((r) => out.end(r)); + } + if (!got) throw Object.assign(new Error('empty file'), { status: 400 }); + if (declared > 0 && got !== declared) throw Object.assign(new Error(`upload cut short (${got} of ${declared} bytes)`), { status: 400 }); + const q = (k) => String(c.req.query(k) || ''); + const r = await ingest({ id, ext, target, size: got, filename: name, fields: { title: q('title'), artist: q('artist'), album: q('album') } }); + return c.json({ ok: true, ...r }); + } catch (err) { + cleanup(id, target); + return c.json({ ok: false, error: err.message }, err.status || 500); + } + }); + + // Attach / replace the cover of an existing upload (raw image body). + app.put('/api/admin/uploads/:id/art', requireAdminOrToken, async (c) => { + const id = c.req.param('id'); + if (!UPLOAD_ID_RE.test(id) || !(await db.getUpload(id))) return c.json({ ok: false, error: 'not found' }, 404); + const buf = await c.req.arrayBuffer(); + if (!buf.byteLength || buf.byteLength > 20 * 1024 * 1024) return c.json({ ok: false, error: 'send an image under 20 MB' }, 400); + const tmp = join(uploadDir, `${id}.art.src`); + try { + await Bun.write(tmp, buf); + await run(ffmpeg, ['-v', 'error', '-y', '-i', tmp, '-frames:v', '1', '-vf', "scale='min(800,iw)':-2", artPath(id)]); + await db.setUploadArt(id, 'file'); + return c.json({ ok: true }); + } catch (err) { + return c.json({ ok: false, error: 'unusable image: ' + err.message }, 400); + } finally { + try { unlinkSync(tmp); } catch { /* gone */ } + } + }); + app.delete('/api/admin/uploads/:id', requireAdminOrToken, async (c) => { const id = c.req.param('id'); if (!UPLOAD_ID_RE.test(id)) return c.json({ ok: false, error: 'invalid id' }, 400); diff --git a/server/uploads.test.js b/server/uploads.test.js index 5fb6762..4e9ef7b 100644 --- a/server/uploads.test.js +++ b/server/uploads.test.js @@ -131,3 +131,37 @@ describe('uploads', () => { expect((await app.request(`/api/uploads/${audioId}`)).status).toBe(404); }); }); + +describe('streaming upload (large / long files)', () => { + const put = (path, body, headers = { 'x-admin': 'yes' }) => app.request(path, { method: 'PUT', headers, body }); + + test('raw body is streamed to disk and registered', async () => { + const buf = readFileSync(fx('clip.mp4')); + const r = await put('/api/admin/uploads/stream?name=Long%20Talk.mp4&title=Long%20Talk', buf); + const j = await r.json(); + expect(r.status).toBe(200); + expect(j.upload).toMatchObject({ kind: 'video', title: 'Long Talk', size: buf.length }); + expect(existsSync(join(root, 'uploads', `${j.upload.id}.mp4`))).toBe(true); + }); + + test('needs the admin, a known extension and a non-empty body', async () => { + const buf = readFileSync(fx('clip.mp4')); + expect((await put('/api/admin/uploads/stream?name=a.mp4', buf, {})).status).toBe(401); + expect((await put('/api/admin/uploads/stream?name=a.exe', buf)).status).toBe(415); + expect((await put('/api/admin/uploads/stream?name=a.mp4', '')).status).toBe(400); + }); + + test('a file that is not media is rejected and leaves nothing behind', async () => { + const before = (await db.listUploads({ limit: 500 })).length; + const r = await put('/api/admin/uploads/stream?name=junk.mp4', new Uint8Array(5000).fill(7)); + expect(r.status).toBe(500); + expect((await db.listUploads({ limit: 500 })).length).toBe(before); + }); + + test('cover image can be attached afterwards', async () => { + const up = await (await put('/api/admin/uploads/stream?name=song.mp3', readFileSync(fx('song.mp3')))).json(); + const r = await put(`/api/admin/uploads/${up.upload.id}/art`, readFileSync(fx('cover.png'))); + expect((await r.json()).ok).toBe(true); + expect((await db.getUpload(up.upload.id)).art).toBeTruthy(); + }); +});