Proxy YouTube playback through server to fix intermittent 403s

Direct googlevideo URLs are bound to the extractor's IP and expire, so the
browser fetching them from a different IP got intermittent 403s on playback.
Route playback through a same-origin /api/play proxy that fetches the stream
server-side (matching the extractor IP) with yt-dlp's own http_headers and
forwards Range headers for seeking; fall back to piping yt-dlp for SABR/itag-18
formats a plain GET can't fetch. Prefer adaptive streams over progressive in
the playback fallback order.
This commit is contained in:
Jonathan Sykes
2026-08-23 15:01:28 +08:00
parent ee78c8561b
commit defc8a7f1a
2 changed files with 227 additions and 31 deletions

View File

@@ -1441,9 +1441,11 @@ const Player = {
} }
}, },
// Ordered list of qualities to try if playback errors out. Progressive // Ordered list of qualities to try if playback errors out. Adaptive
// (single-file, has audio) goes near the front because it's the most reliable; // (video-only + separate audio via /api/play) goes near the front: those
// then we walk from the lowest resolution up. // streams proxy reliably, whereas progressive single-file formats (itag 18
// and friends) are increasingly SABR-gated and 403 even server-side, so they
// now go LAST. We then walk from the lowest resolution up.
fallbackQueue: [], fallbackQueue: [],
fbIndex: 0, fbIndex: 0,
buildFallbackQueue(chosen) { buildFallbackQueue(chosen) {
@@ -1452,8 +1454,8 @@ const Player = {
const queue = []; const queue = [];
const push = (x) => { if (x && !seen.has(x.label)) { seen.add(x.label); queue.push(x); } }; const push = (x) => { if (x && !seen.has(x.label)) { seen.add(x.label); queue.push(x); } };
push(chosen); push(chosen);
qs.filter((q) => q.hasAudio).forEach(push); // progressive = sturdiest qs.filter((q) => !q.hasAudio).forEach(push); // adaptive = sturdiest (proxied)
qs.forEach(push); // then everything, low → high qs.forEach(push); // then everything, incl. progressive
this.fallbackQueue = queue; this.fallbackQueue = queue;
this.fbIndex = 0; this.fbIndex = 0;
}, },

View File

@@ -8,7 +8,8 @@
* Endpoints: * Endpoints:
* GET /api/search?q=<query> yt-dlp search → slim card array * GET /api/search?q=<query> yt-dlp search → slim card array
* GET /api/channel?c=<channel> yt-dlp channel uploads → slim card array * GET /api/channel?c=<channel> yt-dlp channel uploads → slim card array
* GET /api/streams?v=<videoId> yt-dlp stream info → {meta, audioUrl, qualities} * GET /api/streams?v=<videoId> yt-dlp stream info → {meta, audioUrl, qualities} (proxied URLs)
* GET /api/play?v=<id>&f=<fmt> same-origin playback proxy (Range-aware) → media bytes
* GET /api/download/:videoId proxy best progressive stream → binary * GET /api/download/:videoId proxy best progressive stream → binary
* GET /api/version { version } * GET /api/version { version }
* POST /api/user/sync upsert user playlists + last-seen version * POST /api/user/sync upsert user playlists + last-seen version
@@ -308,44 +309,133 @@ app.get('/api/channel', async (c) => {
} }
}); });
// GET /api/streams?v=<videoId> // ============================================================================
// Stream resolution + same-origin playback proxy
// ----------------------------------------------------------------------------
// googlevideo stream URLs are bound to the innertube client AND the IP that
// extracted them, and they expire (~6h via their `expire=` param). Handing
// them straight to the browser — which fetches from a DIFFERENT IP than this
// server — is what produced the intermittent 403s on playback (only the
// download path was ever fixed, in 70b5214). So playback now flows through
// /api/play: the browser hits our own origin, and THIS server fetches the
// googlevideo bytes (its IP matches the extractor) using yt-dlp's own
// http_headers, forwarding the browser's Range header so seeking still works.
// ============================================================================
// videoId -> { info, formats:[{formatId,height,vcodec,acodec,abr,ext,url,headers}], expiresAt }
const streamCache = new Map();
const STREAM_CACHE_MAX = 200;
// googlevideo URLs carry `expire=<unix-seconds>`; return that as an ms epoch.
function parseExpiry(url) {
const m = /[?&]expire=(\d+)/.exec(url || '');
return m ? Number(m[1]) * 1000 : 0;
}
// Resolve (and briefly cache) a video's playable formats via yt-dlp -J. The
// cache spares a fresh ~2-3s yt-dlp run on every Range request the media
// element fires; its TTL is bounded a minute inside the URLs' own expiry.
async function resolveStreams(videoId) {
const now = Date.now();
const cached = streamCache.get(videoId);
if (cached && now < cached.expiresAt) return cached;
const out = await runYtdlp(['-J', '--no-warnings', `https://www.youtube.com/watch?v=${videoId}`]);
const info = JSON.parse(out);
const raw = Array.isArray(info.formats) ? info.formats : [];
const formats = [];
let soonest = Infinity;
for (const f of raw) {
if (!f.url) continue;
const exp = parseExpiry(f.url);
if (exp) soonest = Math.min(soonest, exp);
formats.push({
formatId: String(f.format_id || ''),
height: f.height || 0,
vcodec: f.vcodec,
acodec: f.acodec,
abr: f.abr || 0,
ext: f.ext || '',
url: f.url,
headers: f.http_headers || {},
});
}
const ttlUntil = soonest === Infinity ? now + 30 * 60_000 : soonest - 60_000;
const expiresAt = Math.max(now + 60_000, Math.min(ttlUntil, now + 3 * 3600_000));
const entry = { info, formats, expiresAt };
if (streamCache.size >= STREAM_CACHE_MAX) streamCache.delete(streamCache.keys().next().value);
streamCache.set(videoId, entry);
return entry;
}
const isVideoFmt = (f) => f.vcodec && f.vcodec !== 'none';
const isAudioFmt = (f) => f.acodec && f.acodec !== 'none';
// Highest-bitrate audio-only format (prefer mp4a/m4a).
function pickBestAudio(formats) {
let best = null, bestScore = -1;
for (const f of formats) {
if (isVideoFmt(f) || !isAudioFmt(f)) continue;
let score = f.abr || 0;
if (f.acodec && f.acodec.includes('mp4a')) score += 1000;
if (score > bestScore) { bestScore = score; best = f; }
}
return best;
}
// Pick the format /api/play should serve — by exact id first, then by the same
// height/progressive-first ordering /api/streams used to build the quality.
function pickFormat(formats, { formatId, wantAudio, wantHeight }) {
if (formatId) {
const byId = formats.find((f) => f.formatId === formatId);
if (byId) return byId;
}
if (wantAudio) return pickBestAudio(formats);
if (wantHeight) {
for (const wantProg of [true, false]) {
for (const f of formats) {
if (!isVideoFmt(f)) continue;
if (isAudioFmt(f) !== wantProg) continue;
if ((f.height || 0) === wantHeight) return f;
}
}
}
return null;
}
// GET /api/streams?v=<videoId> — meta + proxied audio/quality URLs.
app.get('/api/streams', async (c) => { app.get('/api/streams', async (c) => {
const videoId = (c.req.query('v') || '').replace(/[/\\:?<>|*"]/g, '').trim(); const videoId = (c.req.query('v') || '').replace(/[/\\:?<>|*"]/g, '').trim();
if (!videoId) return c.json({ ok: false, error: 'missing videoId' }, 400); if (!videoId) return c.json({ ok: false, error: 'missing videoId' }, 400);
try { try {
const url = `https://www.youtube.com/watch?v=${videoId}`; const { info, formats } = await resolveStreams(videoId);
const out = await runYtdlp(['-J', '--no-warnings', url]);
const info = JSON.parse(out);
const formats = Array.isArray(info.formats) ? info.formats : [];
// Best audio-only stream (prefer mp4a/m4a by bitrate) // Proxied audio URL (same-origin, re-resolved fresh on each play).
let bestAudioUrl = null; const bestAudio = pickBestAudio(formats);
let bestAudioScore = -1; const audioUrl = bestAudio
for (const f of formats) { ? `/api/play?v=${videoId}&audio=1&f=${encodeURIComponent(bestAudio.formatId)}`
if (!f.url) continue; : null;
const hasVideo = f.vcodec && f.vcodec !== 'none';
const hasAudio = f.acodec && f.acodec !== 'none';
if (hasVideo || !hasAudio) continue;
let score = f.abr || 0;
if (f.acodec && f.acodec.includes('mp4a')) score += 1000;
if (score > bestAudioScore) { bestAudioScore = score; bestAudioUrl = f.url; }
}
// Quality list — progressive (single-file, hasAudio) first, then adaptive // Quality list — progressive (single-file) first, then adaptive; one entry
// per height. Each URL points at our proxy, carrying the format id so the
// proxy serves the exact same stream this list advertised.
const qualities = []; const qualities = [];
const seen = new Set(); const seen = new Set();
for (const wantProg of [true, false]) { for (const wantProg of [true, false]) {
for (const f of formats) { for (const f of formats) {
if (!f.url) continue; if (!isVideoFmt(f)) continue;
const hasVideo = f.vcodec && f.vcodec !== 'none'; if (isAudioFmt(f) !== wantProg) continue;
const hasAudio = f.acodec && f.acodec !== 'none';
if (!hasVideo) continue;
if ((hasAudio) !== wantProg) continue;
const h = f.height || 0; const h = f.height || 0;
if (h <= 0 || seen.has(h)) continue; if (h <= 0 || seen.has(h)) continue;
seen.add(h); seen.add(h);
qualities.push({ label: h + 'p', height: h, hasAudio: !!hasAudio, url: f.url, ext: f.ext || '' }); qualities.push({
label: h + 'p',
height: h,
hasAudio: wantProg,
url: `/api/play?v=${videoId}&h=${h}&f=${encodeURIComponent(f.formatId)}`,
ext: f.ext || '',
});
} }
} }
qualities.sort((a, b) => b.height - a.height); qualities.sort((a, b) => b.height - a.height);
@@ -362,7 +452,7 @@ app.get('/api/streams', async (c) => {
duration: typeof info.duration === 'number' ? info.duration : 0, duration: typeof info.duration === 'number' ? info.duration : 0,
thumbnail: `https://i.ytimg.com/vi/${videoId}/hqdefault.jpg`, thumbnail: `https://i.ytimg.com/vi/${videoId}/hqdefault.jpg`,
}, },
audioUrl: bestAudioUrl, audioUrl,
qualities, qualities,
}, },
}); });
@@ -371,6 +461,110 @@ app.get('/api/streams', async (c) => {
} }
}); });
// Stream a single format straight out of yt-dlp's stdout. Some formats
// (notably progressive itag 18, and any SABR-gated stream) 403 on a plain GET
// even from the extractor IP — the bytes are only reachable through yt-dlp's
// full protocol. This can't honour a byte Range (yt-dlp writes start-to-end to
// the pipe), so it always answers 200; the media element still plays, it just
// can't seek past what it has buffered. Used only as the /api/play fallback.
//
// It PEEKS the first chunk before committing to a 200: if yt-dlp errors or
// exits without emitting any bytes (stale binary, ffmpeg segfault on merge,
// dead itag), it resolves to null so the caller can return a real error and
// the frontend advances to the next candidate instead of playing an empty 200.
function ytdlpPipeResponse(videoId, formatArgs, contentType) {
return new Promise((resolve) => {
const child = spawn(YTDLP, [
`https://www.youtube.com/watch?v=${videoId}`,
'--no-warnings', '--no-playlist',
...formatArgs,
'-o', '-',
], { stdio: ['ignore', 'pipe', 'ignore'] });
let settled = false;
const finish = (val) => { if (!settled) { settled = true; resolve(val); } };
child.stdout.once('data', (first) => {
// Re-emit the peeked chunk so the response stream is byte-complete.
child.stdout.unshift(first);
finish(new Response(Readable.toWeb(child.stdout), {
status: 200,
headers: {
'Content-Type': contentType,
'Accept-Ranges': 'none',
'Cache-Control': 'no-store',
},
}));
});
child.on('error', () => { try { child.kill(); } catch {} finish(null); });
child.on('close', () => finish(null)); // closed before any data → failure
});
}
// GET /api/play?v=<id>&f=<formatId>[&h=<height>][&audio=1]
// Same-origin streaming proxy for playback. Fetches the googlevideo bytes from
// THIS server (whose IP matches the extractor) with yt-dlp's own http_headers,
// forwarding the browser's Range header so seeking works. This is what makes
// the previously-403ing direct URLs play reliably. If the direct URL still
// 403s (SABR/itag-18 formats reachable only via yt-dlp's protocol), it falls
// back to piping yt-dlp itself.
app.get('/api/play', async (c) => {
const videoId = (c.req.query('v') || '').replace(/[/\\:?<>|*"]/g, '').trim();
if (!videoId) return c.json({ ok: false, error: 'missing videoId' }, 400);
const sel = {
formatId: (c.req.query('f') || '').trim(),
wantAudio: c.req.query('audio') === '1',
wantHeight: Number(c.req.query('h') || 0),
};
const range = c.req.header('range');
// Fetch the chosen format's bytes; on a stale-URL 403/410 drop the cache and
// re-resolve once before giving up.
async function upstreamFetch(forceFresh) {
if (forceFresh) streamCache.delete(videoId);
const { formats } = await resolveStreams(videoId);
const fmt = pickFormat(formats, sel);
if (!fmt) return { error: 'no matching format' };
const headers = { ...fmt.headers };
if (range) headers['Range'] = range;
return { fmt, res: await fetch(fmt.url, { headers, redirect: 'follow' }) };
}
try {
let attempt = await upstreamFetch(false);
if (attempt.error) return c.json({ ok: false, error: attempt.error }, 404);
if (attempt.res.status === 403 || attempt.res.status === 410) {
const retry = await upstreamFetch(true);
if (!retry.error && retry.res) attempt = retry;
}
// Direct URL is genuinely ungettable (SABR / itag-18) — let yt-dlp fetch it.
if (attempt.res.status === 403 || attempt.res.status === 410) {
const fmtArgs = sel.formatId ? ['-f', sel.formatId]
: sel.wantAudio ? ['-f', 'bestaudio[ext=m4a]/bestaudio']
: sel.wantHeight ? ['-f', `best[height=${sel.wantHeight}]/bv*[height=${sel.wantHeight}]`]
: ['-f', 'best'];
const piped = await ytdlpPipeResponse(videoId, fmtArgs, sel.wantAudio ? 'audio/mp4' : 'video/mp4');
if (piped) return piped;
return c.json({ ok: false, error: 'stream unavailable (403)' }, 502);
}
const up = attempt.res;
const headers = new Headers();
for (const k of ['content-type', 'content-length', 'content-range', 'accept-ranges']) {
const v = up.headers.get(k);
if (v) headers.set(k, v);
}
if (!headers.has('accept-ranges')) headers.set('Accept-Ranges', 'bytes');
if (!headers.has('content-type')) headers.set('Content-Type', sel.wantAudio ? 'audio/mp4' : 'video/mp4');
headers.set('Cache-Control', 'no-store');
return new Response(up.body, { status: up.status, headers });
} catch (err) {
return c.json({ ok: false, error: 'stream proxy failed: ' + err.message }, 502);
}
});
// Download via yt-dlp into a self-cleaning temp file, then stream it. // Download via yt-dlp into a self-cleaning temp file, then stream it.
// yt-dlp MUST perform the HTTP fetch itself: googlevideo stream URLs are // yt-dlp MUST perform the HTTP fetch itself: googlevideo stream URLs are
// bound to the innertube client that extracted them, so resolving the URL // bound to the innertube client that extracted them, so resolving the URL