From a79831f4062f8e25b6e962c55f7f5b83217854ef Mon Sep 17 00:00:00 2001 From: Bruno Charest Date: Sun, 27 Sep 2026 20:47:28 -0400 Subject: [PATCH] fix(transcript): pipeline URLs signees yt-dlp + fetch durci anti-429 Les URLs timedtext InnerTube repondent 200-vide (throttle YouTube) : sonde rapide max 2 sans retry, puis dump yt-dlp -> URLs signees en direct (max 4), puis --write-sub. 200-vide traite comme 429 (retryAfterSec), fmt= plus jamais duplique (signature), cookies YT_COOKIES_FILE reutilises par le fetch direct. Tests 26/26. --- .env.example | 3 +- docs/API_MCP_GUIDE.md | 4 +- server/index.mjs | 173 ++++++++++++++++-------- server/providers/youtube-innertube.mjs | 8 +- server/tests/transcript.test.mjs | 28 ++++ server/transcript.mjs | 54 ++++++++ src/components/watch/watch.component.ts | 6 +- 7 files changed, 214 insertions(+), 62 deletions(-) diff --git a/.env.example b/.env.example index 75d6b41..7fd69f1 100644 --- a/.env.example +++ b/.env.example @@ -18,7 +18,8 @@ YT_DLP_TIMEOUT_MS=20000 # Anti-ban : cookies exportés (volume :ro en Docker) + PO-Token + proxy sortant. # Les cookies (session résidentielle de confiance exportée depuis votre # navigateur, format Netscape) sont utilisés par les appels yt-dlp -# (transcripts, downloads) : c'est le déblocage le plus fiable quand +# (transcripts, downloads) ET par le fetch timedtext direct : c'est le +# déblocage le plus fiable quand # YouTube sert du vide/429 aux IPs de datacenter sur timedtext. # YT_COOKIES_FILE=/cookies/youtube.txt # YT_PO_TOKEN= diff --git a/docs/API_MCP_GUIDE.md b/docs/API_MCP_GUIDE.md index bf5c5f8..e60f509 100644 --- a/docs/API_MCP_GUIDE.md +++ b/docs/API_MCP_GUIDE.md @@ -155,12 +155,12 @@ Query : `lang` (défaut `fr`, max 12 chars, filtrée par langues autorisées), ` **Transitoire 502 (à réessayer) :** ```json { "available": false, "languages": ["fr","en"], "lines": [], - "error": "transcript_temporarily_unavailable", "retryable": true, "rateLimited": true } + "error": "transcript_temporarily_unavailable", "retryable": true, "rateLimited": true, "retryAfterSec": 60 } ``` **Autres :** `400 { available:false, error:"invalid_provider_or_video" }`, `429 { available:false, error:"rate_limited" }` (10/min). -Pipeline YouTube : découverte InnerTube (`YT_TRANSCRIPT_SOURCE=innertube-first|ytdlp-only|innertube-only`, 0 piste → `no_subtitles` sans yt-dlp) → candidats ordonnés (`orderedTracks` + `firstPerLanguage`, max 6-8) → fetch timedtext (retry 429 : 2 s puis 8 s) → traduction serveur `&tlang=` en dernier recours → fallback `yt-dlp --write-sub --sub-langs` (90 s max). Parsers : `json3` → `vtt` → XML (`srv1/2/3`, `ttml`, ``), dédup rolling-window auto-captions, `TRANSCRIPT_MAX_LINES` (5000). Cache LRU 200 entrées / 24 h (`TRANSCRIPT_CACHE_TTL`). Anti-ban datacenter : `YT_COOKIES_FILE`, `YT_PO_TOKEN`, `YT_EGRESS_PROXY`. +Pipeline YouTube : découverte InnerTube (`YT_TRANSCRIPT_SOURCE=innertube-first|ytdlp-only|innertube-only`, 0 piste → `no_subtitles` sans yt-dlp) → sonde rapide InnerTube (max 2, sans retry : 200-vide ignoré) → dump yt-dlp → URLs signées fraîches en direct (max 3-4, 1 retry 429) → `&tlang=` ciblé → fallback `yt-dlp --write-sub --sub-langs` (90 s max, le plus robuste). 200-vide traité comme 429 (transitoire, `rateLimited`, `retryAfterSec` 30/60). Parsers : `json3` → `vtt` → XML (`srv1/2/3`, `ttml`, ``), dédup rolling-window auto-captions, `TRANSCRIPT_MAX_LINES` (5000). Cache LRU 200 entrées / 24 h (`TRANSCRIPT_CACHE_TTL`). Anti-ban datacenter : `YT_COOKIES_FILE` (réutilisé par le fetch direct + yt-dlp), `YT_PO_TOKEN`, `YT_EGRESS_PROXY`. Exemples : diff --git a/server/index.mjs b/server/index.mjs index 9c27833..0cdea2a 100644 --- a/server/index.mjs +++ b/server/index.mjs @@ -18,7 +18,7 @@ import rumbleRouter from './rumble.mjs'; import { providerRegistry, validateProviders } from './providers/registry.mjs'; import { dedupeSuggestGroups } from './suggest.mjs'; import { fetchWebSuggest, fetchOdyseeLighthouseSuggest } from './suggest-web.mjs'; -import { pickTrack, parseTrackText, parseVtt, dedupeTranscriptLines, orderedTracks, translatedFallbacks, firstPerLanguage, normalizeTranscriptProvider, transcriptTrackExt, looksLikeHtmlError } from './transcript.mjs'; +import { pickTrack, parseTrackText, parseVtt, dedupeTranscriptLines, orderedTracks, translatedFallbacks, firstPerLanguage, normalizeTranscriptProvider, transcriptTrackExt, looksLikeHtmlError, ensureFmtParam, isEmptyTimedTextBody, mergeTranscriptCandidates } from './transcript.mjs'; import { getUserByUsername, getUserById, @@ -3415,20 +3415,56 @@ async function transcriptViaYtDlp(url, langs, netOpts = {}) { } } -/** Fetch one timedtext track URL and parse it (any format). Returns [] on any failure. */ +/** Build a `Cookie` header from a Netscape-format cookies file (YT_COOKIES_FILE). + * Lets plain timedtext fetches reuse the same trusted residential session as + * yt-dlp instead of going out bare (datacenter egress => 200-empty/429). */ +function youtubeCookiesHeader() { + try { + const file = String(process.env.YT_COOKIES_FILE || '').trim(); + if (!file || !fs.existsSync(file)) return null; + const raw = fs.readFileSync(file, 'utf8'); + const pairs = []; + for (const line of raw.split('\n')) { + const l = line.trim(); + if (!l || l.startsWith('#')) continue; + const parts = l.split('\t'); + if (parts.length < 7) continue; + const domain = parts[0] || ''; + const name = parts[5] || ''; + const value = parts[6] || ''; + if (!name || !/youtube\.com|googlevideo\.com|google\.com/i.test(domain)) continue; + pairs.push(`${name.trim()}=${value.trim()}`); + } + return pairs.length ? Array.from(new Set(pairs)).join('; ') : null; + } catch { + return null; + } +} + +/** Fetch one timedtext track URL and parse it (any format). Throws on any failure. */ async function fetchTimedTextLines(track) { - const resp = await fetch(String(track.url), { - headers: { - 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36', - Accept: 'application/json, text/vtt, text/*;q=0.9, */*;q=0.8', - 'Accept-Language': 'en-US,en;q=0.9,fr;q=0.8', - }, - signal: AbortSignal.timeout(15000), + // Les URLs signées yt-dlp portent déjà `fmt=` : ne jamais le dupliquer + // (signature invalidée => 429/Sorry). + const url = ensureFmtParam(String(track?.url || ''), transcriptTrackExt(track) === 'vtt' ? 'vtt' : 'json3'); + const headers = { + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36', + Accept: 'application/json, text/vtt, text/*;q=0.9, */*;q=0.8', + 'Accept-Language': 'en-US,en;q=0.9,fr;q=0.8', + }; + const cookies = youtubeCookiesHeader(); + if (cookies) headers.Cookie = cookies; + const resp = await fetch(url, { + headers, + signal: AbortSignal.timeout(10000), }); if (!resp.ok) throw new Error(`track_fetch_failed:${resp.status}`); const text = await resp.text(); - if (!text || looksLikeHtmlError(text)) throw new Error('track_fetch_failed:html_error_page'); - return parseTrackText(text, transcriptTrackExt(track)); + // Throttle silencieux de YouTube sur les IPs de datacenter : HTTP 200 avec + // corps vide (ou page Sorry). Transitoire, comme un 429 — jamais "no subtitles". + if (isEmptyTimedTextBody(text)) throw new Error('track_fetch_failed:empty_200'); + const lines = parseTrackText(text, transcriptTrackExt(track)); + if (!lines.length) throw new Error('track_fetch_failed:unparsable'); + return lines; } // NOTE: montée sur le router `r` (et non `app`) pour être servie sous les @@ -3483,7 +3519,7 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => { } catch (e) { console.warn('[transcript] innertube discovery failed, fallback yt-dlp:', e?.code || String(e?.message || e).slice(0, 120)); if (transcriptSource === 'innertube-only') { - return res.status(502).json({ available: false, languages: [], lines: [], error: 'transcript_temporarily_unavailable', retryable: true }); + return res.status(502).json({ available: false, languages: [], lines: [], error: 'transcript_temporarily_unavailable', retryable: true, retryAfterSec: 30 }); } } } @@ -3493,7 +3529,7 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => { meta = (typeof raw === 'string') ? JSON.parse(raw || '{}') : (raw || {}); } catch (e) { console.error('[transcript] yt-dlp failed:', e?.message || e); - return res.status(502).json({ available: false, languages: [], lines: [], error: 'transcript_temporarily_unavailable', retryable: true }); + return res.status(502).json({ available: false, languages: [], lines: [], error: 'transcript_temporarily_unavailable', retryable: true, retryAfterSec: 30 }); } } const { track, languages, lang: chosenLang } = pickTrack(meta, lang); @@ -3506,57 +3542,81 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => { transcriptCacheSet(cacheKey, empty); return res.json(empty); } - // Try candidate tracks in order: requested language, then original (`en`), - // then everything else. Translated tracks are often rate-limited while the - // original still works — never fail on the first track alone. - // Bounded: hammering dozens of timedtext URLs only worsens YouTube 429s. - // YouTube-only last resort: server-side auto-translation (`&tlang=`) of a - // fetchable track when the requested language's own track stays walled. + // Stratégie de récupération (résultat des tests live 2026-09-28) : + // 1. Sonde rapide des URLs InnerTube (découverte, max 2, sans retry) : + // elles répondent désormais 200-vide depuis les egress filtrés. + // 2. Dump yt-dlp -> URLs signées fraîches (`signature`/`expire`) fetchées + // en direct (max 3) : c'est la voie rapide qui fonctionne encore. + // 3. Dernier recours : `yt-dlp --write-sub` (client visionos + retries + // internes, le plus robuste : ~6 s, 80+ Ko vérifiés live). + // On ne martèle jamais plus de 2 URLs mortes d'affilée : chaque 200-vide + // ou 429 supplémentaire aggrave le throttle YouTube à la minute. const sleep = (ms) => new Promise((r) => setTimeout(r, ms)); let lines = []; let workedLang = chosenLang; let sawRateLimit = false; + let sawEmpty = false; const allowedSet = new Set(allowed.map(transcriptPrimary)); const keepAllowed = (cands) => (cands || []).filter(c => allowedSet.has(transcriptPrimary(c.lang))); - const baseCands = keepAllowed(normalized === 'youtube' - ? firstPerLanguage(orderedTracks(meta, lang), 6) - : orderedTracks(meta, lang).slice(0, 5)); - // Server-side auto-translation only towards an allowed target language. - const candidates = normalized === 'youtube' - ? baseCands.concat(keepAllowed(translatedFallbacks(baseCands, lang))).slice(0, 8) - : baseCands; - for (const cand of candidates) { - let attempt = 0; - for (;;) { - try { - const parsed = await fetchTimedTextLines(cand.track); - if (parsed && parsed.length) { - lines = parsed; - workedLang = cand.lang || chosenLang; + const markTransient = (msg) => { + if (msg.includes(':429') || msg.includes('Sorry') || msg.includes('sorry')) sawRateLimit = true; + if (msg.includes(':empty_200') || msg.includes(':unparsable') || msg.includes('html_error')) sawEmpty = true; + }; + const tryCandidates = async (cands, { retry429 = false, label = '' } = {}) => { + for (const cand of cands) { + let attempt = 0; + for (;;) { + try { + const parsed = await fetchTimedTextLines(cand.track); + if (parsed && parsed.length) { + lines = parsed; + workedLang = cand.lang || chosenLang; + } + break; + } catch (e) { + const msg = String(e?.message || e); + console.warn(`[transcript] track fetch failed${label ? ` (${label})` : ''}:`, cand.lang, msg.slice(0, 120)); + markTransient(msg); + if (msg.includes(':429') && retry429 && attempt < 1) { + sawRateLimit = true; + attempt += 1; + await sleep(2000 + Math.floor(Math.random() * 1000)); + continue; + } + break; } - break; - } catch (e) { - const msg = String(e?.message || e); - console.warn('[transcript] track fetch failed:', cand.lang, msg); - // Backoff exponentiel + jitter sur 429 (buckets YouTube à la minute) : - // 1 retry @1.5s ne suffit pas quand l'IP est limitée. - if (msg.includes(':429') && attempt < 2) { - sawRateLimit = true; - attempt += 1; - await sleep([2000, 8000][attempt - 1] + Math.floor(Math.random() * 1000)); - continue; - } - if (msg.includes(':429')) sawRateLimit = true; - break; } + if (lines.length) return true; + } + return false; + }; + const innertubeCands = keepAllowed(normalized === 'youtube' + ? firstPerLanguage(orderedTracks(meta, lang), 2) + : orderedTracks(meta, lang).slice(0, 2)); + // Server-side auto-translation only towards an allowed target language. + const tlangCands = normalized === 'youtube' + ? keepAllowed(translatedFallbacks(innertubeCands, lang)).slice(0, 1) + : []; + await tryCandidates(innertubeCands.concat(tlangCands), { retry429: false, label: 'innertube' }); + if (!lines.length && normalized === 'youtube' && transcriptSource !== 'innertube-only') { + // Étape 2 : URLs signées via dump yt-dlp (même quand InnerTube a déjà + // découvert les pistes : ce sont les seules URLs fetchables en direct). + try { + const raw = await youtubedl(transcriptUrl, { dumpSingleJson: true, skipDownload: true, noWarnings: true, noCheckCertificates: true, ...ytdlpNetOpts() }); + const dump = (typeof raw === 'string') ? JSON.parse(raw || '{}') : (raw || {}); + const signedBase = keepAllowed(firstPerLanguage(orderedTracks(dump, lang), 3)); + const signedTlang = keepAllowed(translatedFallbacks(signedBase, lang)).slice(0, 1); + const signed = mergeTranscriptCandidates({ signed: signedBase.concat(signedTlang), innertube: [], max: 4 }); + if (signed.length) { + console.log(`[transcript] retry via signed yt-dlp urls (${signed.length} candidats)`); + await tryCandidates(signed, { retry429: true, label: 'signed' }); + } + } catch (e) { + console.warn('[transcript] signed-url dump failed:', String(e?.message || e).slice(0, 150)); } - if (lines.length) break; - // Refroidissement entre candidats quand YouTube limite (évite d'aggraver le 429). - if (sawRateLimit) await sleep(700); } if (!lines || lines.length === 0) { - // Direct timedtext fetches failed (429 / Sorry pages / impersonation): - // let yt-dlp download the subtitles instead (handles all of the above). + // Étape 3 : téléchargement via yt-dlp (gère impersonation + retries). // Uniquement les langues autorisées. try { const viaDlp = await transcriptViaYtDlp(transcriptUrl, [lang, chosenLang].filter(l => allowedSet.has(transcriptPrimary(l))), ytdlpNetOpts()); @@ -3568,7 +3628,7 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => { } if (!lines || lines.length === 0) { // Subtitle tracks exist but their content could not be retrieved or - // parsed (YouTube 429 / Sorry pages / impersonation): the video HAS + // parsed (YouTube 429 / 200-vide / Sorry pages) : the video HAS // subtitles, so this is transient — 502 with languages + retryable flag, // never cached as "no subtitles". return res.status(502).json({ @@ -3577,7 +3637,8 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => { lines: [], error: 'transcript_temporarily_unavailable', retryable: true, - rateLimited: sawRateLimit || undefined, + rateLimited: (sawRateLimit || sawEmpty) || undefined, + retryAfterSec: sawRateLimit ? 60 : 30, }); } lines = dedupeTranscriptLines(lines); @@ -3586,7 +3647,7 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => { return res.json(result); } catch (e) { console.error('[transcript] unexpected error:', e?.message || e); - return res.status(502).json({ available: false, languages: [], lines: [], error: 'transcript_temporarily_unavailable', retryable: true }); + return res.status(502).json({ available: false, languages: [], lines: [], error: 'transcript_temporarily_unavailable', retryable: true, retryAfterSec: 30 }); } }); diff --git a/server/providers/youtube-innertube.mjs b/server/providers/youtube-innertube.mjs index 68c806a..b3295ee 100644 --- a/server/providers/youtube-innertube.mjs +++ b/server/providers/youtube-innertube.mjs @@ -299,8 +299,12 @@ export function mapCaptionTracks(captionTracks) { if (!base) continue; const lang = normCaptionLang(t?.language_code) || 'und'; const name = t?.name?.text ?? (typeof t?.name === 'string' ? t.name : lang); - const sep = base.includes('?') ? '&' : '?'; - const track = { url: `${base}${sep}fmt=json3`, ext: 'json3', name: String(name) }; + // Ne jamais dupliquer `fmt=` : une URL signée dont on altère la query + // voit sa `signature` invalidée (YouTube répond 429/Sorry). + const trackUrl = /[?&]fmt=/i.test(base) + ? base + : `${base}${base.includes('?') ? '&' : '?'}fmt=json3`; + const track = { url: trackUrl, ext: 'json3', name: String(name) }; const dict = t?.kind === 'asr' ? automatic_captions : subtitles; if (!Array.isArray(dict[lang])) dict[lang] = []; if (!dict[lang].some((x) => x.url === track.url)) dict[lang].push(track); diff --git a/server/tests/transcript.test.mjs b/server/tests/transcript.test.mjs index b5c56de..5ce9785 100644 --- a/server/tests/transcript.test.mjs +++ b/server/tests/transcript.test.mjs @@ -18,6 +18,9 @@ import { translatedFallbacks, firstPerLanguage, looksLikeHtmlError, + ensureFmtParam, + isEmptyTimedTextBody, + mergeTranscriptCandidates, capLines, normalizeTranscriptProvider, } from '../transcript.mjs'; @@ -234,6 +237,31 @@ describe('dedupeTranscriptLines (rolling-window auto-captions)', () => { }); }); +describe('timedtext fetch guards (anti-429 datacenter)', () => { + it('ensureFmtParam never duplicates fmt on signed urls', () => { + const signed = 'https://www.youtube.com/api/timedtext?v=abc&fmt=json3&signature=XYZ'; + assert.equal(ensureFmtParam(signed), signed); + const bare = 'https://www.youtube.com/api/timedtext?v=abc&caps=asr'; + assert.match(ensureFmtParam(bare), /fmt=json3/); + assert.equal(ensureFmtParam('', 'json3'), ''); + }); + it('isEmptyTimedTextBody treats 200-empty as transient', () => { + assert.equal(isEmptyTimedTextBody(''), true); + assert.equal(isEmptyTimedTextBody(' '), true); + assert.equal(isEmptyTimedTextBody(' sorry'), true); + assert.equal(isEmptyTimedTextBody(JSON.stringify({ events: [{ tStartMs: 0, dDurationMs: 1, segs: [{ utf8: 'Hi' }] }] })), false); + assert.equal(isEmptyTimedTextBody('WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHello\n'), false); + }); + it('mergeTranscriptCandidates prefers signed urls, dedupes, bounds', () => { + const s = { track: { url: 'https://x/signed?sig=1' }, lang: 'fr' }; + const dup = { track: { url: 'https://x/signed?sig=1' }, lang: 'fr' }; + const it = { track: { url: 'https://x/innertube?fmt=json3' }, lang: 'fr' }; + const out = mergeTranscriptCandidates({ signed: [s, dup], innertube: [it], max: 5 }); + assert.deepEqual(out.map((c) => c.track.url), ['https://x/signed?sig=1', 'https://x/innertube?fmt=json3']); + assert.equal(mergeTranscriptCandidates({ signed: [s], innertube: [it], max: 1 }).length, 1); + }); +}); + describe('capLines + normalizeTranscriptProvider', () => { it('truncates very long transcripts', () => { const lines = Array.from({ length: 10 }, (_, i) => ({ t: i, dur: 1, text: `l${i}` })); assert.equal(capLines(lines, 3).length, 3); diff --git a/server/transcript.mjs b/server/transcript.mjs index 85037f2..0f5c471 100644 --- a/server/transcript.mjs +++ b/server/transcript.mjs @@ -319,6 +319,60 @@ export function looksLikeHtmlError(text) { return s.startsWith(', innertube?: Array<{track:any,lang:string|null}>, max?: number }} opts + */ +export function mergeTranscriptCandidates({ signed = [], innertube = [], max = 5 } = {}) { + const n = Math.max(1, Number(max || 5)); + const seen = new Set(); + const out = []; + for (const c of [...(signed || []), ...(innertube || [])]) { + const u = String(c?.track?.url || ''); + if (!u || seen.has(u)) continue; + seen.add(u); + out.push(c); + if (out.length >= n) break; + } + return out; +} + function vttTimestampToSeconds(h, m, s, ms) { return Number(h) * 3600 + Number(m) * 60 + Number(s) + Number(ms) / 1000; } diff --git a/src/components/watch/watch.component.ts b/src/components/watch/watch.component.ts index 0fe0ab3..31899b3 100644 --- a/src/components/watch/watch.component.ts +++ b/src/components/watch/watch.component.ts @@ -1121,8 +1121,12 @@ export class WatchComponent implements OnDestroy, AfterViewInit { }; } if (reason === 'temporarily_unavailable' || code === 'transcript_temporarily_unavailable') { + const wait = Number((data as any)?.retryAfterSec || 0); + const hint = Number.isFinite(wait) && wait > 0 && wait <= 600 + ? ` Réessayez dans ~${Math.round(wait)} s.` + : ' Réessayez dans quelques instants.'; return { - message: 'Sous-titres temporairement indisponibles (limite YouTube). Réessayez dans quelques instants.', + message: `Sous-titres temporairement indisponibles (limite YouTube).${hint}`, retryable: true, reason: reason || 'temporarily_unavailable', };