fix(transcript): tlang en dernier recours + garde anti-blob
CI / build-and-test (push) Successful in 14m5s
CI / build-and-test (push) Successful in 14m5s
Le fallback &tlang= peut renvoyer la piste source en un seul bloc de plusieurs Ko (throttle YouTube) : il passait devant les pistes directes et son blob anglais etait mis en cache comme 'fr'. tlang desormais etape 3 (apres directes signees), isBlobTranscript() refuse les cues uniques > 2000 chars (fetch, yt-dlp et post-dedupe). Tests 27/27.
This commit is contained in:
+43
-16
@@ -18,7 +18,7 @@ import rumbleRouter from './rumble.mjs';
|
||||
import { providerRegistry, validateProviders } from './providers/registry.mjs';
|
||||
import { dedupeSuggestGroups } from './suggest.mjs';
|
||||
import { fetchWebSuggest, fetchOdyseeLighthouseSuggest } from './suggest-web.mjs';
|
||||
import { pickTrack, parseTrackText, parseVtt, dedupeTranscriptLines, orderedTracks, translatedFallbacks, firstPerLanguage, normalizeTranscriptProvider, transcriptTrackExt, looksLikeHtmlError, ensureFmtParam, isEmptyTimedTextBody, mergeTranscriptCandidates } from './transcript.mjs';
|
||||
import { pickTrack, parseTrackText, parseVtt, dedupeTranscriptLines, orderedTracks, translatedFallbacks, firstPerLanguage, normalizeTranscriptProvider, transcriptTrackExt, looksLikeHtmlError, ensureFmtParam, isEmptyTimedTextBody, isBlobTranscript, mergeTranscriptCandidates } from './transcript.mjs';
|
||||
import {
|
||||
getUserByUsername,
|
||||
getUserById,
|
||||
@@ -3543,11 +3543,14 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
return res.json(empty);
|
||||
}
|
||||
// Stratégie de récupération (résultat des tests live 2026-09-28) :
|
||||
// 1. Sonde rapide des URLs InnerTube (découverte, max 2, sans retry) :
|
||||
// 1. Sonde rapide des URLs InnerTube directes (max 2, sans retry) :
|
||||
// elles répondent désormais 200-vide depuis les egress filtrés.
|
||||
// 2. Dump yt-dlp -> URLs signées fraîches (`signature`/`expire`) fetchées
|
||||
// en direct (max 3) : c'est la voie rapide qui fonctionne encore.
|
||||
// 3. Dernier recours : `yt-dlp --write-sub` (client visionos + retries
|
||||
// 3. Traduction serveur `&tlang=` (dernier recours avant yt-dlp) : sous
|
||||
// throttle, YouTube peut renvoyer la piste source en UN seul bloc au
|
||||
// lieu de la traduire — le garde anti-blob refuse ces réponses.
|
||||
// 4. Dernier recours : `yt-dlp --write-sub` (client visionos + retries
|
||||
// internes, le plus robuste : ~6 s, 80+ Ko vérifiés live).
|
||||
// On ne martèle jamais plus de 2 URLs mortes d'affilée : chaque 200-vide
|
||||
// ou 429 supplémentaire aggrave le throttle YouTube à la minute.
|
||||
@@ -3560,7 +3563,7 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
const keepAllowed = (cands) => (cands || []).filter(c => allowedSet.has(transcriptPrimary(c.lang)));
|
||||
const markTransient = (msg) => {
|
||||
if (msg.includes(':429') || msg.includes('Sorry') || msg.includes('sorry')) sawRateLimit = true;
|
||||
if (msg.includes(':empty_200') || msg.includes(':unparsable') || msg.includes('html_error')) sawEmpty = true;
|
||||
if (msg.includes(':empty_200') || msg.includes(':unparsable') || msg.includes(':blob') || msg.includes('html_error')) sawEmpty = true;
|
||||
};
|
||||
const tryCandidates = async (cands, { retry429 = false, label = '' } = {}) => {
|
||||
for (const cand of cands) {
|
||||
@@ -3568,9 +3571,12 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
for (;;) {
|
||||
try {
|
||||
const parsed = await fetchTimedTextLines(cand.track);
|
||||
if (parsed && parsed.length) {
|
||||
if (parsed && parsed.length && !isBlobTranscript(parsed)) {
|
||||
lines = parsed;
|
||||
workedLang = cand.lang || chosenLang;
|
||||
} else if (isBlobTranscript(parsed)) {
|
||||
console.warn(`[transcript] rejected translation blob${label ? ` (${label})` : ''}:`, cand.lang, `1 cue x ${String(parsed[0]?.text || '').length} chars`);
|
||||
markTransient(':blob');
|
||||
}
|
||||
break;
|
||||
} catch (e) {
|
||||
@@ -3593,20 +3599,18 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
const innertubeCands = keepAllowed(normalized === 'youtube'
|
||||
? firstPerLanguage(orderedTracks(meta, lang), 2)
|
||||
: orderedTracks(meta, lang).slice(0, 2));
|
||||
// Server-side auto-translation only towards an allowed target language.
|
||||
const tlangCands = normalized === 'youtube'
|
||||
? keepAllowed(translatedFallbacks(innertubeCands, lang)).slice(0, 1)
|
||||
: [];
|
||||
await tryCandidates(innertubeCands.concat(tlangCands), { retry429: false, label: 'innertube' });
|
||||
await tryCandidates(innertubeCands, { retry429: false, label: 'innertube' });
|
||||
let signedBase = [];
|
||||
if (!lines.length && normalized === 'youtube' && transcriptSource !== 'innertube-only') {
|
||||
// Étape 2 : URLs signées via dump yt-dlp (même quand InnerTube a déjà
|
||||
// découvert les pistes : ce sont les seules URLs fetchables en direct).
|
||||
// Directes uniquement : le `&tlang=` attend l'étape 3 (un blob de
|
||||
// traduction dégradée ne doit jamais éclipser une piste directe).
|
||||
try {
|
||||
const raw = await youtubedl(transcriptUrl, { dumpSingleJson: true, skipDownload: true, noWarnings: true, noCheckCertificates: true, ...ytdlpNetOpts() });
|
||||
const dump = (typeof raw === 'string') ? JSON.parse(raw || '{}') : (raw || {});
|
||||
const signedBase = keepAllowed(firstPerLanguage(orderedTracks(dump, lang), 3));
|
||||
const signedTlang = keepAllowed(translatedFallbacks(signedBase, lang)).slice(0, 1);
|
||||
const signed = mergeTranscriptCandidates({ signed: signedBase.concat(signedTlang), innertube: [], max: 4 });
|
||||
signedBase = keepAllowed(firstPerLanguage(orderedTracks(dump, lang), 3));
|
||||
const signed = mergeTranscriptCandidates({ signed: signedBase, innertube: [], max: 3 });
|
||||
if (signed.length) {
|
||||
console.log(`[transcript] retry via signed yt-dlp urls (${signed.length} candidats)`);
|
||||
await tryCandidates(signed, { retry429: true, label: 'signed' });
|
||||
@@ -3615,14 +3619,24 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
console.warn('[transcript] signed-url dump failed:', String(e?.message || e).slice(0, 150));
|
||||
}
|
||||
}
|
||||
if (!lines.length && normalized === 'youtube') {
|
||||
// Étape 3 : traduction serveur `&tlang=` vers la langue demandée, à
|
||||
// partir des pistes directes qui ont répondu (jamais en premier :
|
||||
// voir garde anti-blob ci-dessus).
|
||||
const tlangCands = keepAllowed(translatedFallbacks(innertubeCands.concat(signedBase), lang)).slice(0, 2);
|
||||
if (tlangCands.length) await tryCandidates(tlangCands, { retry429: false, label: 'tlang' });
|
||||
}
|
||||
if (!lines || lines.length === 0) {
|
||||
// Étape 3 : téléchargement via yt-dlp (gère impersonation + retries).
|
||||
// Uniquement les langues autorisées.
|
||||
// Étape 4 : téléchargement via yt-dlp (gère impersonation + retries).
|
||||
// Uniquement les langues autorisées ; blob refusé comme en direct.
|
||||
try {
|
||||
const viaDlp = await transcriptViaYtDlp(transcriptUrl, [lang, chosenLang].filter(l => allowedSet.has(transcriptPrimary(l))), ytdlpNetOpts());
|
||||
if (viaDlp && viaDlp.lines && viaDlp.lines.length && allowedSet.has(transcriptPrimary(viaDlp.lang))) {
|
||||
if (viaDlp && viaDlp.lines && viaDlp.lines.length && !isBlobTranscript(viaDlp.lines) && allowedSet.has(transcriptPrimary(viaDlp.lang))) {
|
||||
lines = viaDlp.lines;
|
||||
if (viaDlp.lang) workedLang = viaDlp.lang;
|
||||
} else if (isBlobTranscript(viaDlp?.lines)) {
|
||||
console.warn('[transcript] rejected yt-dlp blob:', viaDlp?.lang);
|
||||
sawEmpty = true;
|
||||
}
|
||||
} catch {}
|
||||
}
|
||||
@@ -3642,6 +3656,19 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
});
|
||||
}
|
||||
lines = dedupeTranscriptLines(lines);
|
||||
if (isBlobTranscript(lines)) {
|
||||
// Jamais de cache ni de 200 sur un blob : réponse transitoire.
|
||||
console.warn('[transcript] rejected post-dedupe blob');
|
||||
return res.status(502).json({
|
||||
available: false,
|
||||
languages: visibleLanguages,
|
||||
lines: [],
|
||||
error: 'transcript_temporarily_unavailable',
|
||||
retryable: true,
|
||||
rateLimited: true,
|
||||
retryAfterSec: 60,
|
||||
});
|
||||
}
|
||||
const result = { lang: workedLang, available: true, languages: visibleLanguages, lines };
|
||||
transcriptCacheSet(cacheKey, result);
|
||||
return res.json(result);
|
||||
|
||||
@@ -20,6 +20,7 @@ import {
|
||||
looksLikeHtmlError,
|
||||
ensureFmtParam,
|
||||
isEmptyTimedTextBody,
|
||||
isBlobTranscript,
|
||||
mergeTranscriptCandidates,
|
||||
capLines,
|
||||
normalizeTranscriptProvider,
|
||||
@@ -252,14 +253,22 @@ describe('timedtext fetch guards (anti-429 datacenter)', () => {
|
||||
assert.equal(isEmptyTimedTextBody(JSON.stringify({ events: [{ tStartMs: 0, dDurationMs: 1, segs: [{ utf8: 'Hi' }] }] })), false);
|
||||
assert.equal(isEmptyTimedTextBody('WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHello\n'), false);
|
||||
});
|
||||
it('mergeTranscriptCandidates prefers signed urls, dedupes, bounds', () => {
|
||||
const s = { track: { url: 'https://x/signed?sig=1' }, lang: 'fr' };
|
||||
it('mergeTranscriptCandidates prefers signed urls, dedupes, bounds', () => { const s = { track: { url: 'https://x/signed?sig=1' }, lang: 'fr' };
|
||||
const dup = { track: { url: 'https://x/signed?sig=1' }, lang: 'fr' };
|
||||
const it = { track: { url: 'https://x/innertube?fmt=json3' }, lang: 'fr' };
|
||||
const out = mergeTranscriptCandidates({ signed: [s, dup], innertube: [it], max: 5 });
|
||||
assert.deepEqual(out.map((c) => c.track.url), ['https://x/signed?sig=1', 'https://x/innertube?fmt=json3']);
|
||||
assert.equal(mergeTranscriptCandidates({ signed: [s], innertube: [it], max: 1 }).length, 1);
|
||||
});
|
||||
it('isBlobTranscript rejects single giant cues, accepts real lines', () => {
|
||||
assert.equal(isBlobTranscript([{ t: 0, dur: 1, text: 'x'.repeat(8581) }]), true);
|
||||
assert.equal(isBlobTranscript([{ t: 0, dur: 1, text: 'Bonjour' }]), false);
|
||||
assert.equal(isBlobTranscript([
|
||||
{ t: 0, dur: 1, text: 'a' },
|
||||
{ t: 1, dur: 1, text: 'b' },
|
||||
]), false);
|
||||
assert.equal(isBlobTranscript([]), false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('capLines + normalizeTranscriptProvider', () => { it('truncates very long transcripts', () => {
|
||||
|
||||
@@ -352,6 +352,19 @@ export function isEmptyTimedTextBody(text) {
|
||||
return s.trim().length < 32;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when parsed lines look like a degraded server-side translation blob
|
||||
* rather than real cues: a single giant cue (several KB) instead of
|
||||
* timestamped lines. YouTube serves these under throttle on `&tlang=`
|
||||
* requests (source-language text, untranslated, in ONE event). Caching or
|
||||
* labeling them as the target language would poison the UI + cache.
|
||||
* A legitimate single-cue transcript (very short video) is always tiny.
|
||||
* Pure and testable offline.
|
||||
*/
|
||||
export function isBlobTranscript(lines) {
|
||||
if (!Array.isArray(lines) || lines.length !== 1) return false;
|
||||
return String(lines[0]?.text || '').length > 2000;
|
||||
}
|
||||
/**
|
||||
* Merge fetch candidates, signed yt-dlp URLs first (they carry a fresh
|
||||
* `signature`/`expire` and are the only ones plain fetch can still read),
|
||||
|
||||
Reference in New Issue
Block a user