Files
NewTube/server/tests/transcript.test.mjs
T
bruno a57bc0f7ae
CI / build-and-test (push) Successful in 14m5s
fix(transcript): tlang en dernier recours + garde anti-blob
Le fallback &tlang= peut renvoyer la piste source en un seul bloc
de plusieurs Ko (throttle YouTube) : il passait devant les pistes
directes et son blob anglais etait mis en cache comme 'fr'.
tlang desormais etape 3 (apres directes signees), isBlobTranscript()
refuse les cues uniques > 2000 chars (fetch, yt-dlp et post-dedupe).
Tests 27/27.
2026-09-27 20:56:08 -04:00

331 lines
13 KiB
JavaScript

// Step 16 — Transcript tests (offline fixtures for parsers + API contract).
// Run with: npm run test:transcript
import { describe, it } from 'node:test';
import assert from 'node:assert/strict';
import fs from 'node:fs';
import path from 'node:path';
import os from 'node:os';
import { spawn } from 'node:child_process';
import net from 'node:net';
import {
pickTrack,
parseJson3,
parseVtt,
parseTrackText,
parseXmlCaptions,
dedupeTranscriptLines,
orderedTracks,
translatedFallbacks,
firstPerLanguage,
looksLikeHtmlError,
ensureFmtParam,
isEmptyTimedTextBody,
isBlobTranscript,
mergeTranscriptCandidates,
capLines,
normalizeTranscriptProvider,
} from '../transcript.mjs';
const track = (url, ext = 'vtt') => ({ url, ext });
describe('pickTrack', () => {
it('prefers manual subtitles over automatic captions', () => {
const json = {
subtitles: { en: [track('https://x/en.vtt')] },
automatic_captions: { en: [track('https://x/en.auto.vtt')] },
};
const { track: t, lang } = pickTrack(json, 'en');
assert.equal(t.url, 'https://x/en.vtt');
assert.equal(lang, 'en');
});
it('falls back fr -> fr-* -> auto -> first available', () => {
const json = {
subtitles: { 'fr-CA': [track('https://x/frca.vtt')] },
automatic_captions: { en: [track('https://x/en.auto.vtt')] },
};
assert.equal(pickTrack(json, 'fr').lang, 'fr-CA');
// 'de' matches nothing: first available (manual preferred)
const fb = pickTrack(json, 'de');
assert.equal(fb.lang, 'fr-CA');
// auto-only dict
const autoOnly = pickTrack({ automatic_captions: { en: [track('https://x/e.vtt')] } }, 'en');
assert.equal(autoOnly.track.url, 'https://x/e.vtt');
});
it('returns null track when no subtitles exist', () => {
assert.deepEqual(pickTrack({}, 'fr'), { track: null, languages: [], lang: null });
assert.deepEqual(pickTrack({ subtitles: {}, automatic_captions: {} }, 'en'), {
track: null,
languages: [],
lang: null,
});
});
it('exposes the language list for the UI selector', () => {
const { languages } = pickTrack(
{ subtitles: { fr: [track('a')], en: [track('b')] }, automatic_captions: { es: [track('c')] } },
'fr',
);
assert.deepEqual([...languages].sort(), ['en', 'es', 'fr']);
});
});
describe('parseJson3', () => {
it('converts events with segs to normalized lines', () => {
const lines = parseJson3({
events: [
{ tStartMs: 0, dDurationMs: 2500, segs: [{ utf8: 'Bonjour' }] },
{ tStartMs: 2500, dDurationMs: 3100, segs: [{ utf8: 'Bien' }, { utf8: 'venue' }] },
],
});
assert.deepEqual(lines, [
{ t: 0, dur: 2.5, text: 'Bonjour' },
{ t: 2.5, dur: 3.1, text: 'Bienvenue' },
]);
});
it('drops events without segs or with empty text', () => {
const lines = parseJson3({ events: [{ tStartMs: 0 }, { tStartMs: 1, dDurationMs: 1, segs: [{ utf8: ' ' }] }] });
assert.deepEqual(lines, []);
assert.deepEqual(parseJson3({}), []);
});
});
describe('parseVtt', () => {
const vtt = `WEBVTT
00:00:00.000 --> 00:00:02.500
Bonjour <c.colorE5E5E5>à tous</c>
12
00:00:02.500 --> 00:00:05.600
Bienvenue
dans cette vidéo
`;
it('parses cues with identifiers, tags and multi-line content', () => {
const lines = parseVtt(vtt);
assert.equal(lines.length, 2);
assert.equal(lines[0].t, 0);
assert.equal(lines[0].dur, 2.5);
assert.equal(lines[0].text, 'Bonjour à tous');
assert.equal(lines[1].text, 'Bienvenue dans cette vidéo');
assert.ok(Math.abs(lines[1].t - 2.5) < 1e-9);
});
it('supports MM:SS.mmm timestamps and decodes entities', () => {
const lines = parseVtt('WEBVTT\n\n01:02.000 --> 01:04.500\nFish &amp; chips\n');
assert.equal(lines.length, 1);
assert.equal(lines[0].t, 62);
assert.equal(lines[0].text, 'Fish & chips');
});
it('ignores garbage blocks', () => {
assert.deepEqual(parseVtt('WEBVTT\n\nNOTE nothing here\n'), []);
});
});
describe('orderedTracks + parseTrackText + XML captions', () => {
const auto = {
fr: [{ url: 'https://x/fr.json3', ext: 'json3' }],
en: [{ url: 'https://x/en.json3', ext: 'json3' }],
es: [{ url: 'https://x/es.vtt', ext: 'vtt' }],
};
it('tries requested lang first, then original en', () => {
const langs = orderedTracks({ subtitles: {}, automatic_captions: auto }, 'fr').map((o) => o.lang);
assert.deepEqual(langs, ['fr', 'en', 'es']);
});
it('dedupes identical track URLs', () => {
const dup = { subtitles: {}, automatic_captions: { en: [{ url: 'https://x/same', ext: 'vtt' }], fr: [{ url: 'https://x/same', ext: 'vtt' }] } };
assert.equal(orderedTracks(dup, 'fr').length, 1);
});
it('detects HTML error pages instead of JSON-parsing them', () => {
assert.equal(looksLikeHtmlError('<!DOCTYPE html><html>Sorry</html>'), true);
assert.equal(looksLikeHtmlError('WEBVTT\n\n00:00:00.000 --> 1'), false);
assert.deepEqual(parseTrackText('<!DOCTYPE html> nope', 'json3'), []);
});
it('parses srv/ttml XML captions', () => {
const srv = parseXmlCaptions('<transcript><p t="0" d="2500">Hello</p></transcript>');
assert.equal(srv.length, 1);
assert.equal(srv[0].text, 'Hello');
const ttml = parseXmlCaptions('<tt><body><div><p begin="00:00:01.000" end="00:00:03.500">Bonjour</p></div></body></tt>');
assert.equal(ttml.length, 1);
assert.equal(ttml[0].t, 1);
});
it('parseTrackText handles json3 and vtt payloads', () => {
const j = parseTrackText(JSON.stringify({ events: [{ tStartMs: 0, dDurationMs: 1000, segs: [{ utf8: 'Hi' }] }] }), 'json3');
assert.equal(j.length, 1);
const v = parseTrackText('WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHello\n', 'vtt');
assert.equal(v.length, 1);
});
it('translatedFallbacks appends &tlang= candidates from other languages only', () => { const cands = [
{ track: { url: 'https://x/t?lang=en-US', ext: 'json3' }, lang: 'en-us' },
{ track: { url: 'https://x/t?lang=fr', ext: 'json3' }, lang: 'fr' },
];
const t = translatedFallbacks(cands, 'en');
assert.equal(t.length, 1);
assert.equal(t[0].lang, 'en');
assert.match(t[0].track.url, /tlang=en/);
// Same primary language sources are skipped, not re-translated.
assert.equal(translatedFallbacks(cands, 'fr').length, 1);
assert.equal(translatedFallbacks(cands, '').length, 0);
});
it('firstPerLanguage keeps one candidate per language in order, bounded', () => {
const cands = [
{ track: { url: 'https://x/a' }, lang: 'en-US' },
{ track: { url: 'https://x/b' }, lang: 'en' },
{ track: { url: 'https://x/c' }, lang: 'fr' },
{ track: { url: 'https://x/d' }, lang: null },
];
const out = firstPerLanguage(cands, 10).map((c) => c.track.url);
assert.deepEqual(out, ['https://x/a', 'https://x/c', 'https://x/d']);
assert.equal(firstPerLanguage(cands, 2).length, 2);
});
});
describe('dedupeTranscriptLines (rolling-window auto-captions)', () => {
it('collapses prefix extensions and exact duplicates', () => {
const lines = dedupeTranscriptLines([
{ t: 0, dur: 1, text: 'Tac.' },
{ t: 1, dur: 1, text: 'Tac.' },
{ t: 1, dur: 2, text: 'Tac. David Prou est avec nous. David,' },
{ t: 2, dur: 1, text: 'David Prou est avec nous. David,' },
{ t: 2, dur: 2, text: 'David Prou est avec nous. David, bonjour.' },
{ t: 3, dur: 1, text: 'bonjour.' },
{ t: 3, dur: 2, text: 'bonjour. Bonjour Benois.' },
]);
assert.ok(lines.length < 7, `expected fewer lines, got ${lines.length}`);
assert.ok(lines.length >= 1);
const joined = lines.map((l) => l.text).join(' ');
assert.match(joined, /David Prou/);
assert.match(joined, /Benois/);
// No word lost, no exact consecutive duplicates remain.
for (let i = 1; i < lines.length; i++) {
assert.notEqual(lines[i].text.trim().toLowerCase(), lines[i - 1].text.trim().toLowerCase());
}
});
it('merges sliding partial overlaps without fragments', () => {
const lines = dedupeTranscriptLines([
{ t: 6, dur: 2, text: "Alors ce roman québécois c'était ça ou" },
{ t: 6, dur: 2, text: "Alors ce roman québécois c'était ça ou mourir ?" },
{ t: 7, dur: 1, text: 'mourir ?' },
]);
assert.equal(lines.length, 1);
assert.match(lines[0].text, /mourir/);
});
it('parseJson3 dedupes rolling events end-to-end', () => {
const lines = parseJson3({
events: [
{ tStartMs: 0, dDurationMs: 1000, segs: [{ utf8: 'Tac.' }] },
{ tStartMs: 1000, dDurationMs: 1000, segs: [{ utf8: 'Tac.' }] },
{ tStartMs: 1000, dDurationMs: 2000, segs: [{ utf8: 'Tac. David Prou est avec nous.' }] },
{ tStartMs: 2000, dDurationMs: 1000, segs: [{ utf8: 'David Prou est avec nous.' }] },
],
});
assert.equal(lines.length, 1);
assert.match(lines[0].text, /David Prou/);
});
it('keeps distinct sentences on separate cues', () => {
const lines = dedupeTranscriptLines([
{ t: 0, dur: 2, text: 'Bonjour à tous.' },
{ t: 10, dur: 2, text: 'On parle intelligence artificielle.' },
{ t: 20, dur: 2, text: 'Merci beaucoup.' },
]);
assert.equal(lines.length, 3);
});
});
describe('timedtext fetch guards (anti-429 datacenter)', () => {
it('ensureFmtParam never duplicates fmt on signed urls', () => {
const signed = 'https://www.youtube.com/api/timedtext?v=abc&fmt=json3&signature=XYZ';
assert.equal(ensureFmtParam(signed), signed);
const bare = 'https://www.youtube.com/api/timedtext?v=abc&caps=asr';
assert.match(ensureFmtParam(bare), /fmt=json3/);
assert.equal(ensureFmtParam('', 'json3'), '');
});
it('isEmptyTimedTextBody treats 200-empty as transient', () => {
assert.equal(isEmptyTimedTextBody(''), true);
assert.equal(isEmptyTimedTextBody(' '), true);
assert.equal(isEmptyTimedTextBody('<!DOCTYPE html> sorry'), true);
assert.equal(isEmptyTimedTextBody(JSON.stringify({ events: [{ tStartMs: 0, dDurationMs: 1, segs: [{ utf8: 'Hi' }] }] })), false);
assert.equal(isEmptyTimedTextBody('WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHello\n'), false);
});
it('mergeTranscriptCandidates prefers signed urls, dedupes, bounds', () => { const s = { track: { url: 'https://x/signed?sig=1' }, lang: 'fr' };
const dup = { track: { url: 'https://x/signed?sig=1' }, lang: 'fr' };
const it = { track: { url: 'https://x/innertube?fmt=json3' }, lang: 'fr' };
const out = mergeTranscriptCandidates({ signed: [s, dup], innertube: [it], max: 5 });
assert.deepEqual(out.map((c) => c.track.url), ['https://x/signed?sig=1', 'https://x/innertube?fmt=json3']);
assert.equal(mergeTranscriptCandidates({ signed: [s], innertube: [it], max: 1 }).length, 1);
});
it('isBlobTranscript rejects single giant cues, accepts real lines', () => {
assert.equal(isBlobTranscript([{ t: 0, dur: 1, text: 'x'.repeat(8581) }]), true);
assert.equal(isBlobTranscript([{ t: 0, dur: 1, text: 'Bonjour' }]), false);
assert.equal(isBlobTranscript([
{ t: 0, dur: 1, text: 'a' },
{ t: 1, dur: 1, text: 'b' },
]), false);
assert.equal(isBlobTranscript([]), false);
});
});
describe('capLines + normalizeTranscriptProvider', () => { it('truncates very long transcripts', () => {
const lines = Array.from({ length: 10 }, (_, i) => ({ t: i, dur: 1, text: `l${i}` }));
assert.equal(capLines(lines, 3).length, 3);
});
it('maps short and long provider ids', () => {
assert.equal(normalizeTranscriptProvider('yt'), 'youtube');
assert.equal(normalizeTranscriptProvider('youtube'), 'youtube');
assert.equal(normalizeTranscriptProvider('dm'), 'dailymotion');
assert.equal(normalizeTranscriptProvider('pt'), 'peertube');
assert.equal(normalizeTranscriptProvider('xx'), null);
});
});
describe('API contract', () => {
it('rejects unknown providers with 400 { available: false }', async () => {
const port = await new Promise((resolve) => {
const srv = net.createServer();
srv.listen(0, '127.0.0.1', () => {
const p = srv.address().port;
srv.close(() => resolve(p));
});
});
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'newtube-transcript-test-'));
const server = spawn(process.execPath, ['./server/index.mjs'], {
env: {
...process.env,
PORT: String(port),
NEWTUBE_DB_FILE: path.join(tmpDir, 'transcript.db'),
JWT_SECRET: 'transcript-test-secret',
NODE_ENV: 'test',
},
stdio: ['ignore', 'pipe', 'pipe'],
cwd: path.resolve(import.meta.dirname, '..', '..'),
});
const baseUrl = `http://127.0.0.1:${port}`;
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
try {
let up = false;
const start = Date.now();
while (Date.now() - start < 20000) {
try {
const res = await fetch(`${baseUrl}/`);
if (res.status < 500) { up = true; break; }
} catch {}
await sleep(300);
}
assert.ok(up, 'isolated server starts');
const res = await fetch(`${baseUrl}/api/transcript/xx/abc123?lang=fr`);
assert.equal(res.status, 400);
const body = await res.json();
assert.equal(body.available, false);
assert.ok(typeof body.error === 'string');
} finally {
server.kill();
}
});
});