Files
NewTube/server/tests/transcript.test.mjs
T
bruno 4bdcd393fe feat(youtube): InnerTube-first search, transcripts and watch-next related (Steps 15-18)
- InnerTube layer via pinned youtubei.js 18.1.0 (no quota, no key):
  search with merged continuations (unlimited pages), watch-next
  related with LockupView mapping, caption-track discovery
- 3-layer dispatcher (YT_SEARCH_MODE, default innertube-first):
  innertube -> yt-dlp scrape -> official API, graceful errors.yt
- Robust yt-dlp binary resolution (YT_DLP_PATH > PATH > bundled)
  with systematic API fallback (fixes spawn ENOENT in UI)
- Transcript: InnerTube caption discovery (YT_TRANSCRIPT_SOURCE),
  reusing pickTrack/orderedTracks/parseTrackText; yt-dlp fallback kept
- Watch: sidebar uses real watch-next related[] (/api/details),
  title-search fallback for other providers
- Cache: memory LRU + SQLite (youtube_search_cache, youtube_metrics),
  never persist empty pages; /healthz observability; /api/trending
- Includes pending Step 15/16 leftovers in same files (suggest,
  test scripts); unrelated provider adapters left uncommitted
2026-09-25 19:38:46 -04:00

213 lines
7.6 KiB
JavaScript

// Step 16 — Transcript tests (offline fixtures for parsers + API contract).
// Run with: npm run test:transcript
import { describe, it } from 'node:test';
import assert from 'node:assert/strict';
import fs from 'node:fs';
import path from 'node:path';
import os from 'node:os';
import { spawn } from 'node:child_process';
import net from 'node:net';
import {
pickTrack,
parseJson3,
parseVtt,
parseTrackText,
parseXmlCaptions,
orderedTracks,
looksLikeHtmlError,
capLines,
normalizeTranscriptProvider,
} from '../transcript.mjs';
const track = (url, ext = 'vtt') => ({ url, ext });
describe('pickTrack', () => {
it('prefers manual subtitles over automatic captions', () => {
const json = {
subtitles: { en: [track('https://x/en.vtt')] },
automatic_captions: { en: [track('https://x/en.auto.vtt')] },
};
const { track: t, lang } = pickTrack(json, 'en');
assert.equal(t.url, 'https://x/en.vtt');
assert.equal(lang, 'en');
});
it('falls back fr -> fr-* -> auto -> first available', () => {
const json = {
subtitles: { 'fr-CA': [track('https://x/frca.vtt')] },
automatic_captions: { en: [track('https://x/en.auto.vtt')] },
};
assert.equal(pickTrack(json, 'fr').lang, 'fr-CA');
// 'de' matches nothing: first available (manual preferred)
const fb = pickTrack(json, 'de');
assert.equal(fb.lang, 'fr-CA');
// auto-only dict
const autoOnly = pickTrack({ automatic_captions: { en: [track('https://x/e.vtt')] } }, 'en');
assert.equal(autoOnly.track.url, 'https://x/e.vtt');
});
it('returns null track when no subtitles exist', () => {
assert.deepEqual(pickTrack({}, 'fr'), { track: null, languages: [], lang: null });
assert.deepEqual(pickTrack({ subtitles: {}, automatic_captions: {} }, 'en'), {
track: null,
languages: [],
lang: null,
});
});
it('exposes the language list for the UI selector', () => {
const { languages } = pickTrack(
{ subtitles: { fr: [track('a')], en: [track('b')] }, automatic_captions: { es: [track('c')] } },
'fr',
);
assert.deepEqual([...languages].sort(), ['en', 'es', 'fr']);
});
});
describe('parseJson3', () => {
it('converts events with segs to normalized lines', () => {
const lines = parseJson3({
events: [
{ tStartMs: 0, dDurationMs: 2500, segs: [{ utf8: 'Bonjour' }] },
{ tStartMs: 2500, dDurationMs: 3100, segs: [{ utf8: 'Bien' }, { utf8: 'venue' }] },
],
});
assert.deepEqual(lines, [
{ t: 0, dur: 2.5, text: 'Bonjour' },
{ t: 2.5, dur: 3.1, text: 'Bienvenue' },
]);
});
it('drops events without segs or with empty text', () => {
const lines = parseJson3({ events: [{ tStartMs: 0 }, { tStartMs: 1, dDurationMs: 1, segs: [{ utf8: ' ' }] }] });
assert.deepEqual(lines, []);
assert.deepEqual(parseJson3({}), []);
});
});
describe('parseVtt', () => {
const vtt = `WEBVTT
00:00:00.000 --> 00:00:02.500
Bonjour <c.colorE5E5E5>à tous</c>
12
00:00:02.500 --> 00:00:05.600
Bienvenue
dans cette vidéo
`;
it('parses cues with identifiers, tags and multi-line content', () => {
const lines = parseVtt(vtt);
assert.equal(lines.length, 2);
assert.equal(lines[0].t, 0);
assert.equal(lines[0].dur, 2.5);
assert.equal(lines[0].text, 'Bonjour à tous');
assert.equal(lines[1].text, 'Bienvenue dans cette vidéo');
assert.ok(Math.abs(lines[1].t - 2.5) < 1e-9);
});
it('supports MM:SS.mmm timestamps and decodes entities', () => {
const lines = parseVtt('WEBVTT\n\n01:02.000 --> 01:04.500\nFish &amp; chips\n');
assert.equal(lines.length, 1);
assert.equal(lines[0].t, 62);
assert.equal(lines[0].text, 'Fish & chips');
});
it('ignores garbage blocks', () => {
assert.deepEqual(parseVtt('WEBVTT\n\nNOTE nothing here\n'), []);
});
});
describe('orderedTracks + parseTrackText + XML captions', () => {
const auto = {
fr: [{ url: 'https://x/fr.json3', ext: 'json3' }],
en: [{ url: 'https://x/en.json3', ext: 'json3' }],
es: [{ url: 'https://x/es.vtt', ext: 'vtt' }],
};
it('tries requested lang first, then original en', () => {
const langs = orderedTracks({ subtitles: {}, automatic_captions: auto }, 'fr').map((o) => o.lang);
assert.deepEqual(langs, ['fr', 'en', 'es']);
});
it('dedupes identical track URLs', () => {
const dup = { subtitles: {}, automatic_captions: { en: [{ url: 'https://x/same', ext: 'vtt' }], fr: [{ url: 'https://x/same', ext: 'vtt' }] } };
assert.equal(orderedTracks(dup, 'fr').length, 1);
});
it('detects HTML error pages instead of JSON-parsing them', () => {
assert.equal(looksLikeHtmlError('<!DOCTYPE html><html>Sorry</html>'), true);
assert.equal(looksLikeHtmlError('WEBVTT\n\n00:00:00.000 --> 1'), false);
assert.deepEqual(parseTrackText('<!DOCTYPE html> nope', 'json3'), []);
});
it('parses srv/ttml XML captions', () => {
const srv = parseXmlCaptions('<transcript><p t="0" d="2500">Hello</p></transcript>');
assert.equal(srv.length, 1);
assert.equal(srv[0].text, 'Hello');
const ttml = parseXmlCaptions('<tt><body><div><p begin="00:00:01.000" end="00:00:03.500">Bonjour</p></div></body></tt>');
assert.equal(ttml.length, 1);
assert.equal(ttml[0].t, 1);
});
it('parseTrackText handles json3 and vtt payloads', () => {
const j = parseTrackText(JSON.stringify({ events: [{ tStartMs: 0, dDurationMs: 1000, segs: [{ utf8: 'Hi' }] }] }), 'json3');
assert.equal(j.length, 1);
const v = parseTrackText('WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHello\n', 'vtt');
assert.equal(v.length, 1);
});
});
describe('capLines + normalizeTranscriptProvider', () => { it('truncates very long transcripts', () => {
const lines = Array.from({ length: 10 }, (_, i) => ({ t: i, dur: 1, text: `l${i}` }));
assert.equal(capLines(lines, 3).length, 3);
});
it('maps short and long provider ids', () => {
assert.equal(normalizeTranscriptProvider('yt'), 'youtube');
assert.equal(normalizeTranscriptProvider('youtube'), 'youtube');
assert.equal(normalizeTranscriptProvider('dm'), 'dailymotion');
assert.equal(normalizeTranscriptProvider('pt'), 'peertube');
assert.equal(normalizeTranscriptProvider('xx'), null);
});
});
describe('API contract', () => {
it('rejects unknown providers with 400 { available: false }', async () => {
const port = await new Promise((resolve) => {
const srv = net.createServer();
srv.listen(0, '127.0.0.1', () => {
const p = srv.address().port;
srv.close(() => resolve(p));
});
});
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'newtube-transcript-test-'));
const server = spawn(process.execPath, ['./server/index.mjs'], {
env: {
...process.env,
PORT: String(port),
NEWTUBE_DB_FILE: path.join(tmpDir, 'transcript.db'),
JWT_SECRET: 'transcript-test-secret',
NODE_ENV: 'test',
},
stdio: ['ignore', 'pipe', 'pipe'],
cwd: path.resolve(import.meta.dirname, '..', '..'),
});
const baseUrl = `http://127.0.0.1:${port}`;
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
try {
let up = false;
const start = Date.now();
while (Date.now() - start < 20000) {
try {
const res = await fetch(`${baseUrl}/`);
if (res.status < 500) { up = true; break; }
} catch {}
await sleep(300);
}
assert.ok(up, 'isolated server starts');
const res = await fetch(`${baseUrl}/api/transcript/xx/abc123?lang=fr`);
assert.equal(res.status, 400);
const body = await res.json();
assert.equal(body.available, false);
assert.ok(typeof body.error === 'string');
} finally {
server.kill();
}
});
});