feat(watch): refonte pro page lecture + dedup transcript

- Supprime le hint debug 'touche s' (raccourci conserve, tooltip discret)
- server/transcript.mjs: dedupeTranscriptLines collapse les auto-captions
  a fenetre glissante (12 lignes -> 3 sur l'exemple signale), applique dans
  parseJson3/parseVtt/parseXmlCaptions + filet avant cache API
- watch: layout pro (grille 1fr/360px, carte chaine+actions, labels FR,
  panneaus description/download, sidebar A suivre sticky, a11y/focus)
- transcript interactif: recherche, clic timestamp -> seek, suivi auto +
  surlignage ligne active, copie, retry
- tests: 4 nouveaux cas dedupe (23/23 pass), build prod OK
This commit is contained in:
2026-09-26 08:51:04 -04:00
parent 4bdcd393fe
commit 2db1f2deee
5 changed files with 948 additions and 335 deletions
+69 -10
View File
@@ -16,7 +16,7 @@ import * as cheerio from 'cheerio';
import axios from 'axios';
import rumbleRouter from './rumble.mjs';
import { providerRegistry, validateProviders } from './providers/registry.mjs';
import { pickTrack, parseTrackText, parseVtt, orderedTracks, normalizeTranscriptProvider, transcriptTrackExt, looksLikeHtmlError } from './transcript.mjs';
import { pickTrack, parseTrackText, parseVtt, dedupeTranscriptLines, orderedTracks, translatedFallbacks, firstPerLanguage, normalizeTranscriptProvider, transcriptTrackExt, looksLikeHtmlError } from './transcript.mjs';
import {
getUserByUsername,
getUserById,
@@ -80,6 +80,30 @@ import { fetchChannelContent } from './providers/channel-content.mjs';
import { getSearchMode, getYtDlpBin, hasCookiesFile, metricsSnapshot } from './providers/youtube-common.mjs';
import { ytScrapeCacheStats } from './providers/youtube.mjs';
/**
* Options réseau communes pour les appels yt-dlp : cookies YouTube exportés
* (session résidentielle de confiance) + proxy sortant. Sans eux, les IPs
* de datacenter se voient servir du vide/429 sur timedtext et parfois sur
* les dumps. Absents par défaut -> comportement inchangé.
*/
function ytdlpNetOpts() {
const opts = {};
try {
const cookies = String(process.env.YT_COOKIES_FILE || '').trim();
if (cookies) {
try {
if (fs.existsSync(cookies)) opts.cookies = cookies;
else console.warn(`[transcript] YT_COOKIES_FILE introuvable : ${cookies}`);
} catch {}
}
} catch {}
try {
const proxy = String(process.env.YT_EGRESS_PROXY || '').trim();
if (proxy) opts.proxy = proxy;
} catch {}
return opts;
}
const app = express();
const PORT = Number(process.env.PORT || 4000);
// yt-dlp: prefer the newest binary available. The copy bundled with
@@ -2066,7 +2090,9 @@ app.get('/api/trending', async (req, res) => {
// Alias to support Angular dev proxy paths in both dev and production builds
app.use('/proxy/api', r);
// Mount dedicated Rumble router (browse, search, video)
// Idem transcript : exposé sous /api ET /proxy/api (fallback du front Watch).
app.use('/api/rumble', rumbleRouter);
app.use('/proxy/api/rumble', rumbleRouter);
// -------------------- Client config from environment --------------------
// WARNING: Values served here are exposed to the browser.
@@ -2409,7 +2435,7 @@ const TRANSCRIPT_UNSUPPORTED_PROVIDERS = new Set(['twitch', 'odysee', 'rumble'])
* parsed lines with the language that worked, or null. Bounded by a timeout
* (yt-dlp subtitle downloads can hang on .part files); partial results on
* disk are still scanned when yt-dlp exits non-zero. */
async function transcriptViaYtDlp(url, langs) {
async function transcriptViaYtDlp(url, langs, netOpts = {}) {
const osMod = await import('node:os');
const dir = await fs.promises.mkdtemp(path.join(osMod.tmpdir(), 'newtube-transcript-'));
const scanDir = async (wanted) => {
@@ -2457,6 +2483,7 @@ async function transcriptViaYtDlp(url, langs) {
noWarnings: true,
noCheckCertificates: true,
noPlaylist: true,
...netOpts,
output: path.join(dir, '%(id)s'),
});
const timer = setTimeout(() => { try { child.kill('SIGKILL'); } catch {} }, 90000);
@@ -2496,7 +2523,11 @@ async function fetchTimedTextLines(track) {
return parseTrackText(text, transcriptTrackExt(track));
}
app.get('/api/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
// NOTE: montée sur le router `r` (et non `app`) pour être servie sous les
// DEUX préfixes `/api` et `/proxy/api` (cf. apiBase() côté front + rewrite
// du proxy dev). Une route `app.get('/api/...')` ici répondrait index.html
// (fallback SPA) sous `/proxy/api/...` en prod.
r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
try {
const { provider, videoId } = req.params;
const lang = String(req.query.lang || 'fr').slice(0, 12) || 'fr';
@@ -2541,7 +2572,7 @@ app.get('/api/transcript/:provider/:videoId', transcriptLimiter, async (req, res
}
if (!meta) {
try {
const raw = await youtubedl(transcriptUrl, { dumpSingleJson: true, skipDownload: true, noWarnings: true, noCheckCertificates: true });
const raw = await youtubedl(transcriptUrl, { dumpSingleJson: true, skipDownload: true, noWarnings: true, noCheckCertificates: true, ...ytdlpNetOpts() });
meta = (typeof raw === 'string') ? JSON.parse(raw || '{}') : (raw || {});
} catch (e) {
console.error('[transcript] yt-dlp failed:', e?.message || e);
@@ -2561,11 +2592,19 @@ app.get('/api/transcript/:provider/:videoId', transcriptLimiter, async (req, res
// then everything else. Translated tracks are often rate-limited while the
// original still works — never fail on the first track alone.
// Bounded: hammering dozens of timedtext URLs only worsens YouTube 429s.
// YouTube-only last resort: server-side auto-translation (`&tlang=`) of a
// fetchable track when the requested language's own track stays walled.
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
let lines = [];
let workedLang = chosenLang;
let sawRateLimit = false;
for (const cand of orderedTracks(meta, lang).slice(0, 5)) {
const baseCands = normalized === 'youtube'
? firstPerLanguage(orderedTracks(meta, lang), 10)
: orderedTracks(meta, lang).slice(0, 5);
const candidates = normalized === 'youtube'
? baseCands.concat(translatedFallbacks(baseCands, lang)).slice(0, 12)
: baseCands;
for (const cand of candidates) {
let attempt = 0;
for (;;) {
try {
@@ -2595,7 +2634,7 @@ app.get('/api/transcript/:provider/:videoId', transcriptLimiter, async (req, res
// Direct timedtext fetches failed (429 / Sorry pages / impersonation):
// let yt-dlp download the subtitles instead (handles all of the above).
try {
const viaDlp = await transcriptViaYtDlp(transcriptUrl, [lang, chosenLang].filter(Boolean));
const viaDlp = await transcriptViaYtDlp(transcriptUrl, [lang, chosenLang].filter(Boolean), ytdlpNetOpts());
if (viaDlp && viaDlp.lines && viaDlp.lines.length) {
lines = viaDlp.lines;
if (viaDlp.lang) workedLang = viaDlp.lang;
@@ -2616,6 +2655,7 @@ app.get('/api/transcript/:provider/:videoId', transcriptLimiter, async (req, res
rateLimited: sawRateLimit || undefined,
});
}
lines = dedupeTranscriptLines(lines);
const result = { lang: workedLang, available: true, languages: languages || [], lines };
transcriptCacheSet(cacheKey, result);
return res.json(result);
@@ -2629,15 +2669,34 @@ app.get('/api/transcript/:provider/:videoId', transcriptLimiter, async (req, res
const distRoot = path.join(process.cwd(), 'dist');
const distBrowser = path.join(distRoot, 'browser');
const staticDir = fs.existsSync(distBrowser) ? distBrowser : distRoot;
// Mount static files unconditionally; if path missing, it will just not serve anything
app.use(express.static(staticDir, { maxAge: '1h', index: 'index.html' }));
// SPA fallback: any non-API GET should serve index.html
// Mount static files unconditionally; if path missing, it will just not serve anything.
// Hashed bundles (.js/.css) sont immuables -> cache long ; index.html ne doit
// JAMAIS être caché (sinon le navigateur rejoue d'anciens chunks après un
// redéploiement et appelle l'API avec d'anciens formats d'URL).
app.use(express.static(staticDir, {
maxAge: '1h',
index: 'index.html',
setHeaders: (res, filePath) => {
try {
if (String(filePath || '').toLowerCase().endsWith('.html')) {
res.setHeader('Cache-Control', 'no-cache, no-store, must-revalidate');
res.setHeader('Pragma', 'no-cache');
res.removeHeader('Expires');
}
} catch {}
},
}));
// SPA fallback: any non-API GET should serve index.html (toujours frais, cf. ci-dessus)
app.get('*', (req, res, next) => {
try {
const url = req.originalUrl || req.url || '';
if (url.startsWith('/api/')) return next();
const indexPath = path.join(staticDir, 'index.html');
if (fs.existsSync(indexPath)) return res.sendFile(indexPath);
if (fs.existsSync(indexPath)) {
res.setHeader('Cache-Control', 'no-cache, no-store, must-revalidate');
res.setHeader('Pragma', 'no-cache');
return res.sendFile(indexPath);
}
return next();
} catch {
return next();
+81
View File
@@ -13,7 +13,10 @@ import {
parseVtt,
parseTrackText,
parseXmlCaptions,
dedupeTranscriptLines,
orderedTracks,
translatedFallbacks,
firstPerLanguage,
looksLikeHtmlError,
capLines,
normalizeTranscriptProvider,
@@ -151,6 +154,84 @@ describe('orderedTracks + parseTrackText + XML captions', () => {
const v = parseTrackText('WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHello\n', 'vtt');
assert.equal(v.length, 1);
});
it('translatedFallbacks appends &tlang= candidates from other languages only', () => { const cands = [
{ track: { url: 'https://x/t?lang=en-US', ext: 'json3' }, lang: 'en-us' },
{ track: { url: 'https://x/t?lang=fr', ext: 'json3' }, lang: 'fr' },
];
const t = translatedFallbacks(cands, 'en');
assert.equal(t.length, 1);
assert.equal(t[0].lang, 'en');
assert.match(t[0].track.url, /tlang=en/);
// Same primary language sources are skipped, not re-translated.
assert.equal(translatedFallbacks(cands, 'fr').length, 1);
assert.equal(translatedFallbacks(cands, '').length, 0);
});
it('firstPerLanguage keeps one candidate per language in order, bounded', () => {
const cands = [
{ track: { url: 'https://x/a' }, lang: 'en-US' },
{ track: { url: 'https://x/b' }, lang: 'en' },
{ track: { url: 'https://x/c' }, lang: 'fr' },
{ track: { url: 'https://x/d' }, lang: null },
];
const out = firstPerLanguage(cands, 10).map((c) => c.track.url);
assert.deepEqual(out, ['https://x/a', 'https://x/c', 'https://x/d']);
assert.equal(firstPerLanguage(cands, 2).length, 2);
});
});
describe('dedupeTranscriptLines (rolling-window auto-captions)', () => {
it('collapses prefix extensions and exact duplicates', () => {
const lines = dedupeTranscriptLines([
{ t: 0, dur: 1, text: 'Tac.' },
{ t: 1, dur: 1, text: 'Tac.' },
{ t: 1, dur: 2, text: 'Tac. David Prou est avec nous. David,' },
{ t: 2, dur: 1, text: 'David Prou est avec nous. David,' },
{ t: 2, dur: 2, text: 'David Prou est avec nous. David, bonjour.' },
{ t: 3, dur: 1, text: 'bonjour.' },
{ t: 3, dur: 2, text: 'bonjour. Bonjour Benois.' },
]);
assert.ok(lines.length < 7, `expected fewer lines, got ${lines.length}`);
assert.ok(lines.length >= 1);
const joined = lines.map((l) => l.text).join(' ');
assert.match(joined, /David Prou/);
assert.match(joined, /Benois/);
// No word lost, no exact consecutive duplicates remain.
for (let i = 1; i < lines.length; i++) {
assert.notEqual(lines[i].text.trim().toLowerCase(), lines[i - 1].text.trim().toLowerCase());
}
});
it('merges sliding partial overlaps without fragments', () => {
const lines = dedupeTranscriptLines([
{ t: 6, dur: 2, text: "Alors ce roman québécois c'était ça ou" },
{ t: 6, dur: 2, text: "Alors ce roman québécois c'était ça ou mourir ?" },
{ t: 7, dur: 1, text: 'mourir ?' },
]);
assert.equal(lines.length, 1);
assert.match(lines[0].text, /mourir/);
});
it('parseJson3 dedupes rolling events end-to-end', () => {
const lines = parseJson3({
events: [
{ tStartMs: 0, dDurationMs: 1000, segs: [{ utf8: 'Tac.' }] },
{ tStartMs: 1000, dDurationMs: 1000, segs: [{ utf8: 'Tac.' }] },
{ tStartMs: 1000, dDurationMs: 2000, segs: [{ utf8: 'Tac. David Prou est avec nous.' }] },
{ tStartMs: 2000, dDurationMs: 1000, segs: [{ utf8: 'David Prou est avec nous.' }] },
],
});
assert.equal(lines.length, 1);
assert.match(lines[0].text, /David Prou/);
});
it('keeps distinct sentences on separate cues', () => {
const lines = dedupeTranscriptLines([
{ t: 0, dur: 2, text: 'Bonjour à tous.' },
{ t: 10, dur: 2, text: 'On parle intelligence artificielle.' },
{ t: 20, dur: 2, text: 'Merci beaucoup.' },
]);
assert.equal(lines.length, 3);
});
});
describe('capLines + normalizeTranscriptProvider', () => { it('truncates very long transcripts', () => {
+152 -3
View File
@@ -123,6 +123,103 @@ export function orderedTracks(json, lang = 'fr') {
return out;
}
/**
* Collapse YouTube rolling-window auto-captions into readable cues.
*
* Auto-generated tracks repeat a sliding window on every event:
* `Tac.` -> `Tac. David...` -> `David...` -> `David... bonjour.`
* Without dedup the UI shows each partial 2-3x (see issue example).
*
* Strategy (order-preserving, lossless on words):
* - drop exact consecutive duplicates (extend duration instead),
* - if the current cue starts with / contains the previous one (or vice
* versa), keep the longest and extend its duration,
* - else strip the longest suffix->prefix word overlap and merge the
* remainder into the previous cue when timestamps are close;
* long cues are split to keep every line readable (<= ~220 chars).
* @param {TranscriptLine[]} lines raw parsed cues
* @returns {TranscriptLine[]}
*/
export function dedupeTranscriptLines(lines) {
const list = Array.isArray(lines) ? lines : [];
if (list.length < 2) return list.slice();
const MAX_CHARS = 220;
const MAX_GAP = 3.0;
const clean = (s) => String(s || '').replace(/\s+/g, ' ').trim();
const norm = (s) => clean(s).toLowerCase();
const endOf = (l) => Number(l?.t || 0) + Math.max(0, Number(l?.dur || 0));
/** Longest overlap (in words) where suffix(prev) == prefix(curr). */
const wordOverlap = (prevTokens, currTokens) => {
const n = Math.min(prevTokens.length, currTokens.length);
// Don't consume the whole current cue as "overlap".
for (let k = n - 1; k >= 1; k--) {
let ok = true;
for (let i = 0; i < k; i++) {
if (prevTokens[prevTokens.length - k + i] !== currTokens[i]) { ok = false; break; }
}
if (ok) return k;
}
return 0;
};
const out = [];
for (const raw of list) {
const text = clean(raw?.text);
if (!text) continue;
let t = Number(raw?.t || 0);
if (!Number.isFinite(t) || t < 0) t = 0;
let dur = Number(raw?.dur || 0);
if (!Number.isFinite(dur) || dur < 0) dur = 0;
const cur = { t, dur, text };
const prev = out[out.length - 1];
if (!prev) { out.push(cur); continue; }
const pn = norm(prev.text);
const cn = norm(cur.text);
if (!pn || !cn) continue;
// 1. Exact consecutive duplicate -> extend, skip.
if (pn === cn) {
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
continue;
}
// 2. Inclusion / prefix extension (rolling window) -> keep longest.
if (cn.startsWith(pn)) {
prev.text = cur.text;
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
continue;
}
if (pn.includes(cn) && cur.t - prev.t < MAX_GAP) {
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
continue;
}
// 3. Partial sliding overlap -> merge remainder into previous cue.
const prevTokens = pn.split(' ').filter(Boolean);
const currTokens = cn.split(' ').filter(Boolean);
const k = wordOverlap(prevTokens, currTokens);
if (k >= 1 && cur.t - prev.t <= MAX_GAP) {
const overlapChars = prevTokens.slice(prevTokens.length - k).join(' ').length;
if (overlapChars >= 3) {
const remainder = clean(cur.text.split(/\s+/).slice(k).join(' '));
if (!remainder) {
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
continue;
}
if (prev.text.length + remainder.length + 1 <= MAX_CHARS) {
prev.text = `${prev.text} ${remainder}`.trim();
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
continue;
}
// Previous cue already full: emit the new content as its own cue.
out.push({ t: cur.t, dur: cur.dur, text: remainder });
continue;
}
}
out.push(cur);
}
return out;
}
/**
* Parse a YouTube `json3` timedtext payload into normalized lines.
* @param {any} data parsed JSON
@@ -142,7 +239,7 @@ export function parseJson3(data) {
text,
});
}
return capLines(out);
return capLines(dedupeTranscriptLines(out));
}
/** Decode a handful of HTML entities found in VTT payloads. */
@@ -213,7 +310,7 @@ export function parseXmlCaptions(text) {
out.push({ t: start, dur: dur > 0 ? dur : 0, text: cleaned });
}
}
return capLines(out);
return capLines(dedupeTranscriptLines(out));
}
/** Detect YouTube "Sorry / automated queries" style HTML error pages. */
@@ -257,7 +354,7 @@ export function parseVtt(text) {
if (!decoded || /^WEBVTT/i.test(decoded)) continue;
out.push({ t: start, dur: Math.max(0, end - start), text: decoded });
}
return capLines(out);
return capLines(dedupeTranscriptLines(out));
}
/** Bound response volume for very long transcripts (Phase 1 decision: truncate). */
@@ -292,6 +389,58 @@ export function parseTrackText(text, ext = '') {
return parseXmlCaptions(text);
}
/**
* Build auto-translated fallbacks (`&tlang=xx`) from source candidates.
* Used as LAST resort when no source track in the requested language is
* fetchable but another language is (e.g. en-US walled while fr answers):
* YouTube translates server-side, so `fr_url + &tlang=en` yields English
* lines. Bounded to 2 extra candidates. YouTube-only (timedtext URLs).
* @param {Array<{ track: any, lang: string|null }>} candidates
* @param {string} targetLang requested language
* @returns {Array<{ track: any, lang: string|null }>}
*/
export function translatedFallbacks(candidates, targetLang) {
const primary = String(targetLang || '').toLowerCase().split('-')[0];
if (!primary) return [];
const seen = new Set((candidates || []).map((c) => String(c?.track?.url || '')));
const out = [];
for (const c of candidates || []) {
const srcPrimary = String(c?.lang || '').toLowerCase().split('-')[0];
if (!srcPrimary || srcPrimary === primary) continue;
const base = String(c?.track?.url || '');
if (!base) continue;
const url = `${base}${base.includes('?') ? '&' : '?'}tlang=${encodeURIComponent(primary)}`;
if (seen.has(url)) continue;
seen.add(url);
out.push({ track: { ...c.track, url, name: `${c.track.name || srcPrimary} (auto-translated)` }, lang: primary });
if (out.length >= 2) break;
}
return out;
}
/**
* Keep the first candidate of each primary language (`en-US`+`en` -> one),
* preserving order. Guarantees every available language is tried at least
* once when the caller wants "any language" instead of failing after an
* arbitrary slice. Bounded by `max` (default 10).
* @param {Array<{ track: any, lang: string|null }>} candidates
* @param {number} [max]
* @returns {Array<{ track: any, lang: string|null }>}
*/
export function firstPerLanguage(candidates, max = 10) {
const n = Math.max(1, Number(max || 10));
const seen = new Set();
const out = [];
for (const c of candidates || []) {
const p = String(c?.lang || '').toLowerCase().split('-')[0] || 'und';
if (seen.has(p)) continue;
seen.add(p);
out.push(c);
if (out.length >= n) break;
}
return out;
}
/** Map short (`yt`) and long (`youtube`) provider ids to `providerUrlFrom()` names. */
export function normalizeTranscriptProvider(provider) {
const p = String(provider || '').trim().toLowerCase();