feat(watch): refonte pro page lecture + dedup transcript
- Supprime le hint debug 'touche s' (raccourci conserve, tooltip discret) - server/transcript.mjs: dedupeTranscriptLines collapse les auto-captions a fenetre glissante (12 lignes -> 3 sur l'exemple signale), applique dans parseJson3/parseVtt/parseXmlCaptions + filet avant cache API - watch: layout pro (grille 1fr/360px, carte chaine+actions, labels FR, panneaus description/download, sidebar A suivre sticky, a11y/focus) - transcript interactif: recherche, clic timestamp -> seek, suivi auto + surlignage ligne active, copie, retry - tests: 4 nouveaux cas dedupe (23/23 pass), build prod OK
This commit is contained in:
+152
-3
@@ -123,6 +123,103 @@ export function orderedTracks(json, lang = 'fr') {
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Collapse YouTube rolling-window auto-captions into readable cues.
|
||||
*
|
||||
* Auto-generated tracks repeat a sliding window on every event:
|
||||
* `Tac.` -> `Tac. David...` -> `David...` -> `David... bonjour.`
|
||||
* Without dedup the UI shows each partial 2-3x (see issue example).
|
||||
*
|
||||
* Strategy (order-preserving, lossless on words):
|
||||
* - drop exact consecutive duplicates (extend duration instead),
|
||||
* - if the current cue starts with / contains the previous one (or vice
|
||||
* versa), keep the longest and extend its duration,
|
||||
* - else strip the longest suffix->prefix word overlap and merge the
|
||||
* remainder into the previous cue when timestamps are close;
|
||||
* long cues are split to keep every line readable (<= ~220 chars).
|
||||
* @param {TranscriptLine[]} lines raw parsed cues
|
||||
* @returns {TranscriptLine[]}
|
||||
*/
|
||||
export function dedupeTranscriptLines(lines) {
|
||||
const list = Array.isArray(lines) ? lines : [];
|
||||
if (list.length < 2) return list.slice();
|
||||
const MAX_CHARS = 220;
|
||||
const MAX_GAP = 3.0;
|
||||
const clean = (s) => String(s || '').replace(/\s+/g, ' ').trim();
|
||||
const norm = (s) => clean(s).toLowerCase();
|
||||
const endOf = (l) => Number(l?.t || 0) + Math.max(0, Number(l?.dur || 0));
|
||||
|
||||
/** Longest overlap (in words) where suffix(prev) == prefix(curr). */
|
||||
const wordOverlap = (prevTokens, currTokens) => {
|
||||
const n = Math.min(prevTokens.length, currTokens.length);
|
||||
// Don't consume the whole current cue as "overlap".
|
||||
for (let k = n - 1; k >= 1; k--) {
|
||||
let ok = true;
|
||||
for (let i = 0; i < k; i++) {
|
||||
if (prevTokens[prevTokens.length - k + i] !== currTokens[i]) { ok = false; break; }
|
||||
}
|
||||
if (ok) return k;
|
||||
}
|
||||
return 0;
|
||||
};
|
||||
|
||||
const out = [];
|
||||
for (const raw of list) {
|
||||
const text = clean(raw?.text);
|
||||
if (!text) continue;
|
||||
let t = Number(raw?.t || 0);
|
||||
if (!Number.isFinite(t) || t < 0) t = 0;
|
||||
let dur = Number(raw?.dur || 0);
|
||||
if (!Number.isFinite(dur) || dur < 0) dur = 0;
|
||||
const cur = { t, dur, text };
|
||||
const prev = out[out.length - 1];
|
||||
if (!prev) { out.push(cur); continue; }
|
||||
|
||||
const pn = norm(prev.text);
|
||||
const cn = norm(cur.text);
|
||||
if (!pn || !cn) continue;
|
||||
// 1. Exact consecutive duplicate -> extend, skip.
|
||||
if (pn === cn) {
|
||||
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
||||
continue;
|
||||
}
|
||||
// 2. Inclusion / prefix extension (rolling window) -> keep longest.
|
||||
if (cn.startsWith(pn)) {
|
||||
prev.text = cur.text;
|
||||
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
||||
continue;
|
||||
}
|
||||
if (pn.includes(cn) && cur.t - prev.t < MAX_GAP) {
|
||||
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
||||
continue;
|
||||
}
|
||||
// 3. Partial sliding overlap -> merge remainder into previous cue.
|
||||
const prevTokens = pn.split(' ').filter(Boolean);
|
||||
const currTokens = cn.split(' ').filter(Boolean);
|
||||
const k = wordOverlap(prevTokens, currTokens);
|
||||
if (k >= 1 && cur.t - prev.t <= MAX_GAP) {
|
||||
const overlapChars = prevTokens.slice(prevTokens.length - k).join(' ').length;
|
||||
if (overlapChars >= 3) {
|
||||
const remainder = clean(cur.text.split(/\s+/).slice(k).join(' '));
|
||||
if (!remainder) {
|
||||
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
||||
continue;
|
||||
}
|
||||
if (prev.text.length + remainder.length + 1 <= MAX_CHARS) {
|
||||
prev.text = `${prev.text} ${remainder}`.trim();
|
||||
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
||||
continue;
|
||||
}
|
||||
// Previous cue already full: emit the new content as its own cue.
|
||||
out.push({ t: cur.t, dur: cur.dur, text: remainder });
|
||||
continue;
|
||||
}
|
||||
}
|
||||
out.push(cur);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a YouTube `json3` timedtext payload into normalized lines.
|
||||
* @param {any} data parsed JSON
|
||||
@@ -142,7 +239,7 @@ export function parseJson3(data) {
|
||||
text,
|
||||
});
|
||||
}
|
||||
return capLines(out);
|
||||
return capLines(dedupeTranscriptLines(out));
|
||||
}
|
||||
|
||||
/** Decode a handful of HTML entities found in VTT payloads. */
|
||||
@@ -213,7 +310,7 @@ export function parseXmlCaptions(text) {
|
||||
out.push({ t: start, dur: dur > 0 ? dur : 0, text: cleaned });
|
||||
}
|
||||
}
|
||||
return capLines(out);
|
||||
return capLines(dedupeTranscriptLines(out));
|
||||
}
|
||||
|
||||
/** Detect YouTube "Sorry / automated queries" style HTML error pages. */
|
||||
@@ -257,7 +354,7 @@ export function parseVtt(text) {
|
||||
if (!decoded || /^WEBVTT/i.test(decoded)) continue;
|
||||
out.push({ t: start, dur: Math.max(0, end - start), text: decoded });
|
||||
}
|
||||
return capLines(out);
|
||||
return capLines(dedupeTranscriptLines(out));
|
||||
}
|
||||
|
||||
/** Bound response volume for very long transcripts (Phase 1 decision: truncate). */
|
||||
@@ -292,6 +389,58 @@ export function parseTrackText(text, ext = '') {
|
||||
return parseXmlCaptions(text);
|
||||
}
|
||||
|
||||
/**
|
||||
* Build auto-translated fallbacks (`&tlang=xx`) from source candidates.
|
||||
* Used as LAST resort when no source track in the requested language is
|
||||
* fetchable but another language is (e.g. en-US walled while fr answers):
|
||||
* YouTube translates server-side, so `fr_url + &tlang=en` yields English
|
||||
* lines. Bounded to 2 extra candidates. YouTube-only (timedtext URLs).
|
||||
* @param {Array<{ track: any, lang: string|null }>} candidates
|
||||
* @param {string} targetLang requested language
|
||||
* @returns {Array<{ track: any, lang: string|null }>}
|
||||
*/
|
||||
export function translatedFallbacks(candidates, targetLang) {
|
||||
const primary = String(targetLang || '').toLowerCase().split('-')[0];
|
||||
if (!primary) return [];
|
||||
const seen = new Set((candidates || []).map((c) => String(c?.track?.url || '')));
|
||||
const out = [];
|
||||
for (const c of candidates || []) {
|
||||
const srcPrimary = String(c?.lang || '').toLowerCase().split('-')[0];
|
||||
if (!srcPrimary || srcPrimary === primary) continue;
|
||||
const base = String(c?.track?.url || '');
|
||||
if (!base) continue;
|
||||
const url = `${base}${base.includes('?') ? '&' : '?'}tlang=${encodeURIComponent(primary)}`;
|
||||
if (seen.has(url)) continue;
|
||||
seen.add(url);
|
||||
out.push({ track: { ...c.track, url, name: `${c.track.name || srcPrimary} (auto-translated)` }, lang: primary });
|
||||
if (out.length >= 2) break;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Keep the first candidate of each primary language (`en-US`+`en` -> one),
|
||||
* preserving order. Guarantees every available language is tried at least
|
||||
* once when the caller wants "any language" instead of failing after an
|
||||
* arbitrary slice. Bounded by `max` (default 10).
|
||||
* @param {Array<{ track: any, lang: string|null }>} candidates
|
||||
* @param {number} [max]
|
||||
* @returns {Array<{ track: any, lang: string|null }>}
|
||||
*/
|
||||
export function firstPerLanguage(candidates, max = 10) {
|
||||
const n = Math.max(1, Number(max || 10));
|
||||
const seen = new Set();
|
||||
const out = [];
|
||||
for (const c of candidates || []) {
|
||||
const p = String(c?.lang || '').toLowerCase().split('-')[0] || 'und';
|
||||
if (seen.has(p)) continue;
|
||||
seen.add(p);
|
||||
out.push(c);
|
||||
if (out.length >= n) break;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Map short (`yt`) and long (`youtube`) provider ids to `providerUrlFrom()` names. */
|
||||
export function normalizeTranscriptProvider(provider) {
|
||||
const p = String(provider || '').trim().toLowerCase();
|
||||
|
||||
Reference in New Issue
Block a user