CI / build-and-test (push) Successful in 14m5s
Le fallback &tlang= peut renvoyer la piste source en un seul bloc de plusieurs Ko (throttle YouTube) : il passait devant les pistes directes et son blob anglais etait mis en cache comme 'fr'. tlang desormais etape 3 (apres directes signees), isBlobTranscript() refuse les cues uniques > 2000 chars (fetch, yt-dlp et post-dedupe). Tests 27/27.
526 lines
20 KiB
JavaScript
526 lines
20 KiB
JavaScript
// Step 16 — Pure transcript helpers (no network, no DB).
|
|
// Data source: `yt-dlp --dump-single-json --skip-download` exposes
|
|
// `subtitles` (manual) and `automatic_captions` (auto-generated).
|
|
// Track entries are arrays of { url, ext, name } (or single objects).
|
|
|
|
/**
|
|
* @typedef {{ t: number, dur: number, text: string }} TranscriptLine
|
|
*/
|
|
|
|
const MAX_LINES_DEFAULT = Number(process.env.TRANSCRIPT_MAX_LINES || 5000);
|
|
|
|
/** Normalize a language code for comparison (lowercase, `_` -> `-`). */
|
|
function normLang(code) {
|
|
return String(code || '').trim().toLowerCase().replace(/_/g, '-');
|
|
}
|
|
|
|
/** Pick the first usable track object from an array-or-single entry. */
|
|
function firstTrack(entry) {
|
|
if (!entry) return null;
|
|
const list = Array.isArray(entry) ? entry : [entry];
|
|
return list.find((t) => t && typeof t.url === 'string' && t.url) || null;
|
|
}
|
|
|
|
function trackExtOf(track) {
|
|
const ext = String(track?.ext || '').toLowerCase();
|
|
if (ext) return ext;
|
|
try {
|
|
const u = new URL(String(track?.url || ''));
|
|
const m = /\.([a-z0-9]+)(?:[?#]|$)/i.exec(u.pathname || '');
|
|
if (m) return m[1].toLowerCase();
|
|
} catch {}
|
|
return '';
|
|
}
|
|
|
|
/** Find a language key in `dict` matching `code` exactly or by prefix (`fr` -> `fr-*`). */
|
|
function findLangKey(dict, code) {
|
|
const want = normLang(code);
|
|
if (!want || !dict) return null;
|
|
const keys = Object.keys(dict);
|
|
const exact = keys.find((k) => normLang(k) === want);
|
|
if (exact) return exact;
|
|
// `fr-ca` requested, `fr` available (and vice versa): match on primary subtag
|
|
const primary = want.split('-')[0];
|
|
return keys.find((k) => normLang(k).split('-')[0] === primary) || null;
|
|
}
|
|
|
|
/**
|
|
* Select the best subtitle track.
|
|
* Priority: manual exact > manual prefix/primary > auto exact > auto prefix/primary > first available.
|
|
* @param {any} json yt-dlp dump-single-json
|
|
* @param {string} [lang]
|
|
* @returns {{ track: any|null, languages: string[], lang: string|null }}
|
|
*/
|
|
export function pickTrack(json, lang = 'fr') {
|
|
const manual = (json && json.subtitles) || {};
|
|
const auto = (json && json.automatic_captions) || {};
|
|
const manualKeys = Object.keys(manual);
|
|
const autoKeys = Object.keys(auto);
|
|
const languages = Array.from(new Set([...manualKeys, ...autoKeys]));
|
|
|
|
if (languages.length === 0) return { track: null, languages: [], lang: null };
|
|
|
|
const manualKey = findLangKey(manual, lang);
|
|
if (manualKey) {
|
|
const track = firstTrack(manual[manualKey]);
|
|
if (track) return { track, languages, lang: manualKey };
|
|
}
|
|
const autoKey = findLangKey(auto, lang);
|
|
if (autoKey) {
|
|
const track = firstTrack(auto[autoKey]);
|
|
if (track) return { track, languages, lang: autoKey };
|
|
}
|
|
// Fallback: first available track (manual preferred)
|
|
for (const key of manualKeys) {
|
|
const track = firstTrack(manual[key]);
|
|
if (track) return { track, languages, lang: key };
|
|
}
|
|
for (const key of autoKeys) {
|
|
const track = firstTrack(auto[key]);
|
|
if (track) return { track, languages, lang: key };
|
|
}
|
|
return { track: null, languages, lang: null };
|
|
}
|
|
|
|
/**
|
|
* Order candidate tracks to try when the preferred one cannot be fetched.
|
|
* YouTube auto-translated tracks (`tlang=xx`) are often rate-limited (HTTP 429
|
|
* or "Sorry" HTML pages) while the original language still works: always try
|
|
* the requested language first, then the original (`en`), then everything
|
|
* else (manual preferred). De-duplicated by URL.
|
|
* @param {any} json yt-dlp dump-single-json
|
|
* @param {string} [lang]
|
|
* @returns {Array<{ track: any, lang: string|null }>}
|
|
*/
|
|
export function orderedTracks(json, lang = 'fr') {
|
|
const manual = (json && json.subtitles) || {};
|
|
const auto = (json && json.automatic_captions) || {};
|
|
const seen = new Set();
|
|
const out = [];
|
|
const push = (key, dict) => {
|
|
if (key == null) return;
|
|
const t = firstTrack(dict[key]);
|
|
if (!t || !t.url || seen.has(String(t.url))) return;
|
|
seen.add(String(t.url));
|
|
out.push({ track: t, lang: key });
|
|
};
|
|
const primary = pickTrack(json, lang);
|
|
if (primary.track) {
|
|
seen.add(String(primary.track.url));
|
|
out.push({ track: primary.track, lang: primary.lang });
|
|
}
|
|
// Original language first fallback (avoids translated-track rate limits)
|
|
const reqPrimary = String(lang || '').toLowerCase().split('-')[0];
|
|
for (const fallbackLang of ['en', 'fr']) {
|
|
if (fallbackLang === reqPrimary) continue;
|
|
const k = findLangKey(manual, fallbackLang);
|
|
if (k) push(k, manual);
|
|
const ka = findLangKey(auto, fallbackLang);
|
|
if (ka) push(ka, auto);
|
|
}
|
|
for (const key of Object.keys(manual)) push(key, manual);
|
|
for (const key of Object.keys(auto)) push(key, auto);
|
|
return out;
|
|
}
|
|
|
|
/**
|
|
* Collapse YouTube rolling-window auto-captions into readable cues.
|
|
*
|
|
* Auto-generated tracks repeat a sliding window on every event:
|
|
* `Tac.` -> `Tac. David...` -> `David...` -> `David... bonjour.`
|
|
* Without dedup the UI shows each partial 2-3x (see issue example).
|
|
*
|
|
* Strategy (order-preserving, lossless on words):
|
|
* - drop exact consecutive duplicates (extend duration instead),
|
|
* - if the current cue starts with / contains the previous one (or vice
|
|
* versa), keep the longest and extend its duration,
|
|
* - else strip the longest suffix->prefix word overlap and merge the
|
|
* remainder into the previous cue when timestamps are close;
|
|
* long cues are split to keep every line readable (<= ~220 chars).
|
|
* @param {TranscriptLine[]} lines raw parsed cues
|
|
* @returns {TranscriptLine[]}
|
|
*/
|
|
export function dedupeTranscriptLines(lines) {
|
|
const list = Array.isArray(lines) ? lines : [];
|
|
if (list.length < 2) return list.slice();
|
|
const MAX_CHARS = 220;
|
|
const MAX_GAP = 3.0;
|
|
const clean = (s) => String(s || '').replace(/\s+/g, ' ').trim();
|
|
const norm = (s) => clean(s).toLowerCase();
|
|
const endOf = (l) => Number(l?.t || 0) + Math.max(0, Number(l?.dur || 0));
|
|
|
|
/** Longest overlap (in words) where suffix(prev) == prefix(curr). */
|
|
const wordOverlap = (prevTokens, currTokens) => {
|
|
const n = Math.min(prevTokens.length, currTokens.length);
|
|
// Don't consume the whole current cue as "overlap".
|
|
for (let k = n - 1; k >= 1; k--) {
|
|
let ok = true;
|
|
for (let i = 0; i < k; i++) {
|
|
if (prevTokens[prevTokens.length - k + i] !== currTokens[i]) { ok = false; break; }
|
|
}
|
|
if (ok) return k;
|
|
}
|
|
return 0;
|
|
};
|
|
|
|
const out = [];
|
|
for (const raw of list) {
|
|
const text = clean(raw?.text);
|
|
if (!text) continue;
|
|
let t = Number(raw?.t || 0);
|
|
if (!Number.isFinite(t) || t < 0) t = 0;
|
|
let dur = Number(raw?.dur || 0);
|
|
if (!Number.isFinite(dur) || dur < 0) dur = 0;
|
|
const cur = { t, dur, text };
|
|
const prev = out[out.length - 1];
|
|
if (!prev) { out.push(cur); continue; }
|
|
|
|
const pn = norm(prev.text);
|
|
const cn = norm(cur.text);
|
|
if (!pn || !cn) continue;
|
|
// 1. Exact consecutive duplicate -> extend, skip.
|
|
if (pn === cn) {
|
|
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
|
continue;
|
|
}
|
|
// 2. Inclusion / prefix extension (rolling window) -> keep longest.
|
|
if (cn.startsWith(pn)) {
|
|
prev.text = cur.text;
|
|
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
|
continue;
|
|
}
|
|
if (pn.includes(cn) && cur.t - prev.t < MAX_GAP) {
|
|
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
|
continue;
|
|
}
|
|
// 3. Partial sliding overlap -> merge remainder into previous cue.
|
|
const prevTokens = pn.split(' ').filter(Boolean);
|
|
const currTokens = cn.split(' ').filter(Boolean);
|
|
const k = wordOverlap(prevTokens, currTokens);
|
|
if (k >= 1 && cur.t - prev.t <= MAX_GAP) {
|
|
const overlapChars = prevTokens.slice(prevTokens.length - k).join(' ').length;
|
|
if (overlapChars >= 3) {
|
|
const remainder = clean(cur.text.split(/\s+/).slice(k).join(' '));
|
|
if (!remainder) {
|
|
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
|
continue;
|
|
}
|
|
if (prev.text.length + remainder.length + 1 <= MAX_CHARS) {
|
|
prev.text = `${prev.text} ${remainder}`.trim();
|
|
prev.dur = Math.max(prev.dur, Math.max(0, endOf(cur) - prev.t));
|
|
continue;
|
|
}
|
|
// Previous cue already full: emit the new content as its own cue.
|
|
out.push({ t: cur.t, dur: cur.dur, text: remainder });
|
|
continue;
|
|
}
|
|
}
|
|
out.push(cur);
|
|
}
|
|
return out;
|
|
}
|
|
|
|
/**
|
|
* Parse a YouTube `json3` timedtext payload into normalized lines.
|
|
* @param {any} data parsed JSON
|
|
* @returns {TranscriptLine[]}
|
|
*/
|
|
export function parseJson3(data) {
|
|
const events = (data && data.events) || [];
|
|
if (!Array.isArray(events)) return [];
|
|
const out = [];
|
|
for (const e of events) {
|
|
if (!e || !Array.isArray(e.segs)) continue;
|
|
const text = e.segs.map((s) => String(s?.utf8 ?? '')).join('').replace(/\n+/g, ' ').trim();
|
|
if (!text) continue;
|
|
out.push({
|
|
t: Number(e.tStartMs || 0) / 1000,
|
|
dur: Number(e.dDurationMs || 0) / 1000,
|
|
text,
|
|
});
|
|
}
|
|
return capLines(dedupeTranscriptLines(out));
|
|
}
|
|
|
|
/** Decode a handful of HTML entities found in VTT payloads. */
|
|
function decodeEntities(s) {
|
|
return String(s || '')
|
|
.replace(/&/g, '&')
|
|
.replace(/</g, '<')
|
|
.replace(/>/g, '>')
|
|
.replace(/"/g, '"')
|
|
.replace(/'/g, "'")
|
|
.replace(/ /g, ' ');
|
|
}
|
|
|
|
/** Strip XML/HTML tags and decode entities to plain cue text. */
|
|
function xmlTextToPlain(s) {
|
|
return decodeEntities(String(s || '').replace(/<[^>]*>/g, ' ').replace(/\s+/g, ' ').trim()).trim();
|
|
}
|
|
|
|
function parseTimeAttrToSeconds(v) {
|
|
if (v == null || v === '') return null;
|
|
const s = String(v).trim();
|
|
if (!s) return null;
|
|
if (/^\d+(\.\d+)?$/.test(s)) return Number(s);
|
|
const m = /(?:(\d+):)?([0-5]?\d):([0-5]\d)(?:\.(\d{1,3}))?/.exec(s);
|
|
if (!m) return null;
|
|
return Number(m[1] || 0) * 3600 + Number(m[2]) * 60 + Number(m[3]) + Number((m[4] || '0').padEnd(3, '0')) / 1000;
|
|
}
|
|
|
|
/**
|
|
* Parse YouTube XML caption formats (`srv1`/`srv2`/`srv3` `<p>` cues,
|
|
* `ttml` `<p begin= end=>` cues) into normalized lines.
|
|
* @param {string} text raw XML
|
|
* @returns {TranscriptLine[]}
|
|
*/
|
|
export function parseXmlCaptions(text) {
|
|
const out = [];
|
|
const src = String(text || '');
|
|
if (!src || !/<(p|text|tt|transcript)\b/i.test(src)) return capLines(out);
|
|
// YouTube srv*: <p t="0" d="2500">Hello <s>world</s></p>
|
|
const srvRe = /<p\b[^>]*?(?:t|start)="([^"]*)"[^>]*?(?:d|dur)="([^"]*)"[^>]*>([\s\S]*?)<\/p\s*>/gi;
|
|
// Generic ttml: <p begin="00:00:00.000" end="00:00:02.500">...</p>
|
|
const ttmlRe = /<p\b[^>]*?begin="([^"]*)"[^>]*?end="([^"]*)"[^>]*>([\s\S]*?)<\/p\s*>/gi;
|
|
let m;
|
|
while ((m = srvRe.exec(src)) !== null) {
|
|
const start = Number(m[1]) / 1000;
|
|
const dur = Number(m[2]) / 1000;
|
|
const cleaned = xmlTextToPlain(m[3]);
|
|
if (!cleaned || !Number.isFinite(start)) continue;
|
|
out.push({ t: start, dur: Number.isFinite(dur) && dur > 0 ? dur : 0, text: cleaned });
|
|
}
|
|
if (out.length === 0) {
|
|
while ((m = ttmlRe.exec(src)) !== null) {
|
|
const start = parseTimeAttrToSeconds(m[1]);
|
|
const end = parseTimeAttrToSeconds(m[2]);
|
|
const cleaned = xmlTextToPlain(m[3]);
|
|
if (!cleaned || start == null || end == null) continue;
|
|
out.push({ t: start, dur: Math.max(0, end - start), text: cleaned });
|
|
}
|
|
}
|
|
// Plain <text start="2.5" dur="2.0">Fallback</text> variant
|
|
if (out.length === 0) {
|
|
const textRe = /<text\b[^>]*?start="([^"]*)"[^>]*?(?:dur="([^"]*)")?[^>]*>([\s\S]*?)<\/text\s*>/gi;
|
|
while ((m = textRe.exec(src)) !== null) {
|
|
const start = Number(m[1]);
|
|
const dur = Number(m[2] || 0);
|
|
const cleaned = xmlTextToPlain(m[3]);
|
|
if (!cleaned || !Number.isFinite(start)) continue;
|
|
out.push({ t: start, dur: dur > 0 ? dur : 0, text: cleaned });
|
|
}
|
|
}
|
|
return capLines(dedupeTranscriptLines(out));
|
|
}
|
|
|
|
/** Detect YouTube "Sorry / automated queries" style HTML error pages. */
|
|
export function looksLikeHtmlError(text) {
|
|
const s = String(text || '').trimStart().slice(0, 512).toLowerCase();
|
|
return s.startsWith('<!doctype') || s.startsWith('<html');
|
|
}
|
|
|
|
/**
|
|
* Append `fmt=xx` to a timedtext URL only when it has no `fmt` param yet.
|
|
* Re-signing a signed yt-dlp URL by appending a duplicate `fmt` breaks its
|
|
* `signature` and YouTube answers 429/Sorry: always reuse the URL as-is.
|
|
* Pure and testable offline.
|
|
*/
|
|
export function ensureFmtParam(url, fmt = 'json3') {
|
|
const raw = String(url || '');
|
|
if (!raw) return raw;
|
|
try {
|
|
const u = new URL(raw);
|
|
if (u.searchParams.has('fmt')) return u.toString();
|
|
u.searchParams.set('fmt', fmt);
|
|
return u.toString();
|
|
} catch {
|
|
if (/[?&]fmt=/i.test(raw)) return raw;
|
|
return `${raw}${raw.includes('?') ? '&' : '?'}fmt=${encodeURIComponent(fmt)}`;
|
|
}
|
|
}
|
|
|
|
/**
|
|
* True when a timedtext HTTP 200 body carries no usable caption payload.
|
|
* YouTube throttles datacenter egress IPs with `200 + empty body` (instead
|
|
* of an explicit 429): callers must treat it as transient (same as 429),
|
|
* never as "no subtitles".
|
|
*/
|
|
export function isEmptyTimedTextBody(text) {
|
|
const s = String(text ?? '');
|
|
if (!s) return true;
|
|
if (looksLikeHtmlError(s)) return true;
|
|
return s.trim().length < 32;
|
|
}
|
|
|
|
/**
|
|
* True when parsed lines look like a degraded server-side translation blob
|
|
* rather than real cues: a single giant cue (several KB) instead of
|
|
* timestamped lines. YouTube serves these under throttle on `&tlang=`
|
|
* requests (source-language text, untranslated, in ONE event). Caching or
|
|
* labeling them as the target language would poison the UI + cache.
|
|
* A legitimate single-cue transcript (very short video) is always tiny.
|
|
* Pure and testable offline.
|
|
*/
|
|
export function isBlobTranscript(lines) {
|
|
if (!Array.isArray(lines) || lines.length !== 1) return false;
|
|
return String(lines[0]?.text || '').length > 2000;
|
|
}
|
|
/**
|
|
* Merge fetch candidates, signed yt-dlp URLs first (they carry a fresh
|
|
* `signature`/`expire` and are the only ones plain fetch can still read),
|
|
* then InnerTube base URLs as a quick probe. De-duplicated by URL, bounded.
|
|
* Pure and testable offline.
|
|
* @param {{ signed?: Array<{track:any,lang:string|null}>, innertube?: Array<{track:any,lang:string|null}>, max?: number }} opts
|
|
*/
|
|
export function mergeTranscriptCandidates({ signed = [], innertube = [], max = 5 } = {}) {
|
|
const n = Math.max(1, Number(max || 5));
|
|
const seen = new Set();
|
|
const out = [];
|
|
for (const c of [...(signed || []), ...(innertube || [])]) {
|
|
const u = String(c?.track?.url || '');
|
|
if (!u || seen.has(u)) continue;
|
|
seen.add(u);
|
|
out.push(c);
|
|
if (out.length >= n) break;
|
|
}
|
|
return out;
|
|
}
|
|
|
|
function vttTimestampToSeconds(h, m, s, ms) {
|
|
return Number(h) * 3600 + Number(m) * 60 + Number(s) + Number(ms) / 1000;
|
|
}
|
|
|
|
/**
|
|
* Parse a WebVTT payload into normalized lines.
|
|
* Handles cue identifiers, multi-line cues and inline tags (`<c>`, `<i>`, timestamps).
|
|
* @param {string} text raw VTT
|
|
* @returns {TranscriptLine[]}
|
|
*/
|
|
export function parseVtt(text) {
|
|
if (looksLikeHtmlError(text)) return [];
|
|
const out = [];
|
|
const blocks = String(text || '').replace(/\r\n/g, '\n').split(/\n{2,}/);
|
|
const tsRe = /(?:(\d{2,}):)?([0-5]?\d):([0-5]\d)\.(\d{3})\s*-->\s*(?:(\d{2,}):)?([0-5]?\d):([0-5]\d)\.(\d{3})/;
|
|
for (const block of blocks) {
|
|
const lines = String(block || '').split('\n');
|
|
const timingIdx = lines.findIndex((l) => l.includes('-->'));
|
|
if (timingIdx === -1) continue;
|
|
const m = tsRe.exec(lines[timingIdx]);
|
|
if (!m) continue;
|
|
const start = vttTimestampToSeconds(m[1] || '0', m[2], m[3], m[4]);
|
|
const end = vttTimestampToSeconds(m[5] || '0', m[6], m[7], m[8]);
|
|
const content = lines
|
|
.slice(timingIdx + 1)
|
|
.join(' ')
|
|
// strip inline VTT tags (<c.colorE5E5E5>, <i>, <00:00:01.000>, ...)
|
|
.replace(/<[^>]*>/g, ' ')
|
|
.replace(/\s+/g, ' ')
|
|
.trim();
|
|
const decoded = decodeEntities(content).trim();
|
|
// Skip WEBVTT header leftovers and empty cues
|
|
if (!decoded || /^WEBVTT/i.test(decoded)) continue;
|
|
out.push({ t: start, dur: Math.max(0, end - start), text: decoded });
|
|
}
|
|
return capLines(dedupeTranscriptLines(out));
|
|
}
|
|
|
|
/** Bound response volume for very long transcripts (Phase 1 decision: truncate). */
|
|
export function capLines(lines, max = MAX_LINES_DEFAULT) {
|
|
const list = Array.isArray(lines) ? lines : [];
|
|
const n = Math.max(1, Number(max || MAX_LINES_DEFAULT));
|
|
return list.length > n ? list.slice(0, n) : list;
|
|
}
|
|
|
|
/**
|
|
* Parse a downloaded timedtext payload regardless of its format.
|
|
* Tries `json3`, then `vtt`, then XML captions (`srv*`/`ttml`).
|
|
* HTML error pages always yield `[]`.
|
|
*/
|
|
export function parseTrackText(text, ext = '') {
|
|
if (looksLikeHtmlError(text)) return [];
|
|
const e = String(ext || '').toLowerCase();
|
|
const asJson3 = () => {
|
|
try { return parseJson3(JSON.parse(String(text))); } catch { return []; }
|
|
};
|
|
if (e === 'json3') {
|
|
const lines = asJson3();
|
|
if (lines.length) return lines;
|
|
const vtt = parseVtt(text);
|
|
if (vtt.length) return vtt;
|
|
return parseXmlCaptions(text);
|
|
}
|
|
const vtt = parseVtt(text);
|
|
if (vtt.length) return vtt;
|
|
const j = asJson3();
|
|
if (j.length) return j;
|
|
return parseXmlCaptions(text);
|
|
}
|
|
|
|
/**
|
|
* Build auto-translated fallbacks (`&tlang=xx`) from source candidates.
|
|
* Used as LAST resort when no source track in the requested language is
|
|
* fetchable but another language is (e.g. en-US walled while fr answers):
|
|
* YouTube translates server-side, so `fr_url + &tlang=en` yields English
|
|
* lines. Bounded to 2 extra candidates. YouTube-only (timedtext URLs).
|
|
* @param {Array<{ track: any, lang: string|null }>} candidates
|
|
* @param {string} targetLang requested language
|
|
* @returns {Array<{ track: any, lang: string|null }>}
|
|
*/
|
|
export function translatedFallbacks(candidates, targetLang) {
|
|
const primary = String(targetLang || '').toLowerCase().split('-')[0];
|
|
if (!primary) return [];
|
|
const seen = new Set((candidates || []).map((c) => String(c?.track?.url || '')));
|
|
const out = [];
|
|
for (const c of candidates || []) {
|
|
const srcPrimary = String(c?.lang || '').toLowerCase().split('-')[0];
|
|
if (!srcPrimary || srcPrimary === primary) continue;
|
|
const base = String(c?.track?.url || '');
|
|
if (!base) continue;
|
|
const url = `${base}${base.includes('?') ? '&' : '?'}tlang=${encodeURIComponent(primary)}`;
|
|
if (seen.has(url)) continue;
|
|
seen.add(url);
|
|
out.push({ track: { ...c.track, url, name: `${c.track.name || srcPrimary} (auto-translated)` }, lang: primary });
|
|
if (out.length >= 2) break;
|
|
}
|
|
return out;
|
|
}
|
|
|
|
/**
|
|
* Keep the first candidate of each primary language (`en-US`+`en` -> one),
|
|
* preserving order. Guarantees every available language is tried at least
|
|
* once when the caller wants "any language" instead of failing after an
|
|
* arbitrary slice. Bounded by `max` (default 10).
|
|
* @param {Array<{ track: any, lang: string|null }>} candidates
|
|
* @param {number} [max]
|
|
* @returns {Array<{ track: any, lang: string|null }>}
|
|
*/
|
|
export function firstPerLanguage(candidates, max = 10) {
|
|
const n = Math.max(1, Number(max || 10));
|
|
const seen = new Set();
|
|
const out = [];
|
|
for (const c of candidates || []) {
|
|
const p = String(c?.lang || '').toLowerCase().split('-')[0] || 'und';
|
|
if (seen.has(p)) continue;
|
|
seen.add(p);
|
|
out.push(c);
|
|
if (out.length >= n) break;
|
|
}
|
|
return out;
|
|
}
|
|
|
|
/** Map short (`yt`) and long (`youtube`) provider ids to `providerUrlFrom()` names. */
|
|
export function normalizeTranscriptProvider(provider) {
|
|
const p = String(provider || '').trim().toLowerCase();
|
|
const map = {
|
|
yt: 'youtube', youtube: 'youtube',
|
|
dm: 'dailymotion', dailymotion: 'dailymotion',
|
|
tw: 'twitch', twitch: 'twitch',
|
|
pt: 'peertube', peertube: 'peertube',
|
|
od: 'odysee', odysee: 'odysee',
|
|
ru: 'rumble', rumble: 'rumble',
|
|
};
|
|
return map[p] || null;
|
|
}
|
|
|
|
export { trackExtOf as transcriptTrackExt, firstTrack as transcriptFirstTrack };
|