feat(youtube): InnerTube-first search, transcripts and watch-next related (Steps 15-18)
- InnerTube layer via pinned youtubei.js 18.1.0 (no quota, no key): search with merged continuations (unlimited pages), watch-next related with LockupView mapping, caption-track discovery - 3-layer dispatcher (YT_SEARCH_MODE, default innertube-first): innertube -> yt-dlp scrape -> official API, graceful errors.yt - Robust yt-dlp binary resolution (YT_DLP_PATH > PATH > bundled) with systematic API fallback (fixes spawn ENOENT in UI) - Transcript: InnerTube caption discovery (YT_TRANSCRIPT_SOURCE), reusing pickTrack/orderedTracks/parseTrackText; yt-dlp fallback kept - Watch: sidebar uses real watch-next related[] (/api/details), title-search fallback for other providers - Cache: memory LRU + SQLite (youtube_search_cache, youtube_metrics), never persist empty pages; /healthz observability; /api/trending - Includes pending Step 15/16 leftovers in same files (suggest, test scripts); unrelated provider adapters left uncommitted
This commit is contained in:
@@ -0,0 +1,309 @@
|
||||
// Step 16 — Pure transcript helpers (no network, no DB).
|
||||
// Data source: `yt-dlp --dump-single-json --skip-download` exposes
|
||||
// `subtitles` (manual) and `automatic_captions` (auto-generated).
|
||||
// Track entries are arrays of { url, ext, name } (or single objects).
|
||||
|
||||
/**
|
||||
* @typedef {{ t: number, dur: number, text: string }} TranscriptLine
|
||||
*/
|
||||
|
||||
const MAX_LINES_DEFAULT = Number(process.env.TRANSCRIPT_MAX_LINES || 5000);
|
||||
|
||||
/** Normalize a language code for comparison (lowercase, `_` -> `-`). */
|
||||
function normLang(code) {
|
||||
return String(code || '').trim().toLowerCase().replace(/_/g, '-');
|
||||
}
|
||||
|
||||
/** Pick the first usable track object from an array-or-single entry. */
|
||||
function firstTrack(entry) {
|
||||
if (!entry) return null;
|
||||
const list = Array.isArray(entry) ? entry : [entry];
|
||||
return list.find((t) => t && typeof t.url === 'string' && t.url) || null;
|
||||
}
|
||||
|
||||
function trackExtOf(track) {
|
||||
const ext = String(track?.ext || '').toLowerCase();
|
||||
if (ext) return ext;
|
||||
try {
|
||||
const u = new URL(String(track?.url || ''));
|
||||
const m = /\.([a-z0-9]+)(?:[?#]|$)/i.exec(u.pathname || '');
|
||||
if (m) return m[1].toLowerCase();
|
||||
} catch {}
|
||||
return '';
|
||||
}
|
||||
|
||||
/** Find a language key in `dict` matching `code` exactly or by prefix (`fr` -> `fr-*`). */
|
||||
function findLangKey(dict, code) {
|
||||
const want = normLang(code);
|
||||
if (!want || !dict) return null;
|
||||
const keys = Object.keys(dict);
|
||||
const exact = keys.find((k) => normLang(k) === want);
|
||||
if (exact) return exact;
|
||||
// `fr-ca` requested, `fr` available (and vice versa): match on primary subtag
|
||||
const primary = want.split('-')[0];
|
||||
return keys.find((k) => normLang(k).split('-')[0] === primary) || null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Select the best subtitle track.
|
||||
* Priority: manual exact > manual prefix/primary > auto exact > auto prefix/primary > first available.
|
||||
* @param {any} json yt-dlp dump-single-json
|
||||
* @param {string} [lang]
|
||||
* @returns {{ track: any|null, languages: string[], lang: string|null }}
|
||||
*/
|
||||
export function pickTrack(json, lang = 'fr') {
|
||||
const manual = (json && json.subtitles) || {};
|
||||
const auto = (json && json.automatic_captions) || {};
|
||||
const manualKeys = Object.keys(manual);
|
||||
const autoKeys = Object.keys(auto);
|
||||
const languages = Array.from(new Set([...manualKeys, ...autoKeys]));
|
||||
|
||||
if (languages.length === 0) return { track: null, languages: [], lang: null };
|
||||
|
||||
const manualKey = findLangKey(manual, lang);
|
||||
if (manualKey) {
|
||||
const track = firstTrack(manual[manualKey]);
|
||||
if (track) return { track, languages, lang: manualKey };
|
||||
}
|
||||
const autoKey = findLangKey(auto, lang);
|
||||
if (autoKey) {
|
||||
const track = firstTrack(auto[autoKey]);
|
||||
if (track) return { track, languages, lang: autoKey };
|
||||
}
|
||||
// Fallback: first available track (manual preferred)
|
||||
for (const key of manualKeys) {
|
||||
const track = firstTrack(manual[key]);
|
||||
if (track) return { track, languages, lang: key };
|
||||
}
|
||||
for (const key of autoKeys) {
|
||||
const track = firstTrack(auto[key]);
|
||||
if (track) return { track, languages, lang: key };
|
||||
}
|
||||
return { track: null, languages, lang: null };
|
||||
}
|
||||
|
||||
/**
|
||||
* Order candidate tracks to try when the preferred one cannot be fetched.
|
||||
* YouTube auto-translated tracks (`tlang=xx`) are often rate-limited (HTTP 429
|
||||
* or "Sorry" HTML pages) while the original language still works: always try
|
||||
* the requested language first, then the original (`en`), then everything
|
||||
* else (manual preferred). De-duplicated by URL.
|
||||
* @param {any} json yt-dlp dump-single-json
|
||||
* @param {string} [lang]
|
||||
* @returns {Array<{ track: any, lang: string|null }>}
|
||||
*/
|
||||
export function orderedTracks(json, lang = 'fr') {
|
||||
const manual = (json && json.subtitles) || {};
|
||||
const auto = (json && json.automatic_captions) || {};
|
||||
const seen = new Set();
|
||||
const out = [];
|
||||
const push = (key, dict) => {
|
||||
if (key == null) return;
|
||||
const t = firstTrack(dict[key]);
|
||||
if (!t || !t.url || seen.has(String(t.url))) return;
|
||||
seen.add(String(t.url));
|
||||
out.push({ track: t, lang: key });
|
||||
};
|
||||
const primary = pickTrack(json, lang);
|
||||
if (primary.track) {
|
||||
seen.add(String(primary.track.url));
|
||||
out.push({ track: primary.track, lang: primary.lang });
|
||||
}
|
||||
// Original language first fallback (avoids translated-track rate limits)
|
||||
const reqPrimary = String(lang || '').toLowerCase().split('-')[0];
|
||||
for (const fallbackLang of ['en', 'fr']) {
|
||||
if (fallbackLang === reqPrimary) continue;
|
||||
const k = findLangKey(manual, fallbackLang);
|
||||
if (k) push(k, manual);
|
||||
const ka = findLangKey(auto, fallbackLang);
|
||||
if (ka) push(ka, auto);
|
||||
}
|
||||
for (const key of Object.keys(manual)) push(key, manual);
|
||||
for (const key of Object.keys(auto)) push(key, auto);
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a YouTube `json3` timedtext payload into normalized lines.
|
||||
* @param {any} data parsed JSON
|
||||
* @returns {TranscriptLine[]}
|
||||
*/
|
||||
export function parseJson3(data) {
|
||||
const events = (data && data.events) || [];
|
||||
if (!Array.isArray(events)) return [];
|
||||
const out = [];
|
||||
for (const e of events) {
|
||||
if (!e || !Array.isArray(e.segs)) continue;
|
||||
const text = e.segs.map((s) => String(s?.utf8 ?? '')).join('').replace(/\n+/g, ' ').trim();
|
||||
if (!text) continue;
|
||||
out.push({
|
||||
t: Number(e.tStartMs || 0) / 1000,
|
||||
dur: Number(e.dDurationMs || 0) / 1000,
|
||||
text,
|
||||
});
|
||||
}
|
||||
return capLines(out);
|
||||
}
|
||||
|
||||
/** Decode a handful of HTML entities found in VTT payloads. */
|
||||
function decodeEntities(s) {
|
||||
return String(s || '')
|
||||
.replace(/&/g, '&')
|
||||
.replace(/</g, '<')
|
||||
.replace(/>/g, '>')
|
||||
.replace(/"/g, '"')
|
||||
.replace(/'/g, "'")
|
||||
.replace(/ /g, ' ');
|
||||
}
|
||||
|
||||
/** Strip XML/HTML tags and decode entities to plain cue text. */
|
||||
function xmlTextToPlain(s) {
|
||||
return decodeEntities(String(s || '').replace(/<[^>]*>/g, ' ').replace(/\s+/g, ' ').trim()).trim();
|
||||
}
|
||||
|
||||
function parseTimeAttrToSeconds(v) {
|
||||
if (v == null || v === '') return null;
|
||||
const s = String(v).trim();
|
||||
if (!s) return null;
|
||||
if (/^\d+(\.\d+)?$/.test(s)) return Number(s);
|
||||
const m = /(?:(\d+):)?([0-5]?\d):([0-5]\d)(?:\.(\d{1,3}))?/.exec(s);
|
||||
if (!m) return null;
|
||||
return Number(m[1] || 0) * 3600 + Number(m[2]) * 60 + Number(m[3]) + Number((m[4] || '0').padEnd(3, '0')) / 1000;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse YouTube XML caption formats (`srv1`/`srv2`/`srv3` `<p>` cues,
|
||||
* `ttml` `<p begin= end=>` cues) into normalized lines.
|
||||
* @param {string} text raw XML
|
||||
* @returns {TranscriptLine[]}
|
||||
*/
|
||||
export function parseXmlCaptions(text) {
|
||||
const out = [];
|
||||
const src = String(text || '');
|
||||
if (!src || !/<(p|text|tt|transcript)\b/i.test(src)) return capLines(out);
|
||||
// YouTube srv*: <p t="0" d="2500">Hello <s>world</s></p>
|
||||
const srvRe = /<p\b[^>]*?(?:t|start)="([^"]*)"[^>]*?(?:d|dur)="([^"]*)"[^>]*>([\s\S]*?)<\/p\s*>/gi;
|
||||
// Generic ttml: <p begin="00:00:00.000" end="00:00:02.500">...</p>
|
||||
const ttmlRe = /<p\b[^>]*?begin="([^"]*)"[^>]*?end="([^"]*)"[^>]*>([\s\S]*?)<\/p\s*>/gi;
|
||||
let m;
|
||||
while ((m = srvRe.exec(src)) !== null) {
|
||||
const start = Number(m[1]) / 1000;
|
||||
const dur = Number(m[2]) / 1000;
|
||||
const cleaned = xmlTextToPlain(m[3]);
|
||||
if (!cleaned || !Number.isFinite(start)) continue;
|
||||
out.push({ t: start, dur: Number.isFinite(dur) && dur > 0 ? dur : 0, text: cleaned });
|
||||
}
|
||||
if (out.length === 0) {
|
||||
while ((m = ttmlRe.exec(src)) !== null) {
|
||||
const start = parseTimeAttrToSeconds(m[1]);
|
||||
const end = parseTimeAttrToSeconds(m[2]);
|
||||
const cleaned = xmlTextToPlain(m[3]);
|
||||
if (!cleaned || start == null || end == null) continue;
|
||||
out.push({ t: start, dur: Math.max(0, end - start), text: cleaned });
|
||||
}
|
||||
}
|
||||
// Plain <text start="2.5" dur="2.0">Fallback</text> variant
|
||||
if (out.length === 0) {
|
||||
const textRe = /<text\b[^>]*?start="([^"]*)"[^>]*?(?:dur="([^"]*)")?[^>]*>([\s\S]*?)<\/text\s*>/gi;
|
||||
while ((m = textRe.exec(src)) !== null) {
|
||||
const start = Number(m[1]);
|
||||
const dur = Number(m[2] || 0);
|
||||
const cleaned = xmlTextToPlain(m[3]);
|
||||
if (!cleaned || !Number.isFinite(start)) continue;
|
||||
out.push({ t: start, dur: dur > 0 ? dur : 0, text: cleaned });
|
||||
}
|
||||
}
|
||||
return capLines(out);
|
||||
}
|
||||
|
||||
/** Detect YouTube "Sorry / automated queries" style HTML error pages. */
|
||||
export function looksLikeHtmlError(text) {
|
||||
const s = String(text || '').trimStart().slice(0, 512).toLowerCase();
|
||||
return s.startsWith('<!doctype') || s.startsWith('<html');
|
||||
}
|
||||
|
||||
function vttTimestampToSeconds(h, m, s, ms) {
|
||||
return Number(h) * 3600 + Number(m) * 60 + Number(s) + Number(ms) / 1000;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a WebVTT payload into normalized lines.
|
||||
* Handles cue identifiers, multi-line cues and inline tags (`<c>`, `<i>`, timestamps).
|
||||
* @param {string} text raw VTT
|
||||
* @returns {TranscriptLine[]}
|
||||
*/
|
||||
export function parseVtt(text) {
|
||||
if (looksLikeHtmlError(text)) return [];
|
||||
const out = [];
|
||||
const blocks = String(text || '').replace(/\r\n/g, '\n').split(/\n{2,}/);
|
||||
const tsRe = /(?:(\d{2,}):)?([0-5]?\d):([0-5]\d)\.(\d{3})\s*-->\s*(?:(\d{2,}):)?([0-5]?\d):([0-5]\d)\.(\d{3})/;
|
||||
for (const block of blocks) {
|
||||
const lines = String(block || '').split('\n');
|
||||
const timingIdx = lines.findIndex((l) => l.includes('-->'));
|
||||
if (timingIdx === -1) continue;
|
||||
const m = tsRe.exec(lines[timingIdx]);
|
||||
if (!m) continue;
|
||||
const start = vttTimestampToSeconds(m[1] || '0', m[2], m[3], m[4]);
|
||||
const end = vttTimestampToSeconds(m[5] || '0', m[6], m[7], m[8]);
|
||||
const content = lines
|
||||
.slice(timingIdx + 1)
|
||||
.join(' ')
|
||||
// strip inline VTT tags (<c.colorE5E5E5>, <i>, <00:00:01.000>, ...)
|
||||
.replace(/<[^>]*>/g, ' ')
|
||||
.replace(/\s+/g, ' ')
|
||||
.trim();
|
||||
const decoded = decodeEntities(content).trim();
|
||||
// Skip WEBVTT header leftovers and empty cues
|
||||
if (!decoded || /^WEBVTT/i.test(decoded)) continue;
|
||||
out.push({ t: start, dur: Math.max(0, end - start), text: decoded });
|
||||
}
|
||||
return capLines(out);
|
||||
}
|
||||
|
||||
/** Bound response volume for very long transcripts (Phase 1 decision: truncate). */
|
||||
export function capLines(lines, max = MAX_LINES_DEFAULT) {
|
||||
const list = Array.isArray(lines) ? lines : [];
|
||||
const n = Math.max(1, Number(max || MAX_LINES_DEFAULT));
|
||||
return list.length > n ? list.slice(0, n) : list;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a downloaded timedtext payload regardless of its format.
|
||||
* Tries `json3`, then `vtt`, then XML captions (`srv*`/`ttml`).
|
||||
* HTML error pages always yield `[]`.
|
||||
*/
|
||||
export function parseTrackText(text, ext = '') {
|
||||
if (looksLikeHtmlError(text)) return [];
|
||||
const e = String(ext || '').toLowerCase();
|
||||
const asJson3 = () => {
|
||||
try { return parseJson3(JSON.parse(String(text))); } catch { return []; }
|
||||
};
|
||||
if (e === 'json3') {
|
||||
const lines = asJson3();
|
||||
if (lines.length) return lines;
|
||||
const vtt = parseVtt(text);
|
||||
if (vtt.length) return vtt;
|
||||
return parseXmlCaptions(text);
|
||||
}
|
||||
const vtt = parseVtt(text);
|
||||
if (vtt.length) return vtt;
|
||||
const j = asJson3();
|
||||
if (j.length) return j;
|
||||
return parseXmlCaptions(text);
|
||||
}
|
||||
|
||||
/** Map short (`yt`) and long (`youtube`) provider ids to `providerUrlFrom()` names. */
|
||||
export function normalizeTranscriptProvider(provider) {
|
||||
const p = String(provider || '').trim().toLowerCase();
|
||||
const map = {
|
||||
yt: 'youtube', youtube: 'youtube',
|
||||
dm: 'dailymotion', dailymotion: 'dailymotion',
|
||||
tw: 'twitch', twitch: 'twitch',
|
||||
pt: 'peertube', peertube: 'peertube',
|
||||
od: 'odysee', odysee: 'odysee',
|
||||
ru: 'rumble', rumble: 'rumble',
|
||||
};
|
||||
return map[p] || null;
|
||||
}
|
||||
|
||||
export { trackExtOf as transcriptTrackExt, firstTrack as transcriptFirstTrack };
|
||||
Reference in New Issue
Block a user