Files
NewTube/server/providers/youtube-scrape.mjs
T
bruno 4bdcd393fe feat(youtube): InnerTube-first search, transcripts and watch-next related (Steps 15-18)
- InnerTube layer via pinned youtubei.js 18.1.0 (no quota, no key):
  search with merged continuations (unlimited pages), watch-next
  related with LockupView mapping, caption-track discovery
- 3-layer dispatcher (YT_SEARCH_MODE, default innertube-first):
  innertube -> yt-dlp scrape -> official API, graceful errors.yt
- Robust yt-dlp binary resolution (YT_DLP_PATH > PATH > bundled)
  with systematic API fallback (fixes spawn ENOENT in UI)
- Transcript: InnerTube caption discovery (YT_TRANSCRIPT_SOURCE),
  reusing pickTrack/orderedTracks/parseTrackText; yt-dlp fallback kept
- Watch: sidebar uses real watch-next related[] (/api/details),
  title-search fallback for other providers
- Cache: memory LRU + SQLite (youtube_search_cache, youtube_metrics),
  never persist empty pages; /healthz observability; /api/trending
- Includes pending Step 15/16 leftovers in same files (suggest,
  test scripts); unrelated provider adapters left uncommitted
2026-09-25 19:38:46 -04:00

198 lines
8.6 KiB
JavaScript

// Recherche / channel / trending YouTube SANS clé API via yt-dlp (Step 17 P1).
// Binaire résolu via YT_DLP_PATH > PATH. Jamais de secret en log.
import { execFile as execFileCb } from 'node:child_process';
import { promisify } from 'node:util';
import { getYtDlpTimeoutMs, buildYtDlpExtraArgs, resolveYtDlpBin } from './youtube-common.mjs';
const execFileAsync = promisify(execFileCb);
/**
* Normalise une entrée yt-dlp flat-playlist vers Suggestion (même contrat que youtube.mjs).
* @param {any} e entrée yt-dlp
*/
export function mapFlatEntry(e) {
if (!e || typeof e !== 'object') return null;
const id = e.id || e.display_id || null;
if (!id) return null;
const thumbs = Array.isArray(e.thumbnails) ? e.thumbnails : [];
const thumb = thumbs.length
? (thumbs.find((t) => t?.url && (t.height || 0) >= 360)?.url || thumbs[thumbs.length - 1]?.url)
: (e.thumbnail || undefined);
const duration = e.duration != null ? Number(e.duration) : undefined;
const views = e.view_count != null ? Number(e.view_count) : (e.views != null ? Number(e.views) : undefined);
const channelId = e.channel_id || e.uploader_id || undefined;
const uploaderName = e.channel || e.uploader || undefined;
return {
title: e.title || '',
id: String(id),
url: String(id).startsWith('http') ? String(id) : `https://www.youtube.com/watch?v=${id}`,
thumbnail: thumb,
uploaderName,
type: 'video',
...(Number.isFinite(duration) && duration > 0 ? { duration } : {}),
...(Number.isFinite(views) ? { views } : {}),
...(e.timestamp ? { publishedAt: new Date(Number(e.timestamp) * 1000).toISOString() } : {}),
...(channelId ? { channelId: String(channelId), channelExternalId: String(channelId), channelUrl: `https://www.youtube.com/channel/${channelId}` } : {}),
...(uploaderName ? { channelHandle: String(uploaderName) } : {}),
};
}
/** Parse le JSON de `yt-dlp --dump-single-json --flat-playlist` (objet ou NDJSON fog). */
export function parseFlatPlaylistJson(raw) {
try {
const text = String(raw || '').trim();
if (!text) return [];
// Cas standard : un seul objet JSON avec .entries
try {
const obj = JSON.parse(text);
const entries = Array.isArray(obj?.entries) ? obj.entries : (Array.isArray(obj) ? obj : []);
return entries.map(mapFlatEntry).filter(Boolean);
} catch {
// Fallback NDJSON (une ligne = un JSON)
const out = [];
for (const line of text.split('\n')) {
const t = line.trim();
if (!t) continue;
try {
const o = JSON.parse(t);
const m = mapFlatEntry(o);
if (m) out.push(m);
} catch {}
}
return out;
}
} catch { return []; }
}
export function classifyScrapeError(e) {
if (e?.code === 'yt_scrape_no_binary') return e;
const msg = String(e?.message || e || '');
if (/ENOENT/i.test(msg) || e?.code === 'ENOENT' || e?.errno === 'ENOENT') {
return Object.assign(
new Error('yt-dlp introuvable (ni PATH ni bundled). Installez yt-dlp ou définissez YT_DLP_PATH.'),
{ ytStatus: 503, code: 'yt_scrape_no_binary' },
);
}
if (/bot|sign in to confirm|challenge|cookies|login/i.test(msg)) {
return Object.assign(new Error('YouTube bot-check (cookies/PO-Token requis)'), { ytStatus: 502, code: 'yt_scrape_bot_check' });
}
if (/timed out|timeout|ETIMEDOUT|killed|SIGKILL/i.test(msg)) {
return Object.assign(new Error('YouTube scrape timeout'), { ytStatus: 504, code: 'yt_scrape_timeout' });
}
if (/unable to|unsupported|not available|private|deleted/i.test(msg)) {
return Object.assign(new Error(`YouTube scrape upstream: ${msg.slice(0, 160)}`), { ytStatus: 502, code: 'yt_scrape_upstream' });
}
return Object.assign(new Error(`YouTube scrape failed: ${msg.slice(0, 200)}`), { ytStatus: 502, code: 'yt_scrape_failed' });
}
async function runYtDlpFlat(queryOrUrl, { limit = 10, playlistStart = 1 } = {}) {
const bin = await resolveYtDlpBin();
const timeout = getYtDlpTimeoutMs();
const end = playlistStart + Math.max(1, Math.min(50, Number(limit || 10))) - 1;
const args = [
'--dump-single-json',
'--flat-playlist',
'--no-warnings',
'--no-check-certificates',
'--skip-download',
'--no-playlist',
'--playlist-start', String(playlistStart),
'--playlist-end', String(end),
...buildYtDlpExtraArgs(),
queryOrUrl,
];
try {
const { stdout } = await execFileAsync(bin, args, { timeout, maxBuffer: 16 * 1024 * 1024 });
return parseFlatPlaylistJson(stdout);
} catch (e) {
// yt-dlp peut écrire du JSON partiel sur stdout même en exit non-zero
const partial = e?.stdout ? parseFlatPlaylistJson(String(e.stdout)) : [];
if (partial.length) return partial;
throw classifyScrapeError(e);
}
}
/** Recherche sans clé. sort: relevance|date|views */
export async function searchViaScrape(q, opts = {}) {
const query = String(q || '').trim();
if (query.length < 2) return [];
const limit = Math.min(50, Math.max(1, Number(opts?.limit || 10)));
const page = Math.max(1, Number(opts?.page || 1));
const sort = String(opts?.sort || 'relevance').toLowerCase();
const perPage = limit;
const start = (page - 1) * perPage + 1;
let prefix = 'ytsearch';
if (sort === 'date') prefix = 'ytsearchdate';
// views : pas de préfixe natif stable -> ytsearch + tri local
const target = `${prefix}${perPage}:${query}`;
let items = await runYtDlpFlat(target, { limit: perPage, playlistStart: start });
if (sort === 'views') items = [...items].sort((a, b) => (b.views || 0) - (a.views || 0));
return items.slice(0, perPage);
}
const CHANNEL_TABS = { videos: 'videos', shorts: 'shorts', streams: 'streams', live: 'streams', playlists: 'playlists' };
/** Contenu chaîne sans clé. type: videos|shorts|playlists|live */
export async function fetchChannelViaScrape(externalId, { type = 'videos', page = 1, limit = 24 } = {}) {
const raw = String(externalId || '').trim().replace(/^@/, '');
if (!raw) return { items: [], nextPage: null };
const tab = CHANNEL_TABS[String(type)] || 'videos';
const perPage = Math.min(50, Math.max(1, Number(limit || 24)));
const pageNum = Math.max(1, Number(page || 1));
const start = (pageNum - 1) * perPage + 1;
// UC... -> /channel/UC.../tab, sinon -> /@/handle/tab
const base = /^UC[\w-]{20,}$/.test(raw)
? `https://www.youtube.com/channel/${raw}/${tab}`
: `https://www.youtube.com/@${raw}/${tab}`;
const items = await runYtDlpFlat(base, { limit: perPage, playlistStart: start });
// playlists renvoient des ids PL... : on les garde tels quels avec un shape playlist
if (tab === 'playlists') {
const mapped = items.map((s) => ({ id: s.id, title: s.title, thumbnail: s.thumbnail, videoCount: null, updatedAt: s.publishedAt || null }));
return { items: mapped, nextPage: items.length >= perPage ? pageNum + 1 : null };
}
let filtered = items;
if (type === 'shorts') filtered = items.filter((s) => !s.duration || s.duration <= 70);
return { items: filtered, nextPage: items.length >= perPage ? pageNum + 1 : null };
}
/** Trending sans clé. Les onglets /feed/trending ont été retirés côté YouTube
* (redirect home -> erreur tab) ; on tente les tabs puis repli ytsearch trié par vues. */
export async function getTrendingViaScrape(limit = 24) {
const n = Math.min(50, Math.max(1, Number(limit || 24)));
const tabs = [
'https://www.youtube.com/feed/trending',
'https://www.youtube.com/trending',
];
for (const url of tabs) {
try {
const items = await runYtDlpFlat(url, { limit: n, playlistStart: 1 });
if (items.length) return items;
} catch {}
}
// Repli : recherche générique triée par vues (0 quota, toujours disponible)
return runYtDlpFlat(`ytsearch${n}:top trending videos world`, { limit: n, playlistStart: 1 });
}
/** Résout un channel_id via scrape (imprime channel_id sans appel API). */
export async function resolveChannelIdViaScrape(externalId) {
const raw = String(externalId || '').trim();
if (/^UC[\w-]{20,}$/.test(raw)) return raw;
const handle = raw.replace(/^@/, '');
const bin = await resolveYtDlpBin().catch(() => null);
if (!bin) {
throw Object.assign(
new Error('yt-dlp introuvable (ni PATH ni bundled). Installez yt-dlp ou définissez YT_DLP_PATH.'),
{ ytStatus: 503, code: 'yt_scrape_no_binary' },
);
}
const timeout = Math.min(15000, getYtDlpTimeoutMs());
try {
const { stdout } = await execFileAsync(bin,
['--print', '%(channel_id)s', '--no-warnings', '--skip-download', '--playlist-end', '1', ...buildYtDlpExtraArgs(), `https://www.youtube.com/@${handle}/videos`],
{ timeout });
const id = String(stdout || '').trim().split('\n')[0].trim();
if (/^UC[\w-]{20,}$/.test(id)) return id;
} catch {}
return raw;
}