Files
NewTube/server/providers/channel-content.mjs
T
bruno 4bdcd393fe feat(youtube): InnerTube-first search, transcripts and watch-next related (Steps 15-18)
- InnerTube layer via pinned youtubei.js 18.1.0 (no quota, no key):
  search with merged continuations (unlimited pages), watch-next
  related with LockupView mapping, caption-track discovery
- 3-layer dispatcher (YT_SEARCH_MODE, default innertube-first):
  innertube -> yt-dlp scrape -> official API, graceful errors.yt
- Robust yt-dlp binary resolution (YT_DLP_PATH > PATH > bundled)
  with systematic API fallback (fixes spawn ENOENT in UI)
- Transcript: InnerTube caption discovery (YT_TRANSCRIPT_SOURCE),
  reusing pickTrack/orderedTracks/parseTrackText; yt-dlp fallback kept
- Watch: sidebar uses real watch-next related[] (/api/details),
  title-search fallback for other providers
- Cache: memory LRU + SQLite (youtube_search_cache, youtube_metrics),
  never persist empty pages; /healthz observability; /api/trending
- Includes pending Step 15/16 leftovers in same files (suggest,
  test scripts); unrelated provider adapters left uncommitted
2026-09-25 19:38:46 -04:00

390 lines
19 KiB
JavaScript

// Contenu d'une chaîne par provider : videos | shorts | playlists | live
// Chaque fonction retourne { items: Suggestion[], nextPage: number|null, total }
// Suggestion suit le format de /api/search (id,title,thumbnail,url,uploaderName,
// duration,views,publishedAt,channelId,channelExternalId...).
const TIMEOUT_MS = Number(process.env.CHANNEL_CONTENT_TIMEOUT_MS || 9000);
function fetchWithTimeout(url, options = {}, timeoutMs = TIMEOUT_MS) {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
const { timeout, ...init } = options || {};
return fetch(url, { ...init, signal: controller.signal }).finally(() => clearTimeout(timer));
}
async function readJson(resp) {
if (!resp.ok) throw new Error(`upstream_${resp.status}`);
return resp.json();
}
function ytKeys() {
const keys = [];
try {
const raw = process.env.YOUTUBE_API_KEYS;
if (raw && String(raw).trim()) {
const s = String(raw).trim();
if (s.startsWith('[')) {
try {
const arr = JSON.parse(s);
if (Array.isArray(arr)) keys.push(...arr.map((v) => String(v || '').trim()).filter(Boolean));
} catch {}
} else keys.push(...s.split(',').map((v) => String(v || '').trim()).filter(Boolean));
}
} catch {}
if (process.env.YOUTUBE_API_KEY) keys.push(String(process.env.YOUTUBE_API_KEY).trim());
return [...new Set(keys.filter(Boolean))];
}
async function ytGet(path, params) {
const keys = ytKeys();
if (!keys.length) throw Object.assign(new Error('youtube_api_key_unavailable'), { status: 503 });
let lastErr = null;
for (const key of keys) {
const qs = new URLSearchParams({ ...params, key });
try {
const resp = await fetchWithTimeout(`https://www.googleapis.com/youtube/v3/${path}?${qs.toString()}`);
const data = await resp.json().catch(() => ({}));
if (resp.ok) return data;
lastErr = new Error(`youtube_${resp.status}`);
const reason = data?.error?.errors?.[0]?.reason || '';
if (!/quota|rateLimit|API_KEY_INVALID|expired/i.test(`${reason} ${data?.error?.message || ''}`)) break;
} catch (e) { lastErr = e; break; }
}
throw lastErr || new Error('youtube_failed');
}
function parseISODuration(iso) {
const m = String(iso || '').match(/PT(?:(\d+)H)?(?:(\d+)M)?(?:(\d+)S)?/);
if (!m) return 0;
return Number(m[1] || 0) * 3600 + Number(m[2] || 0) * 60 + Number(m[3] || 0);
}
async function resolveYouTubeChannelId(externalId) {
const raw = String(externalId || '').trim();
if (/^UC[\w-]{20,}$/.test(raw)) return raw;
const handle = raw.replace(/^@/, '');
// 0) scrape sans clé (Step 17) : ne consomme aucun quota
try {
const { getSearchMode } = await import('./youtube-common.mjs');
const mode = getSearchMode();
if (mode !== 'api-only') {
const { resolveChannelIdViaScrape } = await import('./youtube-scrape.mjs');
const id = await resolveChannelIdViaScrape(raw);
if (id && /^UC[\w-]{20,}$/.test(id)) return id;
}
} catch {}
// 1) channels?forHandle (fonctionne encore pour beaucoup de chaînes)
try {
const data = await ytGet('channels', { part: 'id', forHandle: handle });
const id = data?.items?.[0]?.id;
if (id) return id;
} catch {}
// 2) search type=channel
try {
const data = await ytGet('search', { part: 'snippet', q: handle, type: 'channel', maxResults: '1' });
const id = data?.items?.[0]?.id?.channelId;
if (id) return id;
} catch {}
return raw;
}
async function ytVideoDetails(videoIds) {
const map = new Map();
if (!videoIds.length) return map;
try {
const data = await ytGet('videos', { part: 'contentDetails,statistics,status', id: videoIds.join(',') });
for (const v of data?.items || []) if (v?.id) map.set(v.id, v);
} catch {}
return map;
}
function ytSuggestion(item, details, channelId) {
const videoId = item?.id?.videoId || item?.id;
const sn = item?.snippet || {};
const d = videoId ? details.get(videoId) : null;
const secs = parseISODuration(d?.contentDetails?.duration);
return {
id: videoId,
title: sn.title || '',
url: videoId ? `https://www.youtube.com/watch?v=${videoId}` : undefined,
thumbnail: sn.thumbnails?.high?.url || sn.thumbnails?.medium?.url || sn.thumbnails?.default?.url,
uploaderName: sn.channelTitle,
type: 'video',
duration: secs > 0 ? secs : undefined,
views: d?.statistics?.viewCount != null ? Number(d.statistics.viewCount) : undefined,
publishedAt: sn.publishedAt,
channelId,
channelExternalId: channelId,
channelUrl: channelId ? `https://www.youtube.com/channel/${channelId}` : undefined,
};
}
// L'API YouTube pagine avec des pageTokens opaques, pas des numéros de page.
// Cache process-wide des tokens : clé requête -> tokens[page] (tokens[1] = token
// pour charger la page 2). Sans token connu pour page > 1, on s'arrête (nextPage null)
// au lieu de re-servir la page 1 en boucle.
const ytTokenCache = new Map();
function ytTokenStore(key, page, nextToken) {
if (!nextToken) return;
let arr = ytTokenCache.get(key);
if (!arr) {
if (ytTokenCache.size > 500) ytTokenCache.clear();
arr = [];
ytTokenCache.set(key, arr);
}
arr[page] = nextToken;
}
function ytTokenFor(key, page) {
if (page <= 1) return '';
const arr = ytTokenCache.get(key);
return arr ? arr[page - 1] : undefined;
}
async function ytContent(externalId, { type, page, limit, sort, q }) {
const perPage = Math.min(Math.max(1, Number(limit || 24)), 50);
const pageNum = Math.max(1, Number(page || 1));
// Step 17 : scrape-first sans clé (0 quota). Fallback API si bot-check/timeout.
try {
const { getSearchMode } = await import('./youtube-common.mjs');
const mode = getSearchMode();
if (mode !== 'api-only') {
const { fetchChannelViaScrape } = await import('./youtube-scrape.mjs');
// externalId brut (handle ou UC...) : le scrape gère les deux formes
const scraped = await fetchChannelViaScrape(externalId, { type, page: pageNum, limit: perPage });
if (Array.isArray(scraped?.items) && scraped.items.length) {
if (mode === 'scrape-only') return { ...scraped, total: null };
// scrape-first : retour direct si non vide
return { ...scraped, total: null };
}
// vide -> on tente l'API (chaîne à faible volume ou tab non supporté en scrape)
if (mode === 'scrape-only') return { items: [], nextPage: null };
}
} catch (e) {
console.warn('[channel-content/yt] scrape failed, fallback api:', e?.code || e?.message || e);
try {
const { getSearchMode } = await import('./youtube-common.mjs');
if (getSearchMode() === 'scrape-only') return { items: [], nextPage: null };
} catch {}
}
const channelId = await resolveYouTubeChannelId(externalId);
if (type === 'playlists') {
const key = ['pl', channelId, perPage].join('|');
const token = ytTokenFor(key, pageNum);
if (token === undefined) return { items: [], nextPage: null };
const params = { part: 'snippet,contentDetails', channelId, maxResults: String(perPage) };
// Ne jamais envoyer pageToken=undefined (sérialisé en "undefined" -> 400).
if (token) params.pageToken = token;
let data;
try {
data = await ytGet('playlists', params);
} catch (e) {
console.warn('[channel-content/yt] playlists failed:', e?.message || e);
return { items: [], nextPage: null };
}
ytTokenStore(key, pageNum, data?.nextPageToken);
const items = (data?.items || []).map((pl) => ({
id: pl?.id,
title: pl?.snippet?.title || '',
thumbnail: pl?.snippet?.thumbnails?.high?.url || pl?.snippet?.thumbnails?.medium?.url,
videoCount: typeof pl?.contentDetails?.itemCount === 'number' ? pl.contentDetails.itemCount : null,
updatedAt: pl?.snippet?.publishedAt || null,
}));
return { items, nextPage: data?.nextPageToken ? pageNum + 1 : null, total: data?.pageInfo?.totalResults ?? null };
}
const order = sort === 'popular' ? 'viewCount' : sort === 'recent' ? 'date' : 'relevance';
const key = ['search', channelId, type, order, q || '', perPage].join('|');
const token = ytTokenFor(key, pageNum);
if (token === undefined) return { items: [], nextPage: null };
const params = {
part: 'snippet', channelId, type: 'video', maxResults: String(perPage), order,
videoEmbeddable: 'true', safeSearch: 'moderate',
};
if (token) params.pageToken = token;
if (type === 'live') params.eventType = 'live';
if (type === 'shorts') params.videoDuration = 'short';
if (q) params.q = q;
const data = await ytGet('search', params);
ytTokenStore(key, pageNum, data?.nextPageToken);
const ids = (data?.items || []).map((i) => i?.id?.videoId).filter(Boolean);
const details = await ytVideoDetails(ids);
let items = (data?.items || [])
.filter((i) => i?.id?.videoId)
.map((i) => ytSuggestion(i, details, channelId));
if (type === 'shorts') items = items.filter((s) => !s.duration || s.duration <= 70);
if (sort === 'popular') items = [...items].sort((a, b) => (b.views || 0) - (a.views || 0));
return { items, nextPage: data?.nextPageToken ? pageNum + 1 : null, total: data?.pageInfo?.totalResults ?? null };
}
// ---- Dailymotion ----
async function dmContent(externalId, { type, page, limit, sort, q }) {
const user = String(externalId || '').replace(/^@/, '');
const perPage = Math.min(Math.max(1, Number(limit || 24)), 100);
if (type === 'playlists') {
const qs = new URLSearchParams({ fields: 'id,name,thumbnail_url,videos_total', limit: String(perPage), page: String(page || 1) });
const data = await readJson(await fetchWithTimeout(`https://api.dailymotion.com/user/${encodeURIComponent(user)}/playlists?${qs}`));
return {
items: (data?.list || []).map((p) => ({ id: p.id, title: p.name, thumbnail: p.thumbnail_url, videoCount: p.videos_total ?? null })),
nextPage: data?.has_more ? (page || 1) + 1 : null,
total: data?.total ?? null,
};
}
if (type === 'live' || type === 'shorts') return { items: [], nextPage: null };
const qs = new URLSearchParams({
fields: 'id,title,thumbnail_720_url,thumbnail_480_url,duration,views_total,owner.id,owner.screenname,created_time',
limit: String(perPage), page: String(page || 1),
sort: sort === 'popular' ? 'visited' : 'recent',
});
if (q) qs.set('search', q);
const data = await readJson(await fetchWithTimeout(`https://api.dailymotion.com/user/${encodeURIComponent(user)}/videos?${qs}`));
let items = (data?.list || []).map((v) => ({
id: v.id, title: v.title, thumbnail: v.thumbnail_720_url || v.thumbnail_480_url,
url: `https://www.dailymotion.com/video/${v.id}`, uploaderName: v['owner.screenname'],
channelId: v['owner.id'], channelExternalId: v['owner.id'] || user,
duration: Number(v.duration || 0), views: Number(v.views_total || 0),
publishedAt: v.created_time ? new Date(v.created_time * 1000).toISOString() : undefined, type: 'video',
}));
if (q) { const n = q.toLowerCase(); items = items.filter((i) => i.title.toLowerCase().includes(n)); }
return { items, nextPage: data?.has_more ? (page || 1) + 1 : null, total: data?.total ?? null };
}
// ---- Twitch ----
async function twToken() {
const id = process.env.TWITCH_CLIENT_ID;
const secret = process.env.TWITCH_CLIENT_SECRET;
if (!id || !secret) return null;
const resp = await fetchWithTimeout('https://id.twitch.tv/oauth2/token', {
method: 'POST', headers: { 'Content-Type': 'application/x-www-form-urlencoded' },
body: new URLSearchParams({ client_id: id, client_secret: secret, grant_type: 'client_credentials' }),
});
if (!resp.ok) return null;
const data = await resp.json().catch(() => ({}));
return data?.access_token || null;
}
async function twUser(externalId, headers) {
const byId = /^\d+$/.test(String(externalId || ''));
const data = await readJson(await fetchWithTimeout(
`https://api.twitch.tv/helix/users?${byId ? 'id' : 'login'}=${encodeURIComponent(String(externalId).replace(/^@/, '').toLowerCase())}`,
{ headers },
));
return data?.data?.[0] || null;
}
async function twContent(externalId, { type, page, limit, sort, q }) {
const clientId = process.env.TWITCH_CLIENT_ID;
const token = await twToken();
if (!clientId || !token) return { items: [], nextPage: null };
const headers = { 'Client-ID': clientId, Authorization: `Bearer ${token}` };
const user = await twUser(externalId, headers).catch(() => null);
if (!user) return { items: [], nextPage: null };
const perPage = Math.min(Math.max(1, Number(limit || 24)), 100);
if (type === 'live') {
const data = await readJson(await fetchWithTimeout(
`https://api.twitch.tv/helix/streams?user_login=${encodeURIComponent(user.login)}&first=1`, { headers },
)).catch(() => ({ data: [] }));
const items = (data?.data || []).map((s) => ({
id: s.id, title: s.title, thumbnail: String(s.thumbnail_url || '').replace('{width}', '1280').replace('{height}', '720'),
url: `https://www.twitch.tv/${user.login}`, uploaderName: user.display_name,
channelId: user.id, channelExternalId: user.id, views: Number(s.viewer_count || 0),
publishedAt: s.started_at, type: 'video', kind: 'vod',
}));
return { items, nextPage: null };
}
if (type === 'playlists' || type === 'shorts') return { items: [], nextPage: null };
const qs = new URLSearchParams({ user_id: user.id, first: String(perPage), type: 'archive', sort: sort === 'popular' ? 'views' : 'time' });
const data = await readJson(await fetchWithTimeout(`https://api.twitch.tv/helix/videos?${qs}`, { headers }));
let items = (data?.data || []).map((v) => ({
id: v.id, title: v.title, thumbnail: v.thumbnail_url, url: v.url,
uploaderName: user.display_name, channelId: user.id, channelExternalId: user.id,
duration: undefined, views: Number(v.view_count || 0), publishedAt: v.created_at, type: 'video', kind: 'vod',
}));
if (q) { const n = q.toLowerCase(); items = items.filter((i) => i.title.toLowerCase().includes(n)); }
const cursor = data?.pagination?.cursor;
return { items, nextPage: cursor ? (page || 1) + 1 : null };
}
// ---- PeerTube (externalId = instance|channel) ----
async function ptContent(externalId, { type, page, limit, sort, q }) {
const [instance, channel] = String(externalId || '').split('|');
if (!instance || !channel) return { items: [], nextPage: null };
const perPage = Math.min(Math.max(1, Number(limit || 24)), 100);
const start = ((Math.max(1, Number(page || 1))) - 1) * perPage;
if (type === 'playlists') {
const qs = new URLSearchParams({ start: String(start), count: String(perPage), sort: '-updatedAt' });
const data = await readJson(await fetchWithTimeout(`https://${instance}/api/v1/video-channels/${encodeURIComponent(channel)}/video-playlists?${qs}`));
return {
items: (data?.data || []).map((p) => ({
id: String(p.uuid || p.id), title: p.displayName || p.name,
thumbnail: p?.thumbnailPath ? `https://${instance}${p.thumbnailPath}` : null,
videoCount: typeof p.videosLength === 'number' ? p.videosLength : null, updatedAt: p.updatedAt || null,
})),
nextPage: (data?.data || []).length >= perPage ? (page || 1) + 1 : null, total: data?.total ?? null,
};
}
if (type === 'live' || type === 'shorts') return { items: [], nextPage: null };
const qs = new URLSearchParams({ start: String(start), count: String(perPage), sort: sort === 'popular' ? '-views' : '-publishedAt' });
if (q) qs.set('search', q);
const data = await readJson(await fetchWithTimeout(`https://${instance}/api/v1/video-channels/${encodeURIComponent(channel)}/videos?${qs}`));
const items = (data?.data || []).map((v) => ({
id: String(v.uuid || v.id), title: v.name, thumbnail: v?.thumbnailPath ? `https://${instance}${v.thumbnailPath}` : undefined,
url: v?.url, uploaderName: v?.channel?.displayName || channel,
channelId: externalId, channelExternalId: externalId,
duration: Number(v.duration || 0), views: Number(v.views || 0), publishedAt: v.publishedAt, type: 'video',
}));
return { items, nextPage: items.length >= perPage ? (page || 1) + 1 : null, total: data?.total ?? null };
}
// ---- Odysee ----
async function odContent(externalId, { type, page, limit, sort, q }) {
if (type !== 'videos') return { items: [], nextPage: null };
const claim = String(externalId || '').startsWith('@') ? String(externalId) : `@${String(externalId).replace(/^@/, '')}`;
const perPage = Math.min(Math.max(1, Number(limit || 24)), 50);
const body = {
jsonrpc: '2.0', id: 1, method: 'claim_search',
params: { channel: claim, page: Math.max(1, Number(page || 1)), page_size: perPage, claim_type: 'stream', order_by: sort === 'popular' ? ['effective_amount'] : ['release_time'] },
};
const resp = await fetchWithTimeout('https://api.na-backend.odysee.com/api/v1/proxy?m=claim_search', {
method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify(body),
});
const data = await resp.json().catch(() => ({}));
let items = ((data?.result?.items) || []).map((c) => ({
id: c.claim_id, title: c?.value?.title, thumbnail: c?.value?.thumbnail?.url,
url: c.short_url || c.canonical_url, uploaderName: c?.signing_channel?.value?.title || claim,
channelId: externalId, channelExternalId: externalId,
duration: Number(c?.value?.video?.duration || 0), publishedAt: c?.value?.release_time ? new Date(Number(c.value.release_time) * 1000).toISOString() : undefined,
type: 'video', slug: (c.short_url || '').replace('https://odysee.com/', ''),
}));
if (q) { const n = q.toLowerCase(); items = items.filter((i) => String(i.title || '').toLowerCase().includes(n)); }
const totalPages = data?.result?.total_pages;
return { items, nextPage: totalPages && (page || 1) < totalPages ? (page || 1) + 1 : (items.length >= perPage ? (page || 1) + 1 : null) };
}
// ---- Rumble : pas d'API publique -> recherche unifiée filtrée par chaîne ----
async function ruContent(externalId, { type, page, limit, q }, searchRegistry) {
if (type !== 'videos') return { items: [], nextPage: null };
try {
const mod = searchRegistry?.ru;
if (!mod || typeof mod.search !== 'function') return { items: [], nextPage: null };
const needle = String(externalId || '').replace(/^@/, '').toLowerCase();
const results = await mod.search(q || needle || 'videos', { limit: 50, page: 1 });
const items = (results || []).filter((r) => {
const hay = `${r?.uploaderName || ''} ${r?.channelId || ''} ${r?.url || ''}`.toLowerCase();
return !needle || hay.includes(needle);
}).slice(0, Number(limit || 24));
return { items, nextPage: null };
} catch { return { items: [], nextPage: null }; }
}
export async function fetchChannelContent(provider, externalId, opts = {}, ctx = {}) {
const { type = 'videos', page = 1, limit = 24, sort = 'recent', q = '' } = opts || {};
switch (provider) {
case 'yt': return ytContent(externalId, { type, page, limit, sort, q });
case 'dm': return dmContent(externalId, { type, page, limit, sort, q });
case 'tw': return twContent(externalId, { type, page, limit, sort, q });
case 'pt': return ptContent(externalId, { type, page, limit, sort, q });
case 'od': return odContent(externalId, { type, page, limit, sort, q });
case 'ru': return ruContent(externalId, { type, page, limit, q }, ctx.searchRegistry);
default: throw Object.assign(new Error('invalid_provider'), { status: 400 });
}
}