Files
NewTube/server/providers/rumble.mjs
T
bruno 665a0f0ebd
CI / build-and-test (push) Successful in 14m43s
feat(providers): phases 7.3/7.4/7.6/8.1 — provenance, health, NDJSON, contrat unique
7.3: capturedAt/source au registre + 6 adaptateurs + module provenance.ts + ?debug=1 (search-transport.mjs). 7.4: ProviderHealthService + badge source degradee. 7.6: squelettes par provider + snapshots progressifs + transport NDJSON /api/search. 8.1: ProviderAdapter unifie (search enveloppe + channelContent/channelMeta/capabilities) via getProviderAdapter + test de contrat offline.
2026-09-30 07:57:11 -04:00

466 lines
20 KiB
JavaScript

import { load } from 'cheerio';
import { spawn } from 'node:child_process';
import path from 'node:path';
import fs from 'node:fs';
import { fileURLToPath } from 'node:url';
/**
* Rumble provider.
*
* Context (2026-09): rumble.com sits behind Cloudflare. Server-side fetches from
* Node/OpenSSL receive 403 "Just a moment..." challenges on /search/* paths —
* the TLS (JA3) fingerprint is rejected regardless of headers. Verified working
* bypasses:
* - curl built against Windows Schannel (dev machine only, not portable);
* - python curl_cffi with browser impersonation (works on Linux/Docker).
*
* Fetch strategy, best-effort in order:
* 1. Node fetch with a full browser header set + shared __cf_bm cookie jar
* (works on networks where CF does not challenge this fingerprint);
* 2. python3 + curl_cffi helper (rumble_fetch.py) impersonating Chrome —
* the reliable path inside the Docker image (curl_cffi installed there);
* 3. if everything is challenged, return [] — the unified search already
* degrades gracefully per-provider.
*/
const BROWSER_HEADERS = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8',
'Accept-Language': 'en-US,en;q=0.9',
'sec-ch-ua': '"Chromium";v="131", "Not_A Brand";v="24"',
'sec-ch-ua-mobile': '?0',
'sec-ch-ua-platform': '"Windows"',
'sec-fetch-dest': 'document',
'sec-fetch-mode': 'navigate',
'sec-fetch-site': 'none',
'sec-fetch-user': '?1',
'upgrade-insecure-requests': '1',
};
// One shared cookie jar: Cloudflare issues __cf_bm on the first 200 and expects
// it on subsequent requests. Refreshed lazily every ~25 minutes.
let cookieJar = null;
let cookieJarAt = 0;
const COOKIE_JAR_TTL_MS = 25 * 60 * 1000;
const MODULE_DIR = path.dirname(fileURLToPath(import.meta.url));
const PYTHON_HELPER = path.join(MODULE_DIR, 'rumble_fetch.py');
const PYTHON_HELPER_EXISTS = fs.existsSync(PYTHON_HELPER);
function mergeSetCookies(existing, setCookieHeaders) {
if (!Array.isArray(setCookieHeaders) || setCookieHeaders.length === 0) return existing;
const jar = new Map();
for (const pair of String(existing || '').split(';').map(s => s.trim()).filter(Boolean)) {
const idx = pair.indexOf('=');
if (idx > 0) jar.set(pair.slice(0, idx), pair.slice(idx + 1));
}
for (const raw of setCookieHeaders) {
const first = String(raw || '').split(';')[0].trim();
const idx = first.indexOf('=');
if (idx > 0) jar.set(first.slice(0, idx), first.slice(idx + 1));
}
return Array.from(jar.entries()).map(([k, v]) => `${k}=${v}`).join('; ');
}
function headersWithCookies() {
const headers = { ...BROWSER_HEADERS };
if (cookieJar && (Date.now() - cookieJarAt) < COOKIE_JAR_TTL_MS) {
headers['Cookie'] = cookieJar;
}
return headers;
}
/** Node fetch with cookie-jar bookkeeping. Returns {status, html}. */
async function nodeFetch(url, { timeoutMs = 12_000 } = {}) {
const controller = new AbortController();
const tid = setTimeout(() => controller.abort(), timeoutMs);
try {
const res = await fetch(url, { headers: headersWithCookies(), redirect: 'follow', signal: controller.signal });
try {
const setCookies = typeof res.headers.getSetCookie === 'function' ? res.headers.getSetCookie() : [];
if (setCookies.length > 0) {
cookieJar = mergeSetCookies(cookieJar, setCookies);
cookieJarAt = Date.now();
}
} catch {}
const html = await res.text();
return { status: res.status, html };
} finally {
clearTimeout(tid);
}
}
/** python3 + curl_cffi helper. Returns {status, html} or null when unavailable. */
async function pythonFetch(url, { timeoutMs = 20_000 } = {}) {
if (!PYTHON_HELPER_EXISTS) return null;
const pythons = ['python3', 'python'];
for (const py of pythons) {
const result = await new Promise((resolve) => {
let settled = false;
const done = (v) => { if (!settled) { settled = true; resolve(v); } };
const child = spawn(py, [PYTHON_HELPER, url, String(Math.ceil(timeoutMs / 1000))], {
stdio: ['ignore', 'pipe', 'ignore'],
});
let stdout = '';
let failed = false;
const timer = setTimeout(() => { failed = true; try { child.kill(); } catch {} done(null); }, timeoutMs + 5_000);
child.stdout.on('data', (d) => { stdout += d.toString(); });
child.on('error', () => { clearTimeout(timer); done(null); });
child.on('close', (code) => {
clearTimeout(timer);
if (failed) return;
done(code === 0 && stdout.length > 1000 ? { status: 200, html: stdout } : null);
});
});
if (result) return result;
}
return null;
}
/** Node fetch first, python/curl_cffi fallback. */
async function fetchHtml(url) {
try {
const r = await nodeFetch(url);
if (r.status === 200 && !isChallenge(r.html)) return r;
} catch { /* fallback */ }
const py = await pythonFetch(url);
if (py) return py;
return null;
}
function isChallenge(html) {
return /Just a moment|challenge-platform|cf-chl/i.test(String(html || '').slice(0, 4000));
}
/**
* Phase 3.3 — cache négatif court.
* Sans lui, chaque recherche Rumble déclenchait 2 fetch Node + 2 fetch Python
* (12-20 s) pour finir sur un challenge Cloudflare, à chaque frappe utilisateur.
* TTL volontairement court (5 min) : un challenge se lève vite, et on ne veut
* pas figer une panne de plus de quelques minutes.
* @type {Map<string, number>}
*/
const NEGATIVE_CACHE_TTL_MS = Number(process.env.RUMBLE_NEGATIVE_CACHE_TTL_MS || 5 * 60 * 1000);
const negativeCache = new Map();
function negativeCacheKey(q, page) { return `search:vide:${page}:${String(q || '').trim().toLowerCase()}`; }
function isNegativelyCached(key) {
const ts = negativeCache.get(key);
if (!ts) return false;
if ((Date.now() - ts) >= NEGATIVE_CACHE_TTL_MS) { negativeCache.delete(key); return false; }
return true;
}
function markNegative(key) {
// Garde-fou mémoire : le cache ne doit pas grossir sans borne.
if (negativeCache.size >= 200) {
for (const [k, ts] of negativeCache) {
if ((Date.now() - ts) >= NEGATIVE_CACHE_TTL_MS) negativeCache.delete(k);
if (negativeCache.size < 150) break;
}
}
negativeCache.set(key, Date.now());
}
export function resetRumbleNegativeCache() { negativeCache.clear(); }
/* --------------------------------- parsing -------------------------------- */
function parseDurationToSeconds(raw) {
if (raw == null) return undefined;
const value = String(raw).trim();
if (!value) return undefined;
// Garde-fou : les attributs datetime (dates de publication, ex.
// "2026-09-23T18:11:52-04:00") ne sont JAMAIS des durées — sans ce test,
// l'extraction des chiffres+':' produit des durées absurdes (~351840 s).
if (/^\d{4}-\d{2}-\d{2}/.test(value)) return undefined;
const numeric = Number(value);
if (Number.isFinite(numeric) && numeric > 0) {
// Rumble expose parfois des durées en millisecondes sur les cartes de
// recherche (ex. 351840 ≈ 5:52). Au-delà de ~27 h en "secondes", c'est
// quasi certainement des ms : on convertit (sans impact sur le filtre
// Shorts, car >100 s reste une vidéo dans tous les cas).
if (/^\d+(\.\d+)?$/.test(value) && numeric > 100000) return Math.floor(numeric / 1000);
return Math.floor(numeric);
}
const isoMatch = value.match(/^PT(?:(\d+)H)?(?:(\d+)M)?(?:(\d+)S)?$/i);
if (isoMatch) {
const total = (Number(isoMatch[1] || 0) * 3600) + (Number(isoMatch[2] || 0) * 60) + Number(isoMatch[3] || 0);
if (total > 0) return total;
}
const textMatch = value.match(/^(?:(\d+)\s*h(?:ours?)?)?\s*(?:(\d+)\s*m(?:in(?:utes)?)?)?\s*(?:(\d+)\s*s(?:ec(?:onds)?)?)?$/i);
if (textMatch && (textMatch[1] || textMatch[2] || textMatch[3])) {
const total = (Number(textMatch[1] || 0) * 3600) + (Number(textMatch[2] || 0) * 60) + Number(textMatch[3] || 0);
if (total > 0) return total;
}
const colonCandidate = value.replace(/[^0-9:]/g, '');
if (colonCandidate.includes(':')) {
const segments = colonCandidate.split(':').filter(Boolean).map(s => Number(s));
if (segments.length >= 2 && segments.every(n => Number.isFinite(n))) {
while (segments.length < 3) segments.unshift(0);
const [h, m, s] = segments.slice(-3);
const total = (h * 3600) + (m * 60) + s;
if (total > 0) return total;
}
}
return undefined;
}
function normalizeThumb(raw) {
const t = String(raw || '').replace(/\s+/g, '').trim();
if (!t) return undefined;
return t.startsWith('//') ? `https:${t}` : t;
}
/**
* Compteur de vues localized : "1,2K views", "1.2M views", "3 456", "12,345".
* L'ancien code faisait `.replace(/[^\d]/g, '')` : le suffixe disparaissait et
* "1,2K" devenait 12 (au lieu de 1 200). Les vues étaient donc affichées
* fausses sur la quasi-totalité des cartes Rumble au-delà de 999.
*/
export function parseRumbleViews(raw) {
const s = String(raw ?? '').trim();
if (!s) return undefined;
const m = s.match(/([\d][\d\s\u00a0.,]*)\s*([KMBkmb])?/);
if (!m) return undefined;
// 1 234 -> "1234" ; 1,2 / 1.2 -> "1.2" (décimal) ; 1,234 -> "1234" (milliers)
let digits = m[1].replace(/[\s\u00a0]/g, '');
if (/^\d{1,3},\d{3}$/.test(digits) || /^\d{1,3}\.\d{3}$/.test(digits)) digits = digits.replace(/[.,]/g, '');
else digits = digits.replace(',', '.');
const num = Number(digits);
if (!Number.isFinite(num) || num < 0) return undefined;
const suffix = (m[2] || '').toLowerCase();
const mult = suffix === 'b' ? 1e9 : suffix === 'm' ? 1e6 : suffix === 'k' ? 1e3 : 1;
const total = Math.round(num * mult);
return total > 0 ? total : undefined;
}
/** Clé de rapprochement stable : dernier segment d'URL, sans extension ni tracking. */
function urlKey(u) {
const s = String(u || '').split('?')[0].replace(/\.html$/i, '').replace(/\/$/, '');
return (s.split('/').filter(Boolean).pop() || '').toLowerCase();
}
/**
* Phase 3.1 — extraction du JSON-LD (`application/ld+json`).
* Plus stable que le DOM : quand le balisage change, le JSON-LD reste. On
* accepte `VideoObject`, `ItemList` et `@graph`, et on n'indexe QUE ce qui est
* explicitement présent (aucune valeur déduite).
* @returns {Map<string, { publishedAt?: string, views?: number, thumbnail?: string, duration?: number, channelId?: string }>}
*/
export function parseJsonLd(html) {
/** @type {Map<string, any>} */
const out = new Map();
const re = /<script[^>]+type=["']application\/ld\+json["'][^>]*>([\s\S]*?)<\/script>/gi;
let m;
while ((m = re.exec(String(html || ''))) !== null) {
let data;
try { data = JSON.parse(m[1].trim()); } catch { continue; }
const nodes = Array.isArray(data) ? data
: Array.isArray(data?.['@graph']) ? data['@graph']
: data ? [data] : [];
for (const node of nodes) visit(node, out, 0);
}
return out;
}
function visit(node, out, depth) {
if (!node || typeof node !== 'object' || depth > 4) return;
const list = Array.isArray(node['@graph'])
? node['@graph']
: (Array.isArray(node.itemListElement) ? node.itemListElement.map((e) => e?.item ?? e) : null);
if (list) { for (const c of list) visit(c, out, depth + 1); }
const type = String(node['@type'] || '').toLowerCase();
if (!type.includes('video')) return;
const key = urlKey(node.url || node.embedUrl || node.contentUrl || node['@id']);
if (!key) return;
const stats = node.interactionStatistic;
const statList = Array.isArray(stats) ? stats : (stats ? [stats] : []);
let views;
for (const st of statList) {
const t = String(st?.interactionType || '').toLowerCase();
if (!t.includes('watch') && !t.includes('view')) continue;
const n = Number(st?.userInteractionCount);
if (Number.isFinite(n) && n >= 0) { views = Math.round(n); break; }
}
const dateRaw = node.uploadDate || node.datePublished;
const parsed = dateRaw ? Date.parse(dateRaw) : NaN;
const publishedAt = Number.isFinite(parsed) && parsed > 0 ? new Date(parsed).toISOString() : undefined;
const duration = parseDurationToSeconds(node.duration);
const thumbRaw = Array.isArray(node.thumbnailUrl)
? node.thumbnailUrl[0]
: (typeof node.thumbnailUrl === 'string' ? node.thumbnailUrl : node.thumbnailUrl?.url);
const authorUrl = typeof node.author === 'string' ? node.author : node.author?.url;
const channelId = (String(authorUrl || '').match(/\/c\/([^/?#]+)/i)?.[1] || '').trim() || undefined;
out.set(key, {
...(publishedAt ? { publishedAt } : {}),
...(views !== undefined && views > 0 ? { views } : {}),
...(duration !== undefined ? { duration } : {}),
...(thumbRaw ? { thumbnail: normalizeThumb(thumbRaw) } : {}),
...(channelId ? { channelId } : {}),
});
}
/** Classic parser: li.video-listing-entry cards. */
export function parseSearchHtml(html, { limit = 50 } = {}) {
const $ = load(html);
const items = [];
$('li.video-listing-entry').each((_idx, el) => {
if (items.length >= limit) return false;
const $el = $(el);
const anchor = ($el.find('a.video-item--a').attr('href') || '').trim();
if (!anchor) return;
const img = $el.find('img.video-item--img');
const rawThumbnail = img.attr('data-src') || img.attr('data-original') || img.attr('src') || '';
const title = $el.find('h3.video-item--title').text().replace(/\s+/g, ' ').trim();
const uploaderName = $el.find('.ellipsis-1').text().replace(/\s+/g, ' ').trim();
// Strip tracking params (e9s, sci, …) from the anchor for a canonical URL
const cleanAnchor = anchor.startsWith('http')
? anchor
: `https://rumble.com${anchor.split('?')[0]}`;
const url = cleanAnchor;
const id = $el.attr('data-id')
|| (url.split('/').filter(Boolean).pop() || '').replace(/\.html$/, '')
|| String(Math.random());
// Rumble expose la durée dans <span class="video-item--duration" data-value="3:06:42">
// (le span n'a pas de contenu texte). L'attribut datetime des <time> est
// une DATE de publication — jamais une durée (cf. garde-fou ci-dessus).
const durEl = $el.find('.video-item--duration, .video-item--meta time, .video-item--meta .duration').first();
const durationCandidates = [
$el.attr('data-duration'),
$el.attr('data-video-duration'),
$el.find('[data-duration]').attr('data-duration'),
durEl.attr('data-value'),
durEl.attr('data-duration'),
durEl.attr('title'),
durEl.text(),
];
let durationSeconds;
for (const candidate of durationCandidates) {
const parsed = parseDurationToSeconds(candidate);
if (typeof parsed === 'number' && parsed > 0) { durationSeconds = parsed; break; }
}
const viewsText = $el.find('.video-item--views').first().text().trim();
const views = parseRumbleViews(viewsText);
// Phase 1.4 - date de publication. L'attribut `datetime` du <time> est la
// vraie date ISO ; son texte est un libelle relatif ("2 months ago") que
// Date.parse() ne sait pas lire, d'ou l'absence historique de publishedAt.
const timeEl = $el.find('.video-item--meta time, time.video-item--date, time').first();
const publishedRaw = (timeEl.attr('datetime') || timeEl.attr('data-datetime') || '').trim();
const publishedMs = publishedRaw ? Date.parse(publishedRaw) : NaN;
const publishedAt = Number.isFinite(publishedMs) && publishedMs > 0
? new Date(publishedMs).toISOString()
: undefined;
// Phase 1.5 - identifiant de chaine. L'URL porte soit /c/<username>/,
// soit (en fallback) le sous-domaine ; l'avatar est sur l'image du by-line.
const channelLink = $el.find('a.video-item--channel-link, a[href*="/c/"]').first();
const channelHref = (channelLink.attr('href') || '').trim();
const channelId = (channelHref.match(/\/c\/([^/?#]+)/i)?.[1] || '').trim() || undefined;
const avatarEl = $el.find('img.video-item--channel-thumb, .video-item--by-line img, a.video-item--channel-link img').first();
const uploaderAvatar = normalizeThumb(avatarEl.attr('src') || avatarEl.attr('data-src') || '');
items.push({
title: title || url,
id,
url,
thumbnail: normalizeThumb(rawThumbnail),
uploaderName: uploaderName || undefined,
// Phase 1.4 / 1.5 / 1.6 - les trois champs etaient presents dans le HTML
// mais jamais extraits, ce qui vidait la page chaine de sa date, de son
// lien "voir la chaine" et de son avatar.
publishedAt,
channelId,
uploaderAvatar,
views,
type: 'video',
duration: durationSeconds,
// Pas de flag maison ici : le filtre central (isShortItem,
// durée ≤75 s) tranche. Le scraper HTML n'expose pas de dimensions.
});
});
// Phase 3.1 — surcouche JSON-LD : complète ce que le DOM n'expose pas
// (date de publication, vues, auteur) SANS jamais écraser une valeur réellement
// présente dans la liste. Le DOM reste la source primaire, le JSON-LD le filet.
try {
const ld = parseJsonLd(html);
if (ld.size > 0) {
for (const it of items) {
const patch = ld.get(urlKey(it.url)) || ld.get(urlKey(it.id));
if (!patch) continue;
if (it.publishedAt === undefined && patch.publishedAt) it.publishedAt = patch.publishedAt;
if (it.views === undefined && patch.views) it.views = patch.views;
if (it.duration === undefined && patch.duration) it.duration = patch.duration;
if (it.thumbnail === undefined && patch.thumbnail) it.thumbnail = patch.thumbnail;
if (it.channelId === undefined && patch.channelId) it.channelId = patch.channelId;
}
}
} catch { /* la surcouche est un bonus, jamais un motif d'échec */ }
return items;
}
/* --------------------------------- handler -------------------------------- */
const handler = {
id: 'ru',
label: 'Rumble',
/**
* @param {string} q
* @param {{ limit: number, page?: number, sort?: string }} opts
* @returns {Promise<Array<any>>}
*/
async search(q, opts) {
const { limit = 10, page = 1 } = opts || {};
const perPage = Math.min(Math.max(1, Number(limit || 10)), 50);
const pageNum = Math.max(1, Number(page || 1));
const query = String(q || '').trim();
if (!query) return [];
// Phase 3.3 — short-circuit si l'échec vient d'être constaté.
const cacheKey = negativeCacheKey(query, pageNum);
if (isNegativelyCached(cacheKey)) {
// Phase 3.4 — un échec est une ERREUR, pas un résultat vide : le front
// peut ainsi afficher un bandeau au lieu d'un « aucun résultat » trompeur.
throw Object.assign(new Error('rumble_cloudflare_challenge'), {
code: 'rumble_cloudflare_challenge', rumbleChallenged: true,
});
}
// --- Attempt 1: canonical search page (Node fetch, then python/curl_cffi) ---
try {
const params = new URLSearchParams({ q: query });
if (pageNum > 1) params.set('page', String(pageNum));
const r = await fetchHtml(`https://rumble.com/search/video?${params.toString()}`);
if (r && !isChallenge(r.html)) {
const items = parseSearchHtml(r.html, { limit: perPage });
if (items.length > 0) return items;
}
} catch { /* next */ }
// --- Attempt 2: search/all (all-types page, video facet included) ---
try {
const params = new URLSearchParams({ 'search-videos': '1', q: query });
if (pageNum > 1) params.set('page', String(pageNum));
const r = await fetchHtml(`https://rumble.com/search/all?${params.toString()}`);
if (r && !isChallenge(r.html)) {
const items = parseSearchHtml(r.html, { limit: perPage });
if (items.length > 0) return items;
}
} catch { /* give up */ }
// Phase 3.4 — on distingue « échec » de « vide ». Une recherche Rumble qui
// renvoie 0 résultat à travers deux chemins est presque toujours un blocage
// Cloudflare, pas une absence de contenu : on le remonte, on met en cache
// négatif, et le groupe ne passe pas pour un recherche normale sans résultat.
markNegative(cacheKey);
throw Object.assign(new Error('rumble_unavailable'), {
code: 'rumble_unavailable', rumbleChallenged: true,
});
},
};
export default handler;