Files
NewTube/server/providers/rumble.mjs
T
bruno 97e5de74f5
CI / build-and-test (push) Successful in 13m34s
feat(channel): page details chaine multi-providers avec onglets et infinite scroll
- Strategy Pattern front: ChannelProviderInterface + 6 providers (yt/dm/tw/pt/od/ru) + factory (ids longs/courts)
- ChannelContentService: cache par onglet/tri/recherche (TTL 5min), pagination nextPage
- ChannelPage: header (banniere/avatar/subs/description ...plus), tabs dynamiques par capabilities, grilles videos/shorts 9:16/playlists/live, recherche debounced, tri recent/populaire, infinite scroll IntersectionObserver, skeletons/empty/error
- Backend: GET /api/channels/:provider/:id/content (videos/shorts/playlists/live) natif par provider
2026-09-25 10:15:12 -04:00

284 lines
11 KiB
JavaScript

import { load } from 'cheerio';
import { spawn } from 'node:child_process';
import path from 'node:path';
import fs from 'node:fs';
import { fileURLToPath } from 'node:url';
/**
* Rumble provider.
*
* Context (2026-09): rumble.com sits behind Cloudflare. Server-side fetches from
* Node/OpenSSL receive 403 "Just a moment..." challenges on /search/* paths —
* the TLS (JA3) fingerprint is rejected regardless of headers. Verified working
* bypasses:
* - curl built against Windows Schannel (dev machine only, not portable);
* - python curl_cffi with browser impersonation (works on Linux/Docker).
*
* Fetch strategy, best-effort in order:
* 1. Node fetch with a full browser header set + shared __cf_bm cookie jar
* (works on networks where CF does not challenge this fingerprint);
* 2. python3 + curl_cffi helper (rumble_fetch.py) impersonating Chrome —
* the reliable path inside the Docker image (curl_cffi installed there);
* 3. if everything is challenged, return [] — the unified search already
* degrades gracefully per-provider.
*/
const BROWSER_HEADERS = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8',
'Accept-Language': 'en-US,en;q=0.9',
'sec-ch-ua': '"Chromium";v="131", "Not_A Brand";v="24"',
'sec-ch-ua-mobile': '?0',
'sec-ch-ua-platform': '"Windows"',
'sec-fetch-dest': 'document',
'sec-fetch-mode': 'navigate',
'sec-fetch-site': 'none',
'sec-fetch-user': '?1',
'upgrade-insecure-requests': '1',
};
// One shared cookie jar: Cloudflare issues __cf_bm on the first 200 and expects
// it on subsequent requests. Refreshed lazily every ~25 minutes.
let cookieJar = null;
let cookieJarAt = 0;
const COOKIE_JAR_TTL_MS = 25 * 60 * 1000;
const MODULE_DIR = path.dirname(fileURLToPath(import.meta.url));
const PYTHON_HELPER = path.join(MODULE_DIR, 'rumble_fetch.py');
const PYTHON_HELPER_EXISTS = fs.existsSync(PYTHON_HELPER);
function mergeSetCookies(existing, setCookieHeaders) {
if (!Array.isArray(setCookieHeaders) || setCookieHeaders.length === 0) return existing;
const jar = new Map();
for (const pair of String(existing || '').split(';').map(s => s.trim()).filter(Boolean)) {
const idx = pair.indexOf('=');
if (idx > 0) jar.set(pair.slice(0, idx), pair.slice(idx + 1));
}
for (const raw of setCookieHeaders) {
const first = String(raw || '').split(';')[0].trim();
const idx = first.indexOf('=');
if (idx > 0) jar.set(first.slice(0, idx), first.slice(idx + 1));
}
return Array.from(jar.entries()).map(([k, v]) => `${k}=${v}`).join('; ');
}
function headersWithCookies() {
const headers = { ...BROWSER_HEADERS };
if (cookieJar && (Date.now() - cookieJarAt) < COOKIE_JAR_TTL_MS) {
headers['Cookie'] = cookieJar;
}
return headers;
}
/** Node fetch with cookie-jar bookkeeping. Returns {status, html}. */
async function nodeFetch(url, { timeoutMs = 12_000 } = {}) {
const controller = new AbortController();
const tid = setTimeout(() => controller.abort(), timeoutMs);
try {
const res = await fetch(url, { headers: headersWithCookies(), redirect: 'follow', signal: controller.signal });
try {
const setCookies = typeof res.headers.getSetCookie === 'function' ? res.headers.getSetCookie() : [];
if (setCookies.length > 0) {
cookieJar = mergeSetCookies(cookieJar, setCookies);
cookieJarAt = Date.now();
}
} catch {}
const html = await res.text();
return { status: res.status, html };
} finally {
clearTimeout(tid);
}
}
/** python3 + curl_cffi helper. Returns {status, html} or null when unavailable. */
async function pythonFetch(url, { timeoutMs = 20_000 } = {}) {
if (!PYTHON_HELPER_EXISTS) return null;
const pythons = ['python3', 'python'];
for (const py of pythons) {
const result = await new Promise((resolve) => {
let settled = false;
const done = (v) => { if (!settled) { settled = true; resolve(v); } };
const child = spawn(py, [PYTHON_HELPER, url, String(Math.ceil(timeoutMs / 1000))], {
stdio: ['ignore', 'pipe', 'ignore'],
});
let stdout = '';
let failed = false;
const timer = setTimeout(() => { failed = true; try { child.kill(); } catch {} done(null); }, timeoutMs + 5_000);
child.stdout.on('data', (d) => { stdout += d.toString(); });
child.on('error', () => { clearTimeout(timer); done(null); });
child.on('close', (code) => {
clearTimeout(timer);
if (failed) return;
done(code === 0 && stdout.length > 1000 ? { status: 200, html: stdout } : null);
});
});
if (result) return result;
}
return null;
}
/** Node fetch first, python/curl_cffi fallback. */
async function fetchHtml(url) {
try {
const r = await nodeFetch(url);
if (r.status === 200 && !isChallenge(r.html)) return r;
} catch { /* fallback */ }
const py = await pythonFetch(url);
if (py) return py;
return null;
}
function isChallenge(html) {
return /Just a moment|challenge-platform|cf-chl/i.test(String(html || '').slice(0, 4000));
}
/* --------------------------------- parsing -------------------------------- */
function parseDurationToSeconds(raw) {
if (raw == null) return undefined;
const value = String(raw).trim();
if (!value) return undefined;
// Garde-fou : les attributs datetime (dates de publication, ex.
// "2026-09-23T18:11:52-04:00") ne sont JAMAIS des durées — sans ce test,
// l'extraction des chiffres+':' produit des durées absurdes (~351840 s).
if (/^\d{4}-\d{2}-\d{2}/.test(value)) return undefined;
const numeric = Number(value);
if (Number.isFinite(numeric) && numeric > 0) {
// Rumble expose parfois des durées en millisecondes sur les cartes de
// recherche (ex. 351840 ≈ 5:52). Au-delà de ~27 h en "secondes", c'est
// quasi certainement des ms : on convertit (sans impact sur le filtre
// Shorts, car >100 s reste une vidéo dans tous les cas).
if (/^\d+(\.\d+)?$/.test(value) && numeric > 100000) return Math.floor(numeric / 1000);
return Math.floor(numeric);
}
const isoMatch = value.match(/^PT(?:(\d+)H)?(?:(\d+)M)?(?:(\d+)S)?$/i);
if (isoMatch) {
const total = (Number(isoMatch[1] || 0) * 3600) + (Number(isoMatch[2] || 0) * 60) + Number(isoMatch[3] || 0);
if (total > 0) return total;
}
const textMatch = value.match(/^(?:(\d+)\s*h(?:ours?)?)?\s*(?:(\d+)\s*m(?:in(?:utes)?)?)?\s*(?:(\d+)\s*s(?:ec(?:onds)?)?)?$/i);
if (textMatch && (textMatch[1] || textMatch[2] || textMatch[3])) {
const total = (Number(textMatch[1] || 0) * 3600) + (Number(textMatch[2] || 0) * 60) + Number(textMatch[3] || 0);
if (total > 0) return total;
}
const colonCandidate = value.replace(/[^0-9:]/g, '');
if (colonCandidate.includes(':')) {
const segments = colonCandidate.split(':').filter(Boolean).map(s => Number(s));
if (segments.length >= 2 && segments.every(n => Number.isFinite(n))) {
while (segments.length < 3) segments.unshift(0);
const [h, m, s] = segments.slice(-3);
const total = (h * 3600) + (m * 60) + s;
if (total > 0) return total;
}
}
return undefined;
}
function normalizeThumb(raw) {
const t = String(raw || '').replace(/\s+/g, '').trim();
if (!t) return undefined;
return t.startsWith('//') ? `https:${t}` : t;
}
/** Classic parser: li.video-listing-entry cards. */
function parseSearchHtml(html, { limit = 50 } = {}) {
const $ = load(html);
const items = [];
$('li.video-listing-entry').each((_idx, el) => {
if (items.length >= limit) return false;
const $el = $(el);
const anchor = ($el.find('a.video-item--a').attr('href') || '').trim();
if (!anchor) return;
const img = $el.find('img.video-item--img');
const rawThumbnail = img.attr('data-src') || img.attr('data-original') || img.attr('src') || '';
const title = $el.find('h3.video-item--title').text().replace(/\s+/g, ' ').trim();
const uploaderName = $el.find('.ellipsis-1').text().replace(/\s+/g, ' ').trim();
// Strip tracking params (e9s, sci, …) from the anchor for a canonical URL
const cleanAnchor = anchor.startsWith('http')
? anchor
: `https://rumble.com${anchor.split('?')[0]}`;
const url = cleanAnchor;
const id = $el.attr('data-id')
|| (url.split('/').filter(Boolean).pop() || '').replace(/\.html$/, '')
|| String(Math.random());
// Rumble expose la durée dans <span class="video-item--duration" data-value="3:06:42">
// (le span n'a pas de contenu texte). L'attribut datetime des <time> est
// une DATE de publication — jamais une durée (cf. garde-fou ci-dessus).
const durEl = $el.find('.video-item--duration, .video-item--meta time, .video-item--meta .duration').first();
const durationCandidates = [
$el.attr('data-duration'),
$el.attr('data-video-duration'),
$el.find('[data-duration]').attr('data-duration'),
durEl.attr('data-value'),
durEl.attr('data-duration'),
durEl.attr('title'),
durEl.text(),
];
let durationSeconds;
for (const candidate of durationCandidates) {
const parsed = parseDurationToSeconds(candidate);
if (typeof parsed === 'number' && parsed > 0) { durationSeconds = parsed; break; }
}
const viewsText = $el.find('.video-item--views').first().text().trim();
const views = Number(String(viewsText).replace(/[^\d]/g, '')) || undefined;
items.push({
title: title || url,
id,
url,
thumbnail: normalizeThumb(rawThumbnail),
uploaderName: uploaderName || undefined,
views,
type: 'video',
duration: durationSeconds,
// Pas de flag natif côté Rumble : durée courte (1..75 s) ⇒ Short probable.
isShort: typeof durationSeconds === 'number' && durationSeconds > 0 && durationSeconds <= 75,
});
});
return items;
}
/* --------------------------------- handler -------------------------------- */
const handler = {
id: 'ru',
label: 'Rumble',
/**
* @param {string} q
* @param {{ limit: number, page?: number, sort?: string }} opts
* @returns {Promise<Array<any>>}
*/
async search(q, opts) {
const { limit = 10, page = 1 } = opts || {};
const perPage = Math.min(Math.max(1, Number(limit || 10)), 50);
const pageNum = Math.max(1, Number(page || 1));
const query = String(q || '').trim();
if (!query) return [];
// --- Attempt 1: canonical search page (Node fetch, then python/curl_cffi) ---
try {
const params = new URLSearchParams({ q: query });
if (pageNum > 1) params.set('page', String(pageNum));
const r = await fetchHtml(`https://rumble.com/search/video?${params.toString()}`);
if (r && !isChallenge(r.html)) {
const items = parseSearchHtml(r.html, { limit: perPage });
if (items.length > 0) return items;
}
} catch { /* next */ }
// --- Attempt 2: search/all (all-types page, video facet included) ---
try {
const params = new URLSearchParams({ 'search-videos': '1', q: query });
if (pageNum > 1) params.set('page', String(pageNum));
const r = await fetchHtml(`https://rumble.com/search/all?${params.toString()}`);
if (r && !isChallenge(r.html)) {
const items = parseSearchHtml(r.html, { limit: perPage });
if (items.length > 0) return items;
}
} catch { /* give up */ }
return [];
},
};
export default handler;