CI / build-and-test (push) Successful in 13m34s
- Strategy Pattern front: ChannelProviderInterface + 6 providers (yt/dm/tw/pt/od/ru) + factory (ids longs/courts) - ChannelContentService: cache par onglet/tri/recherche (TTL 5min), pagination nextPage - ChannelPage: header (banniere/avatar/subs/description ...plus), tabs dynamiques par capabilities, grilles videos/shorts 9:16/playlists/live, recherche debounced, tri recent/populaire, infinite scroll IntersectionObserver, skeletons/empty/error - Backend: GET /api/channels/:provider/:id/content (videos/shorts/playlists/live) natif par provider
284 lines
11 KiB
JavaScript
284 lines
11 KiB
JavaScript
import { load } from 'cheerio';
|
|
import { spawn } from 'node:child_process';
|
|
import path from 'node:path';
|
|
import fs from 'node:fs';
|
|
import { fileURLToPath } from 'node:url';
|
|
|
|
/**
|
|
* Rumble provider.
|
|
*
|
|
* Context (2026-09): rumble.com sits behind Cloudflare. Server-side fetches from
|
|
* Node/OpenSSL receive 403 "Just a moment..." challenges on /search/* paths —
|
|
* the TLS (JA3) fingerprint is rejected regardless of headers. Verified working
|
|
* bypasses:
|
|
* - curl built against Windows Schannel (dev machine only, not portable);
|
|
* - python curl_cffi with browser impersonation (works on Linux/Docker).
|
|
*
|
|
* Fetch strategy, best-effort in order:
|
|
* 1. Node fetch with a full browser header set + shared __cf_bm cookie jar
|
|
* (works on networks where CF does not challenge this fingerprint);
|
|
* 2. python3 + curl_cffi helper (rumble_fetch.py) impersonating Chrome —
|
|
* the reliable path inside the Docker image (curl_cffi installed there);
|
|
* 3. if everything is challenged, return [] — the unified search already
|
|
* degrades gracefully per-provider.
|
|
*/
|
|
|
|
const BROWSER_HEADERS = {
|
|
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
|
|
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8',
|
|
'Accept-Language': 'en-US,en;q=0.9',
|
|
'sec-ch-ua': '"Chromium";v="131", "Not_A Brand";v="24"',
|
|
'sec-ch-ua-mobile': '?0',
|
|
'sec-ch-ua-platform': '"Windows"',
|
|
'sec-fetch-dest': 'document',
|
|
'sec-fetch-mode': 'navigate',
|
|
'sec-fetch-site': 'none',
|
|
'sec-fetch-user': '?1',
|
|
'upgrade-insecure-requests': '1',
|
|
};
|
|
|
|
// One shared cookie jar: Cloudflare issues __cf_bm on the first 200 and expects
|
|
// it on subsequent requests. Refreshed lazily every ~25 minutes.
|
|
let cookieJar = null;
|
|
let cookieJarAt = 0;
|
|
const COOKIE_JAR_TTL_MS = 25 * 60 * 1000;
|
|
|
|
const MODULE_DIR = path.dirname(fileURLToPath(import.meta.url));
|
|
const PYTHON_HELPER = path.join(MODULE_DIR, 'rumble_fetch.py');
|
|
const PYTHON_HELPER_EXISTS = fs.existsSync(PYTHON_HELPER);
|
|
|
|
function mergeSetCookies(existing, setCookieHeaders) {
|
|
if (!Array.isArray(setCookieHeaders) || setCookieHeaders.length === 0) return existing;
|
|
const jar = new Map();
|
|
for (const pair of String(existing || '').split(';').map(s => s.trim()).filter(Boolean)) {
|
|
const idx = pair.indexOf('=');
|
|
if (idx > 0) jar.set(pair.slice(0, idx), pair.slice(idx + 1));
|
|
}
|
|
for (const raw of setCookieHeaders) {
|
|
const first = String(raw || '').split(';')[0].trim();
|
|
const idx = first.indexOf('=');
|
|
if (idx > 0) jar.set(first.slice(0, idx), first.slice(idx + 1));
|
|
}
|
|
return Array.from(jar.entries()).map(([k, v]) => `${k}=${v}`).join('; ');
|
|
}
|
|
|
|
function headersWithCookies() {
|
|
const headers = { ...BROWSER_HEADERS };
|
|
if (cookieJar && (Date.now() - cookieJarAt) < COOKIE_JAR_TTL_MS) {
|
|
headers['Cookie'] = cookieJar;
|
|
}
|
|
return headers;
|
|
}
|
|
|
|
/** Node fetch with cookie-jar bookkeeping. Returns {status, html}. */
|
|
async function nodeFetch(url, { timeoutMs = 12_000 } = {}) {
|
|
const controller = new AbortController();
|
|
const tid = setTimeout(() => controller.abort(), timeoutMs);
|
|
try {
|
|
const res = await fetch(url, { headers: headersWithCookies(), redirect: 'follow', signal: controller.signal });
|
|
try {
|
|
const setCookies = typeof res.headers.getSetCookie === 'function' ? res.headers.getSetCookie() : [];
|
|
if (setCookies.length > 0) {
|
|
cookieJar = mergeSetCookies(cookieJar, setCookies);
|
|
cookieJarAt = Date.now();
|
|
}
|
|
} catch {}
|
|
const html = await res.text();
|
|
return { status: res.status, html };
|
|
} finally {
|
|
clearTimeout(tid);
|
|
}
|
|
}
|
|
|
|
/** python3 + curl_cffi helper. Returns {status, html} or null when unavailable. */
|
|
async function pythonFetch(url, { timeoutMs = 20_000 } = {}) {
|
|
if (!PYTHON_HELPER_EXISTS) return null;
|
|
const pythons = ['python3', 'python'];
|
|
for (const py of pythons) {
|
|
const result = await new Promise((resolve) => {
|
|
let settled = false;
|
|
const done = (v) => { if (!settled) { settled = true; resolve(v); } };
|
|
const child = spawn(py, [PYTHON_HELPER, url, String(Math.ceil(timeoutMs / 1000))], {
|
|
stdio: ['ignore', 'pipe', 'ignore'],
|
|
});
|
|
let stdout = '';
|
|
let failed = false;
|
|
const timer = setTimeout(() => { failed = true; try { child.kill(); } catch {} done(null); }, timeoutMs + 5_000);
|
|
child.stdout.on('data', (d) => { stdout += d.toString(); });
|
|
child.on('error', () => { clearTimeout(timer); done(null); });
|
|
child.on('close', (code) => {
|
|
clearTimeout(timer);
|
|
if (failed) return;
|
|
done(code === 0 && stdout.length > 1000 ? { status: 200, html: stdout } : null);
|
|
});
|
|
});
|
|
if (result) return result;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/** Node fetch first, python/curl_cffi fallback. */
|
|
async function fetchHtml(url) {
|
|
try {
|
|
const r = await nodeFetch(url);
|
|
if (r.status === 200 && !isChallenge(r.html)) return r;
|
|
} catch { /* fallback */ }
|
|
const py = await pythonFetch(url);
|
|
if (py) return py;
|
|
return null;
|
|
}
|
|
|
|
function isChallenge(html) {
|
|
return /Just a moment|challenge-platform|cf-chl/i.test(String(html || '').slice(0, 4000));
|
|
}
|
|
|
|
/* --------------------------------- parsing -------------------------------- */
|
|
|
|
function parseDurationToSeconds(raw) {
|
|
if (raw == null) return undefined;
|
|
const value = String(raw).trim();
|
|
if (!value) return undefined;
|
|
// Garde-fou : les attributs datetime (dates de publication, ex.
|
|
// "2026-09-23T18:11:52-04:00") ne sont JAMAIS des durées — sans ce test,
|
|
// l'extraction des chiffres+':' produit des durées absurdes (~351840 s).
|
|
if (/^\d{4}-\d{2}-\d{2}/.test(value)) return undefined;
|
|
const numeric = Number(value);
|
|
if (Number.isFinite(numeric) && numeric > 0) {
|
|
// Rumble expose parfois des durées en millisecondes sur les cartes de
|
|
// recherche (ex. 351840 ≈ 5:52). Au-delà de ~27 h en "secondes", c'est
|
|
// quasi certainement des ms : on convertit (sans impact sur le filtre
|
|
// Shorts, car >100 s reste une vidéo dans tous les cas).
|
|
if (/^\d+(\.\d+)?$/.test(value) && numeric > 100000) return Math.floor(numeric / 1000);
|
|
return Math.floor(numeric);
|
|
}
|
|
const isoMatch = value.match(/^PT(?:(\d+)H)?(?:(\d+)M)?(?:(\d+)S)?$/i);
|
|
if (isoMatch) {
|
|
const total = (Number(isoMatch[1] || 0) * 3600) + (Number(isoMatch[2] || 0) * 60) + Number(isoMatch[3] || 0);
|
|
if (total > 0) return total;
|
|
}
|
|
const textMatch = value.match(/^(?:(\d+)\s*h(?:ours?)?)?\s*(?:(\d+)\s*m(?:in(?:utes)?)?)?\s*(?:(\d+)\s*s(?:ec(?:onds)?)?)?$/i);
|
|
if (textMatch && (textMatch[1] || textMatch[2] || textMatch[3])) {
|
|
const total = (Number(textMatch[1] || 0) * 3600) + (Number(textMatch[2] || 0) * 60) + Number(textMatch[3] || 0);
|
|
if (total > 0) return total;
|
|
}
|
|
const colonCandidate = value.replace(/[^0-9:]/g, '');
|
|
if (colonCandidate.includes(':')) {
|
|
const segments = colonCandidate.split(':').filter(Boolean).map(s => Number(s));
|
|
if (segments.length >= 2 && segments.every(n => Number.isFinite(n))) {
|
|
while (segments.length < 3) segments.unshift(0);
|
|
const [h, m, s] = segments.slice(-3);
|
|
const total = (h * 3600) + (m * 60) + s;
|
|
if (total > 0) return total;
|
|
}
|
|
}
|
|
return undefined;
|
|
}
|
|
|
|
function normalizeThumb(raw) {
|
|
const t = String(raw || '').replace(/\s+/g, '').trim();
|
|
if (!t) return undefined;
|
|
return t.startsWith('//') ? `https:${t}` : t;
|
|
}
|
|
|
|
/** Classic parser: li.video-listing-entry cards. */
|
|
function parseSearchHtml(html, { limit = 50 } = {}) {
|
|
const $ = load(html);
|
|
const items = [];
|
|
$('li.video-listing-entry').each((_idx, el) => {
|
|
if (items.length >= limit) return false;
|
|
const $el = $(el);
|
|
const anchor = ($el.find('a.video-item--a').attr('href') || '').trim();
|
|
if (!anchor) return;
|
|
const img = $el.find('img.video-item--img');
|
|
const rawThumbnail = img.attr('data-src') || img.attr('data-original') || img.attr('src') || '';
|
|
const title = $el.find('h3.video-item--title').text().replace(/\s+/g, ' ').trim();
|
|
const uploaderName = $el.find('.ellipsis-1').text().replace(/\s+/g, ' ').trim();
|
|
// Strip tracking params (e9s, sci, …) from the anchor for a canonical URL
|
|
const cleanAnchor = anchor.startsWith('http')
|
|
? anchor
|
|
: `https://rumble.com${anchor.split('?')[0]}`;
|
|
const url = cleanAnchor;
|
|
const id = $el.attr('data-id')
|
|
|| (url.split('/').filter(Boolean).pop() || '').replace(/\.html$/, '')
|
|
|| String(Math.random());
|
|
// Rumble expose la durée dans <span class="video-item--duration" data-value="3:06:42">
|
|
// (le span n'a pas de contenu texte). L'attribut datetime des <time> est
|
|
// une DATE de publication — jamais une durée (cf. garde-fou ci-dessus).
|
|
const durEl = $el.find('.video-item--duration, .video-item--meta time, .video-item--meta .duration').first();
|
|
const durationCandidates = [
|
|
$el.attr('data-duration'),
|
|
$el.attr('data-video-duration'),
|
|
$el.find('[data-duration]').attr('data-duration'),
|
|
durEl.attr('data-value'),
|
|
durEl.attr('data-duration'),
|
|
durEl.attr('title'),
|
|
durEl.text(),
|
|
];
|
|
let durationSeconds;
|
|
for (const candidate of durationCandidates) {
|
|
const parsed = parseDurationToSeconds(candidate);
|
|
if (typeof parsed === 'number' && parsed > 0) { durationSeconds = parsed; break; }
|
|
}
|
|
const viewsText = $el.find('.video-item--views').first().text().trim();
|
|
const views = Number(String(viewsText).replace(/[^\d]/g, '')) || undefined;
|
|
items.push({
|
|
title: title || url,
|
|
id,
|
|
url,
|
|
thumbnail: normalizeThumb(rawThumbnail),
|
|
uploaderName: uploaderName || undefined,
|
|
views,
|
|
type: 'video',
|
|
duration: durationSeconds,
|
|
// Pas de flag natif côté Rumble : durée courte (1..75 s) ⇒ Short probable.
|
|
isShort: typeof durationSeconds === 'number' && durationSeconds > 0 && durationSeconds <= 75,
|
|
});
|
|
});
|
|
return items;
|
|
}
|
|
|
|
/* --------------------------------- handler -------------------------------- */
|
|
|
|
const handler = {
|
|
id: 'ru',
|
|
label: 'Rumble',
|
|
/**
|
|
* @param {string} q
|
|
* @param {{ limit: number, page?: number, sort?: string }} opts
|
|
* @returns {Promise<Array<any>>}
|
|
*/
|
|
async search(q, opts) {
|
|
const { limit = 10, page = 1 } = opts || {};
|
|
const perPage = Math.min(Math.max(1, Number(limit || 10)), 50);
|
|
const pageNum = Math.max(1, Number(page || 1));
|
|
const query = String(q || '').trim();
|
|
if (!query) return [];
|
|
|
|
// --- Attempt 1: canonical search page (Node fetch, then python/curl_cffi) ---
|
|
try {
|
|
const params = new URLSearchParams({ q: query });
|
|
if (pageNum > 1) params.set('page', String(pageNum));
|
|
const r = await fetchHtml(`https://rumble.com/search/video?${params.toString()}`);
|
|
if (r && !isChallenge(r.html)) {
|
|
const items = parseSearchHtml(r.html, { limit: perPage });
|
|
if (items.length > 0) return items;
|
|
}
|
|
} catch { /* next */ }
|
|
|
|
// --- Attempt 2: search/all (all-types page, video facet included) ---
|
|
try {
|
|
const params = new URLSearchParams({ 'search-videos': '1', q: query });
|
|
if (pageNum > 1) params.set('page', String(pageNum));
|
|
const r = await fetchHtml(`https://rumble.com/search/all?${params.toString()}`);
|
|
if (r && !isChallenge(r.html)) {
|
|
const items = parseSearchHtml(r.html, { limit: perPage });
|
|
if (items.length > 0) return items;
|
|
}
|
|
} catch { /* give up */ }
|
|
|
|
return [];
|
|
},
|
|
};
|
|
|
|
export default handler;
|