refactor(rumble): un seul cœur de scraping, routeur fin, contrat 503 CF
CI / build-and-test (push) Successful in 14m57s
CI / build-and-test (push) Successful in 14m57s
- server/providers/rumble.mjs devient l'unique implémentation : fetch (cookie jar __cf_bm + curl_cffi + cooldown global 60 s après un échec total), parsing liste/vidéo avec surcouche JSON-LD partagée, vues via parseRumbleViews partout (« 1,2K » n'est plus lu en 12 sur /api/rumble). - server/rumble.mjs = couche HTTP mince : limiter 20/min, cache 60 s, 5 routes qui délèguent au cœur ; /search passe par le registre de providers (cache SQLite, negative cache et channelRef partagés avec /api/search). - Contrat d'erreur : un blocage Cloudflare est un 503 rumble_cloudflare_challenge, plus jamais un 404 (qui faisait retirer les vidéos du flux Shorts lors d'un challenge) ; page reçue sans identité = 404 rumble_video_not_found. - Code mort supprimé : scrapeRumbleVideo d'index.mjs (191 l., 3e implémentation jamais appelée), rumbleLimiter 10/min jamais monté, import cheerio devenu inutile. Net -238 lignes. - Tests : parseur générique (vues, durées, ids /shorts/), contrat 503 sans réseau via le cooldown, durées absentes -> undefined. - Docs : API_MCP_GUIDE §5.5 (routes, rate-limit réel, erreurs) + note README « Rumble peut être temporairement absent selon le réseau ».
This commit is contained in:
+87
-515
@@ -1,15 +1,29 @@
|
||||
import express from 'express';
|
||||
import * as cheerio from 'cheerio';
|
||||
import axios from 'axios';
|
||||
import rateLimit from 'express-rate-limit';
|
||||
import { spawn } from 'node:child_process';
|
||||
import path from 'node:path';
|
||||
import fs from 'node:fs';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { getProviderAdapter } from './providers/registry.mjs';
|
||||
import {
|
||||
normalizeRumbleId,
|
||||
scrapeRumbleList,
|
||||
scrapeRumbleVideo,
|
||||
} from './providers/rumble.mjs';
|
||||
|
||||
/**
|
||||
* Routes HTTP Rumble — COUCHE MINCE uniquement.
|
||||
*
|
||||
* Tout le scraping (fetch Cloudflare à 3 niveaux, cookie jar __cf_bm, cooldown
|
||||
* global, negative cache, parsing DOM/JSON-LD, vues K/M/B) vit dans
|
||||
* `server/providers/rumble.mjs`, le cœur partagé avec la recherche unifiée :
|
||||
* aucune logique de fetch ni de parse dans ce fichier (l'implémentation
|
||||
* dupliquée — axios + spawn python + parseur de cartes, vues « 1,2K » lues en
|
||||
* 12 — a été supprimée ici). `/search` passe par le registre de providers :
|
||||
* cache SQLite, negative cache et channelRef sont partagés avec `/api/search`.
|
||||
*/
|
||||
|
||||
const router = express.Router();
|
||||
|
||||
/* ----------------------------- Rate limiting ----------------------------- */
|
||||
// 20/min : la valeur vivante historique (l'ancien rumbleLimiter 10/min
|
||||
// d'index.mjs n'était jamais monté — supprimé avec le code mort).
|
||||
const rumbleLimiter = rateLimit({
|
||||
windowMs: 60 * 1000,
|
||||
max: 20,
|
||||
@@ -20,6 +34,9 @@ const rumbleLimiter = rateLimit({
|
||||
router.use(rumbleLimiter);
|
||||
|
||||
/* --------------------------------- Cache -------------------------------- */
|
||||
// Cache positif court : les pages Rumble changent peu et chaque miss coûte un
|
||||
// fetch (+ éventuellement un spawn python). Les ÉCHECS ne sont pas cachés ici :
|
||||
// c'est le negative cache + le cooldown du cœur qui les portent.
|
||||
const cache = new Map();
|
||||
const TTL_MS = 60 * 1000; // 60s
|
||||
|
||||
@@ -36,483 +53,20 @@ function getCache(key) {
|
||||
return hit.data;
|
||||
}
|
||||
|
||||
/* ------------------------------- HTTP GET -------------------------------- */
|
||||
/**
|
||||
* Stratégie best-effort (cf. server/providers/rumble.mjs) :
|
||||
* 1. axios/Node avec headers navigateur (rapide quand CF ne challenge pas) ;
|
||||
* 2. helper python3 + curl_cffi (impersonation Chrome) — indispensable derrière
|
||||
* Cloudflare (les pages /search/* retournent sinon 403 "Just a moment").
|
||||
* Retourne le HTML brut ou lève une erreur si tout est challengé.
|
||||
* Contrat d'erreur : un blocage Cloudflare est un 503 typé, JAMAIS un 404
|
||||
* (le 404 ferait retirer la vidéo du flux Shorts — « introuvable de façon
|
||||
* certaine ») et jamais un 200 vide sans diagnostic. Les erreurs sans statut
|
||||
* (bug inattendu) restent en 500 typé.
|
||||
*/
|
||||
const ROUTER_DIR = path.dirname(fileURLToPath(import.meta.url));
|
||||
const ROUTER_PY_HELPER = path.join(ROUTER_DIR, 'providers', 'rumble_fetch.py');
|
||||
const ROUTER_PY_HELPER_EXISTS = fs.existsSync(ROUTER_DIR) && fs.existsSync(ROUTER_PY_HELPER);
|
||||
|
||||
function isChallengeHtml(html) {
|
||||
return /Just a moment|challenge-platform|cf-chl/i.test(String(html || '').slice(0, 4000));
|
||||
}
|
||||
|
||||
async function pythonFetchHtml(url, { timeoutMs = 20_000 } = {}) {
|
||||
if (!ROUTER_PY_HELPER_EXISTS) return null;
|
||||
for (const py of ['python3', 'python']) {
|
||||
const result = await new Promise((resolve) => {
|
||||
let settled = false;
|
||||
const done = (v) => { if (!settled) { settled = true; resolve(v); } };
|
||||
let child;
|
||||
try {
|
||||
child = spawn(py, [ROUTER_PY_HELPER, url, String(Math.ceil(timeoutMs / 1000))], {
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
});
|
||||
} catch { return done(null); }
|
||||
let stdout = '';
|
||||
let failed = false;
|
||||
const timer = setTimeout(() => { failed = true; try { child.kill(); } catch {} done(null); }, timeoutMs + 5_000);
|
||||
child.stdout.on('data', (d) => { stdout += d.toString(); });
|
||||
child.on('error', () => { clearTimeout(timer); done(null); });
|
||||
child.on('close', (code) => {
|
||||
clearTimeout(timer);
|
||||
if (failed) return;
|
||||
done(code === 0 && stdout.length > 1000 ? stdout : null);
|
||||
});
|
||||
});
|
||||
if (result) return result;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
async function httpGet(url) {
|
||||
// Tentative 1 : axios/Node direct
|
||||
try {
|
||||
const resp = await axios.get(url, {
|
||||
headers: {
|
||||
// UA “desktop” moderne pour minimiser les anti-bot simples
|
||||
'User-Agent':
|
||||
'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
|
||||
'Accept':
|
||||
'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8',
|
||||
'Accept-Language': 'en-US,en;q=0.9',
|
||||
'sec-ch-ua': '"Chromium";v="131", "Not_A Brand";v="24"',
|
||||
'sec-ch-ua-mobile': '?0',
|
||||
'sec-ch-ua-platform': '"Windows"',
|
||||
'sec-fetch-dest': 'document',
|
||||
'sec-fetch-mode': 'navigate',
|
||||
'sec-fetch-site': 'none',
|
||||
'sec-fetch-user': '?1',
|
||||
'upgrade-insecure-requests': '1',
|
||||
},
|
||||
timeout: 15000,
|
||||
// Important: pas de redirects inter-domain hasardeux
|
||||
maxRedirects: 3,
|
||||
validateStatus: s => s >= 200 && s < 400
|
||||
});
|
||||
const html = resp.data;
|
||||
if (typeof html === 'string' && html.length > 1000 && !isChallengeHtml(html)) return html;
|
||||
} catch (e) {
|
||||
// Statuts 403 Cloudflare → on bascule vers le helper python ci-dessous
|
||||
const status = e?.response?.status;
|
||||
if (status && status !== 403) throw e;
|
||||
}
|
||||
// Tentative 2 : python + curl_cffi (impersonation navigateur)
|
||||
const pyHtml = await pythonFetchHtml(url);
|
||||
if (pyHtml) return pyHtml;
|
||||
const err = new Error('Request failed with status code 403');
|
||||
err.status = 403;
|
||||
throw err;
|
||||
}
|
||||
|
||||
/* ------------------------- Utils: normalisation ID ------------------------ */
|
||||
/**
|
||||
* Rumble expose plusieurs formes:
|
||||
* - Page canoniques: https://rumble.com/v6siqxf-some-title.html
|
||||
* - Ancienne forme: https://rumble.com/video/12345
|
||||
* - URL d’embed officielle: https://rumble.com/embed/v6siqxf/
|
||||
* - ID brut attendu: v6siqxf (toujours commence par 'v' + base62)
|
||||
*
|
||||
* Cette fonction accepte: ID ou URL et renvoie { id: 'vXXXX', urlCanonique, embedUrl }
|
||||
*/
|
||||
function normalizeRumbleId(input, { preferEmbed = true } = {}) {
|
||||
if (!input) return null;
|
||||
|
||||
let id = null;
|
||||
let urlCanonique = null;
|
||||
let embedUrl = null;
|
||||
|
||||
// 1) Si on nous donne déjà un ID "vXXXX"
|
||||
const clean = String(input).trim();
|
||||
const mIdOnly = /^v[0-9A-Za-z]+$/.exec(clean);
|
||||
if (mIdOnly) {
|
||||
id = clean;
|
||||
urlCanonique = `https://rumble.com/${id}`;
|
||||
embedUrl = `https://rumble.com/embed/${id}/`;
|
||||
return { id, urlCanonique, embedUrl };
|
||||
}
|
||||
|
||||
// 2) Si on nous donne une URL
|
||||
try {
|
||||
const u = new URL(clean, 'https://rumble.com');
|
||||
// /embed/vXXXX/
|
||||
let m = /\/embed\/(v[0-9A-Za-z]+)/.exec(u.pathname);
|
||||
if (!m) m = /\/(v[0-9A-Za-z]+)(?:[-/.]|$)/.exec(u.pathname);
|
||||
if (!m) {
|
||||
// ancienne forme /video/123 → on ne sait pas convertir de manière fiable
|
||||
const mOld = /\/video\/([0-9A-Za-z]+)/.exec(u.pathname);
|
||||
if (mOld) {
|
||||
// On garde l’URL telle quelle et laissera le parseur de la page extraire le vrai vID.
|
||||
return { id: null, urlCanonique: u.href, embedUrl: null };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
id = m[1];
|
||||
urlCanonique = `https://rumble.com/${id}`;
|
||||
embedUrl = `https://rumble.com/embed/${id}/`;
|
||||
return { id, urlCanonique, embedUrl };
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/* ---------------------- Parsing robuste d’une PAGE vidéo ------------------ */
|
||||
/**
|
||||
* Source d’autorité pour le vrai ID: le JS inline:
|
||||
* Rumble("play", {..., "video":"vXXXX", ...})
|
||||
* On prend ensuite en fallback: <meta property="og:video"> (souvent /embed/vXXXX/)
|
||||
* puis <link rel="canonical"> ou <meta property="og:url"> (contenant /vXXXX-...).
|
||||
*
|
||||
* NB: ce choix est basé sur l’observation publique: la valeur "video":"vXXXX"
|
||||
* est exactement l’ID attendu par l’embed officiel.
|
||||
*/
|
||||
function extractVideoIdentity($) {
|
||||
// 1) Script "Rumble('play', {... "video":"vXXXX" ...})"
|
||||
// On évite d’exécuter quoi que ce soit; simple regex sur tout le HTML.
|
||||
const html = $.html() || '';
|
||||
let m = /Rumble\(\s*["']play["']\s*,\s*{[^}]*["']video["']\s*:\s*["'](v[0-9A-Za-z]+)["']/s.exec(html);
|
||||
if (m && m[1]) {
|
||||
const id = m[1];
|
||||
return {
|
||||
id,
|
||||
embedUrl: `https://rumble.com/embed/${id}/`,
|
||||
urlCanonique: `https://rumble.com/${id}`
|
||||
};
|
||||
}
|
||||
|
||||
// 2) og:video → .../embed/vXXXX/...
|
||||
let embed = $('meta[property="og:video"]').attr('content')
|
||||
|| $('meta[name="twitter:player"]').attr('content');
|
||||
if (embed) {
|
||||
if (embed.startsWith('//')) embed = 'https:' + embed;
|
||||
const mm = /\/embed\/(v[0-9A-Za-z]+)/.exec(embed);
|
||||
if (mm) {
|
||||
const id = mm[1];
|
||||
return { id, embedUrl: `https://rumble.com/embed/${id}/`, urlCanonique: `https://rumble.com/${id}` };
|
||||
}
|
||||
}
|
||||
|
||||
// 3) Canonical / og:url → .../vXXXX-...
|
||||
let canon = $('link[rel="canonical"]').attr('href')
|
||||
|| $('meta[property="og:url"]').attr('content');
|
||||
if (canon) {
|
||||
const mm = /\/(v[0-9A-Za-z]+)(?:[-/.]|$)/.exec(canon);
|
||||
if (mm) {
|
||||
const id = mm[1];
|
||||
return { id, embedUrl: `https://rumble.com/embed/${id}/`, urlCanonique: `https://rumble.com/${id}` };
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
/* -------------------------- Scraper d’une vidéo -------------------------- */
|
||||
async function scrapeRumbleVideo(videoIdOrUrl) {
|
||||
try {
|
||||
// Accepte /:videoId ou une URL complète.
|
||||
let norm = normalizeRumbleId(videoIdOrUrl);
|
||||
const fetchUrl = norm?.urlCanonique || `https://rumble.com/${videoIdOrUrl}`;
|
||||
const html = await httpGet(fetchUrl);
|
||||
const $ = cheerio.load(html);
|
||||
|
||||
// Identité fiable (id + embed + canonique)
|
||||
let ident = extractVideoIdentity($);
|
||||
if (!ident) {
|
||||
// dernier recours: ré-essayer avec la page telle quelle si on est venu via /video/123
|
||||
if (!norm?.id && norm?.urlCanonique) {
|
||||
ident = extractVideoIdentity($);
|
||||
}
|
||||
}
|
||||
if (!ident?.id) {
|
||||
return { error: 'Unable to determine Rumble video ID', input: videoIdOrUrl };
|
||||
}
|
||||
|
||||
// Métadonnées robustes
|
||||
const title =
|
||||
$('h1.video-title, .video-title h1').first().text().trim()
|
||||
|| $('meta[property="og:title"]').attr('content') || 'Untitled Video';
|
||||
|
||||
let thumbnail = $('meta[property="og:image"]').attr('content') || '';
|
||||
if (thumbnail && thumbnail.startsWith('//')) thumbnail = 'https:' + thumbnail;
|
||||
|
||||
const uploaderName =
|
||||
$('.media-by--a, .channel-name, a[href*="/c/"]').first().text().trim() || '';
|
||||
|
||||
const viewsText =
|
||||
$('.rumbles-views, .video-views, .media-view-count, [data-view-count]').first().text().trim() || '';
|
||||
const views = parseInt(viewsText.replace(/[^\d]/g, ''), 10) || 0;
|
||||
|
||||
const duration = parseInt($('meta[property="video:duration"]').attr('content') || '', 10) || 0;
|
||||
|
||||
const uploadedDate =
|
||||
$('meta[property="article:published_time"]').attr('content')
|
||||
|| $('time[datetime]').attr('datetime') || '';
|
||||
|
||||
const description =
|
||||
$('meta[property="og:description"]').attr('content')
|
||||
|| $('meta[name="description"]').attr('content') || '';
|
||||
|
||||
// embedUrl final — toujours la forme officielle
|
||||
const embedUrl = ident.embedUrl;
|
||||
|
||||
return {
|
||||
videoId: ident.id,
|
||||
title,
|
||||
thumbnail,
|
||||
uploaderName,
|
||||
views,
|
||||
duration,
|
||||
uploadedDate,
|
||||
description,
|
||||
url: ident.urlCanonique,
|
||||
embedUrl,
|
||||
type: 'video'
|
||||
};
|
||||
} catch (e) {
|
||||
const msg = (e && e.message) ? e.message : String(e);
|
||||
return { error: `Scraping failed: ${msg}` };
|
||||
}
|
||||
}
|
||||
|
||||
/* ------------------ Scraper de liste (search / browse) ------------------ */
|
||||
function parseDurationToSeconds(text) {
|
||||
if (!text) return 0;
|
||||
// Garde-fou : les attributs datetime (dates de publication) ne sont jamais des durées.
|
||||
if (/^\d{4}-\d{2}-\d{2}/.test(String(text).trim())) return 0;
|
||||
|
||||
// Nettoyer le texte en supprimant les espaces et caractères non numériques inutiles
|
||||
const cleanText = text.trim().replace(/\s+/g, '');
|
||||
|
||||
// Format hh:mm:ss
|
||||
let m = cleanText.match(/^(\d+):(\d{2}):(\d{2})$/);
|
||||
if (m) {
|
||||
const h = parseInt(m[1], 10) || 0;
|
||||
const mn = parseInt(m[2], 10) || 0;
|
||||
const s = parseInt(m[3], 10) || 0;
|
||||
return h * 3600 + mn * 60 + s;
|
||||
}
|
||||
|
||||
// Format mm:ss
|
||||
m = cleanText.match(/^(\d+):(\d{2})$/);
|
||||
if (m) {
|
||||
const mn = parseInt(m[1], 10) || 0;
|
||||
const s = parseInt(m[2], 10) || 0;
|
||||
return mn * 60 + s;
|
||||
}
|
||||
|
||||
// Format avec unités (ex: 1h 30m 45s)
|
||||
m = cleanText.match(/(\d+h)?(\d+m)?(\d+s)?/);
|
||||
if (m) {
|
||||
const hours = m[1] ? parseInt(m[1], 10) : 0;
|
||||
const minutes = m[2] ? parseInt(m[2], 10) : 0;
|
||||
const seconds = m[3] ? parseInt(m[3], 10) : 0;
|
||||
return (hours * 3600) + (minutes * 60) + seconds;
|
||||
}
|
||||
|
||||
// Si on arrive ici, on essaie d'extraire tous les nombres et on suppose un format mmss
|
||||
const numbers = cleanText.match(/\d+/g);
|
||||
if (numbers && numbers.length > 0) {
|
||||
// Si un seul nombre, on suppose que c'est en secondes
|
||||
if (numbers.length === 1) {
|
||||
return parseInt(numbers[0], 10) || 0;
|
||||
}
|
||||
// Si deux nombres, on suppose mm:ss
|
||||
if (numbers.length === 2) {
|
||||
return (parseInt(numbers[0], 10) * 60) + (parseInt(numbers[1], 10) || 0);
|
||||
}
|
||||
// Si trois nombres, on suppose hh:mm:ss
|
||||
if (numbers.length >= 3) {
|
||||
return (parseInt(numbers[0], 10) * 3600) +
|
||||
(parseInt(numbers[1], 10) * 60) +
|
||||
(parseInt(numbers[2], 10) || 0);
|
||||
}
|
||||
}
|
||||
|
||||
return 0; // Par défaut si aucun format n'est reconnu
|
||||
}
|
||||
|
||||
async function scrapeRumbleList({ q, page = 1, limit = 24, sort = 'viral', mode = 'videos' }) {
|
||||
try {
|
||||
// Mode "shorts" : le flux natif rumble.com/shorts (vertical ≤ 90 s).
|
||||
// Les liens sont de la forme /shorts/vXXXX (plus éventuellement ?tracking).
|
||||
const isShortsMode = mode === 'shorts';
|
||||
const url = isShortsMode
|
||||
? `https://rumble.com/shorts?page=${page}`
|
||||
: q
|
||||
? `https://rumble.com/search/video?q=${encodeURIComponent(q)}&page=${page}`
|
||||
: `https://rumble.com/videos?sort=${encodeURIComponent(sort)}&page=${page}`;
|
||||
|
||||
const html = await httpGet(url);
|
||||
const $ = cheerio.load(html);
|
||||
|
||||
const found = [];
|
||||
// 1) Cartes "vidéos" standards (li/article/div)
|
||||
const linkSelector = isShortsMode
|
||||
? 'a[href*="/shorts/v"], a[href^="/v"], a[href^="/video/"]'
|
||||
: 'a[href^="/v"], a[href^="/video/"]';
|
||||
$(linkSelector).each((_, el) => {
|
||||
const href = ($(el).attr('href') || '').split('?')[0];
|
||||
// On préfère STRICTEMENT l’ID /vXXXX (aussi après /shorts/).
|
||||
let m = /\/(v[0-9A-Za-z]+)(?:[-/.]|$)/.exec(href);
|
||||
let id = m?.[1] || null;
|
||||
|
||||
// Fallback minimaliste pour /video/123 → on ne convertit pas ici; on laissera /video/:id passer au détails qui normalise par parse de la page.
|
||||
const isLegacy = !id && /^\/video\//.test(href);
|
||||
|
||||
if (!id && !isLegacy) return;
|
||||
|
||||
const card = $(el).closest('li, article, .video-listing-entry, .video-item, .video-card, div');
|
||||
|
||||
// Titre : préférer le vrai titre de la carte (h3/h2) au texte brut du lien
|
||||
// (le lien contient souvent des badges "N watching" / durées qui polluaient le titre).
|
||||
const cardTitle = card.find('h3, h2, .video-item--title').first().text().replace(/\s+/g, ' ').trim();
|
||||
const linkTitle = ($(el).attr('title') || '').replace(/\s+/g, ' ').trim();
|
||||
const linkText = $(el).text().replace(/\s+/g, ' ').trim();
|
||||
let title = cardTitle || linkTitle || linkText;
|
||||
// Filtrer les faux titres (badges viewers, durées seules, chaînes vides)
|
||||
if (!title || /^\d+\s+watching$/i.test(title) || /^[0-9:.,\s]+$/.test(title) || title.length < 3) {
|
||||
// Dernier recours : attribut alt de l'image de la carte
|
||||
const alt = (card.find('img').first().attr('alt') || '').replace(/\s+/g, ' ').trim();
|
||||
if (alt && alt.length >= 3 && !/^\d+\s+watching$/i.test(alt)) title = alt;
|
||||
else return; // ignorer cette carte (badge live, doublon de lien, …)
|
||||
}
|
||||
|
||||
// Thumb robuste: data-src > src
|
||||
let thumb =
|
||||
card.find('img').first().attr('data-src')
|
||||
|| card.find('img').first().attr('src')
|
||||
|| '';
|
||||
if (thumb && thumb.startsWith('//')) thumb = 'https:' + thumb;
|
||||
|
||||
// Essayer plusieurs sélecteurs pour la durée, y compris les attributs data-
|
||||
// Ajout de plus de sélecteurs spécifiques à Rumble pour la durée
|
||||
const durationElement = card.find(
|
||||
'.video-item--duration, .video-duration, .duration, .video-item__duration, ' +
|
||||
'[data-duration], .videoDuration, .video-time, .time, ' +
|
||||
'.video-card__duration, .media__duration, .thumb-time, ' +
|
||||
'.video-listing-entry__duration, .video-item__duration, time'
|
||||
).first();
|
||||
|
||||
const durationCandidates = [];
|
||||
if (durationElement.length) {
|
||||
durationCandidates.push(
|
||||
durationElement.attr('data-value'), // motif Rumble : <span class="video-item--duration" data-value="3:06:42">
|
||||
durationElement.attr('data-duration'),
|
||||
durationElement.attr('data-time'),
|
||||
durationElement.attr('aria-label'),
|
||||
durationElement.attr('title'),
|
||||
durationElement.text()?.trim()
|
||||
);
|
||||
}
|
||||
|
||||
// Chercher dans le HTML de la carte un motif HH:MM(:SS)
|
||||
try {
|
||||
const htmlSnippet = card.html() || '';
|
||||
const match = />\s*([0-9]+:[0-9]{2}(?::[0-9]{2})?)\s*</.exec(htmlSnippet);
|
||||
if (match && match[1]) durationCandidates.push(match[1]);
|
||||
} catch {}
|
||||
|
||||
let durationSeconds = 0;
|
||||
for (const candidate of durationCandidates) {
|
||||
const parsed = parseDurationToSeconds(candidate);
|
||||
if (parsed > 0) {
|
||||
durationSeconds = parsed;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Extraire les vues
|
||||
const viewsText =
|
||||
card.find('.video-item--views, .rumbles-views, .views, .video-item__views, [data-views]').first()
|
||||
.attr('data-views') ||
|
||||
card.find('.video-item--views, .rumbles-views, .views, .video-item__views, .video-views').first().text().trim();
|
||||
|
||||
const views = parseInt((viewsText || '').replace(/[^\d]/g, ''), 10) || 0;
|
||||
|
||||
// Important: on renvoie TOUJOURS une URL canonique cohérente
|
||||
let url = null;
|
||||
let videoId = null;
|
||||
|
||||
if (id) {
|
||||
videoId = id;
|
||||
url = `https://rumble.com/${id}`;
|
||||
} else if (isLegacy) {
|
||||
// Laisse l’endpoint /video/:slug gérer la normalisation
|
||||
videoId = href.replace(/^\//, ''); // "video/123..."
|
||||
url = `https://rumble.com/${videoId}`;
|
||||
}
|
||||
|
||||
// Filtrage doublons par videoId (id ou "video/123...")
|
||||
const key = videoId;
|
||||
// Uploader : meilleur effort (nom de chaîne sous la carte)
|
||||
const uploaderName = card.find('.video-item--channel, .channel-name, a[href^="/c/"], a[href^="/user/"]').first().text().replace(/\s+/g, ' ').trim() || '';
|
||||
found.push({
|
||||
videoId: key,
|
||||
title,
|
||||
thumbnail: thumb,
|
||||
uploaderName,
|
||||
views,
|
||||
duration: durationSeconds,
|
||||
// Indicateur natif pour le front. En mode "shorts", tout ce qui sort
|
||||
// du flux rumble.com/shorts (vertical ≤ 90 s) est un Short, même si
|
||||
// le badge de durée est absent du HTML.
|
||||
// (Rumble n'expose pas de flag natif ; la durée reste la source de vérité.)
|
||||
isShort: isShortsMode ? (durationSeconds <= 0 || durationSeconds <= 90) : (durationSeconds > 0 && durationSeconds <= 75),
|
||||
uploadedDate: '',
|
||||
url,
|
||||
type: 'video'
|
||||
});
|
||||
});
|
||||
|
||||
// De-dupe
|
||||
const seen = new Set();
|
||||
const unique = [];
|
||||
for (const it of found) {
|
||||
if (!it.videoId) continue;
|
||||
if (seen.has(it.videoId)) continue;
|
||||
seen.add(it.videoId);
|
||||
unique.push(it);
|
||||
}
|
||||
|
||||
// Limite + nextCursor (page-based)
|
||||
const list = unique.slice(0, limit);
|
||||
const nextCursor = list.length === limit ? String(Number(page) + 1) : null;
|
||||
|
||||
return {
|
||||
items: list,
|
||||
total: unique.length,
|
||||
page: Number(page),
|
||||
limit: Number(limit),
|
||||
nextCursor
|
||||
};
|
||||
} catch (e) {
|
||||
return {
|
||||
items: [],
|
||||
total: 0,
|
||||
page: Number(page),
|
||||
limit: Number(limit),
|
||||
nextCursor: null,
|
||||
error: (e && e.message) ? e.message : String(e)
|
||||
};
|
||||
}
|
||||
function sendError(res, e) {
|
||||
const status = Number(e?.status) || 500;
|
||||
return res.status(status).json({ error: e?.code || 'rumble_failed' });
|
||||
}
|
||||
|
||||
/* --------------------------------- Routes -------------------------------- */
|
||||
|
||||
// Tendances / classements : https://rumble.com/videos?sort=…
|
||||
router.get('/browse', async (req, res) => {
|
||||
const page = Math.max(1, parseInt(String(req.query.page || '1'), 10) || 1);
|
||||
const limit = Math.min(50, Math.max(1, parseInt(String(req.query.limit || '24'), 10) || 24));
|
||||
@@ -522,16 +76,19 @@ router.get('/browse', async (req, res) => {
|
||||
const cached = getCache(key);
|
||||
if (cached) return res.json(cached);
|
||||
|
||||
const data = await scrapeRumbleList({ page, limit, sort });
|
||||
setCache(key, data);
|
||||
return res.json(data);
|
||||
try {
|
||||
const data = await scrapeRumbleList({ page, limit, sort });
|
||||
setCache(key, data);
|
||||
return res.json(data);
|
||||
} catch (e) {
|
||||
return sendError(res, e);
|
||||
}
|
||||
});
|
||||
|
||||
/* ---------------- Route "shorts" natifs (rumble.com/shorts) -------------- */
|
||||
/**
|
||||
* GET /api/rumble/shorts?page=&limit=
|
||||
* Scrape le flux natif des Shorts Rumble (vertical ≤ 90 s) : IDs /vXXXX
|
||||
* propres, compatibles avec l'embed officiel /embed/vXXXX/.
|
||||
* Flux natif des Shorts Rumble (rumble.com/shorts, vertical ≤ 90 s) :
|
||||
* IDs /vXXXX propres, compatibles avec l'embed officiel /embed/vXXXX/.
|
||||
*/
|
||||
router.get('/shorts', async (req, res) => {
|
||||
const page = Math.max(1, parseInt(String(req.query.page || '1'), 10) || 1);
|
||||
@@ -541,11 +98,17 @@ router.get('/shorts', async (req, res) => {
|
||||
const cached = getCache(key);
|
||||
if (cached) return res.json(cached);
|
||||
|
||||
const data = await scrapeRumbleList({ page, limit, mode: 'shorts' });
|
||||
setCache(key, data);
|
||||
return res.json(data);
|
||||
try {
|
||||
const data = await scrapeRumbleList({ page, limit, mode: 'shorts' });
|
||||
setCache(key, data);
|
||||
return res.json(data);
|
||||
} catch (e) {
|
||||
return sendError(res, e);
|
||||
}
|
||||
});
|
||||
|
||||
// Recherche : même adaptateur (et donc même cache SQLite, negative cache et
|
||||
// channelRef) que le groupe `ru` de la recherche unifiée.
|
||||
router.get('/search', async (req, res) => {
|
||||
const q = String(req.query.q || '').trim();
|
||||
if (!q) return res.status(400).json({ error: 'Query parameter required' });
|
||||
@@ -563,12 +126,25 @@ router.get('/search', async (req, res) => {
|
||||
const cached = getCache(key);
|
||||
if (cached) return res.json(cached);
|
||||
|
||||
const data = await scrapeRumbleList({ q, page, limit });
|
||||
setCache(key, data);
|
||||
return res.json(data);
|
||||
try {
|
||||
const items = await getProviderAdapter('ru').search(q, { limit, page });
|
||||
const data = {
|
||||
items,
|
||||
total: items.length,
|
||||
page,
|
||||
limit,
|
||||
nextCursor: items.length === limit ? String(page + 1) : null,
|
||||
};
|
||||
setCache(key, data);
|
||||
return res.json(data);
|
||||
} catch (e) {
|
||||
return sendError(res, e);
|
||||
}
|
||||
});
|
||||
|
||||
// Endpoint details. Accepte :videoId pouvant être "vXXXX" OU "video/123..."
|
||||
// Détails d'une vidéo (résolution d'embed fiable pour /watch et /shorts).
|
||||
// Accepte :videoId pouvant être "vXXXX" OU "video/123..." — l'ancien pattern
|
||||
// :videoId(*) est conservé tel quel (comportement de route inchangé).
|
||||
router.get('/video/:videoId(*)', async (req, res) => {
|
||||
try {
|
||||
const raw = String(req.params.videoId);
|
||||
@@ -576,40 +152,36 @@ router.get('/video/:videoId(*)', async (req, res) => {
|
||||
const cached = getCache(key);
|
||||
if (cached) return res.json(cached);
|
||||
|
||||
// Normalise au maximum avant scrape
|
||||
const norm = normalizeRumbleId(raw) || { urlCanonique: `https://rumble.com/${raw}` };
|
||||
const data = await scrapeRumbleVideo(norm.id || norm.urlCanonique);
|
||||
if (data.error) return res.status(404).json(data);
|
||||
|
||||
setCache(key, data);
|
||||
return res.json(data);
|
||||
} catch (error) {
|
||||
return res.status(500).json({ error: 'Failed to scrape video' });
|
||||
} catch (e) {
|
||||
return sendError(res, e);
|
||||
}
|
||||
});
|
||||
|
||||
/* ----------------- Option: “prélecteur” sans pub (non-embed) -------------- */
|
||||
/**
|
||||
* On NE désactive PAS les pubs côté Rumble (pas de param officiel fiable).
|
||||
* Mais on peut servir un “preplay”:
|
||||
* - On affiche miniature/titre.
|
||||
* - Au clic: (A) ouvrir dans Rumble (UX la plus propre), ou (B) injecter l’iframe
|
||||
* officiellement (ce qui déclenchera leur logique pub).
|
||||
* Cette route renvoie juste les meta nécessaires pour ce composant prélecteur.
|
||||
* Option : « prélecteur » sans pub (non-embed).
|
||||
* On NE désactive PAS les pubs côté Rumble (pas de paramètre officiel fiable) :
|
||||
* cette route renvoie juste les métadonnées du composant prélecteur
|
||||
* (miniature/titre + lien d'ouverture fournisseur, iframe en dernier recours).
|
||||
*/
|
||||
router.get('/video/:videoId/preplay', async (req, res) => {
|
||||
const raw = String(req.params.videoId);
|
||||
const norm = normalizeRumbleId(raw) || { urlCanonique: `https://rumble.com/${raw}` };
|
||||
const data = await scrapeRumbleVideo(norm.id || norm.urlCanonique);
|
||||
if (data.error) return res.status(404).json(data);
|
||||
const preplay = {
|
||||
videoId: data.videoId,
|
||||
title: data.title,
|
||||
thumbnail: data.thumbnail,
|
||||
rumbleUrl: data.url, // bouton "Ouvrir sur le site du fournisseur"
|
||||
embedUrl: data.embedUrl // injection différée si l’utilisateur insiste pour lire ici
|
||||
};
|
||||
res.json(preplay);
|
||||
try {
|
||||
const raw = String(req.params.videoId);
|
||||
const norm = normalizeRumbleId(raw) || { urlCanonique: `https://rumble.com/${raw}` };
|
||||
const data = await scrapeRumbleVideo(norm.id || norm.urlCanonique);
|
||||
return res.json({
|
||||
videoId: data.videoId,
|
||||
title: data.title,
|
||||
thumbnail: data.thumbnail,
|
||||
rumbleUrl: data.url, // bouton "Ouvrir sur le site du fournisseur"
|
||||
embedUrl: data.embedUrl // injection différée si l'utilisateur insiste
|
||||
});
|
||||
} catch (e) {
|
||||
return sendError(res, e);
|
||||
}
|
||||
});
|
||||
|
||||
export default router;
|
||||
|
||||
Reference in New Issue
Block a user