diff --git a/README.md b/README.md index fc0fead..c34adce 100644 --- a/README.md +++ b/README.md @@ -43,6 +43,8 @@ Agrégez, explorez et regardez des vidéos depuis \* Playlists = intégration locale NewTube (création/gestion); la synchro native dépend de l’API publique de chaque fournisseur. +* **Rumble 🟠** : scraping maison derrière Cloudflare (fetch navigateur + helper `curl_cffi`). Si le réseau du serveur est challengé, l’API répond `503 { error: "rumble_cloudflare_challenge" }` (« bloqué temporairement ») au lieu d’un résultat vide : Rumble peut donc être temporairement absent selon le réseau — voir `server/providers/rumble.mjs`. + --- ## 🔎 Recherche unifiée (multi-providers) diff --git a/docs/API_MCP_GUIDE.md b/docs/API_MCP_GUIDE.md index e60f509..241258d 100644 --- a/docs/API_MCP_GUIDE.md +++ b/docs/API_MCP_GUIDE.md @@ -126,7 +126,7 @@ curl "http://localhost:4000/api/details/youtube/dQw4w9WgXcQ" | jq '{title,upload Tendances YT sans clé (scrape). Phase 1 : `yt` uniquement (`400` sinon), fallback `{ items:[] }` jamais 500. ### 5.5 Rumble dédié (`/api/rumble/…`) -`GET /browse`, `GET /search?q=…`, `GET /video/:videoId`, `GET /video/:videoId/preplay`. Rate-limit 10/min. +`GET /browse?page=&limit=&sort=`, `GET /shorts?page=&limit=`, `GET /search?q=…&page=|offset=`, `GET /video/:videoId`, `GET /video/:videoId/preplay`. Rate-limit 20/min. Le scraping vit dans `server/providers/rumble.mjs` (cœur partagé avec `/api/search` : cache SQLite, negative cache, cookie jar Cloudflare) ; `/search` passe par le registre de providers. Réponses : `{ items, total, page, limit, nextCursor }` (liste), `{ videoId, title, …, embedUrl }` (vidéo). Erreurs : `503 { error:"rumble_cloudflare_challenge" }` = blocage Cloudflare (cooldown global 60 s côté serveur), `404 { error:"rumble_video_not_found" }` = vidéo réellement absente, `400 { error:"Query parameter required" }` (search sans q). ## 6. Référence REST — transcript détaillé @@ -208,7 +208,7 @@ Statut `GET /api/oauth/status`, URL `GET /api/oauth/:provider/url` (JWT), callba | YouTube 429 épuisé | 502 | `{ available:false, error:"transcript_temporarily_unavailable", retryable:true }` | | Erreur interne | 500 | `{ error:"search_failed"|"details_failed", details }` | -Seaux : login 5/min, suggest 60/min, transcript 10/min, downloads lecture 120 / écriture 30 / formats 30, AI 10/min, Rumble 10/min. +Seaux : login 5/min, suggest 60/min, transcript 10/min, downloads lecture 120 / écriture 30 / formats 30, AI 10/min, Rumble 20/min. ## 11. Serveur MCP — installation diff --git a/server/index.mjs b/server/index.mjs index 564f9e3..dc8a679 100644 --- a/server/index.mjs +++ b/server/index.mjs @@ -12,7 +12,6 @@ import { execFile as execFileCb } from 'node:child_process'; import { promisify } from 'node:util'; import { fileURLToPath as serverFileURLToPath } from 'node:url'; import ffmpegPath from 'ffmpeg-static'; -import * as cheerio from 'cheerio'; import axios from 'axios'; import rumbleRouter from './rumble.mjs'; import { providerRegistry, getProviderAdapter, validateProviders, SUGGESTION_CONTRACT_VERSION } from './providers/registry.mjs'; @@ -1140,15 +1139,6 @@ function formatsCacheSet(key, data) { } } -// Rate limiter for Rumble scraping to prevent being blocked -const rumbleLimiter = rateLimit({ - windowMs: 60 * 1000, // 1 min - max: 10, // Limit to 10 requests per minute per IP - standardHeaders: true, - legacyHeaders: false, - message: { error: 'Too many requests to Rumble API. Please try again later.' } -}); - function makeAccessToken(userId, sessionId) { const payload = { sub: userId, sid: sessionId }; return jwt.sign(payload, JWT_SECRET, { expiresIn: `${ACCESS_TTL_MIN}m` }); @@ -1300,198 +1290,6 @@ function providerUrlFrom(provider, videoId, { instance, slug, sourceUrl }) { } } -async function scrapeRumbleVideo(videoId) { - const url = `https://rumble.com/${videoId}`; - - try { - const response = await axios.get(url, { - headers: { - 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36', - 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8', - 'Accept-Language': 'en-US,en;q=0.5', - 'Accept-Encoding': 'gzip, deflate, br', - 'DNT': '1', - 'Connection': 'keep-alive', - 'Upgrade-Insecure-Requests': '1', - 'Cache-Control': 'no-cache' - }, - timeout: 15000, - maxRedirects: 5, - validateStatus: function (status) { - return status >= 200 && status < 400; // Accept redirects - } - }); - - const $ = cheerio.load(response.data); - const html = response.data; - - // Extract basic video information - let title = ''; - let thumbnail = ''; - let uploaderName = ''; - let uploaderAvatar = ''; - let views = 0; - let duration = 0; - let uploadedDate = ''; - let description = ''; - - // Try multiple selectors for title (Rumble's HTML structure can vary) - title = $('h1.video-title, .video-title h1, [data-video-title]').first().text().trim() || - $('meta[property="og:title"]').attr('content') || - $('title').text().trim() || - $('h1').first().text().trim() || ''; - - // Clean up title (remove site name if present) - title = title.replace(/\s*\|\s*Rumble$/i, '').trim(); - - // Extract thumbnail with fallbacks - thumbnail = $('meta[property="og:image"], meta[name="twitter:image"]').attr('content') || - $('meta[property="og:image:secure_url"]').attr('content') || - $('.video-thumbnail img, .thumbnail img').attr('src') || ''; - - // Make thumbnail URL absolute if relative - if (thumbnail && !thumbnail.startsWith('http')) { - thumbnail = thumbnail.startsWith('//') ? 'https:' + thumbnail : 'https://rumble.com' + thumbnail; - } - - // Extract uploader information with multiple selectors - uploaderName = $('.media-by--a, .channel-name, .uploader-name').first().text().trim() || - $('meta[property="article:author"]').attr('content') || - $('.author-name, .channel-link').first().text().trim() || ''; - - uploaderAvatar = $('.channel-avatar img, .uploader-avatar img').attr('src') || - $('.author-avatar img').attr('src') || ''; - - // Make uploader avatar URL absolute - if (uploaderAvatar && !uploaderAvatar.startsWith('http')) { - uploaderAvatar = uploaderAvatar.startsWith('//') ? 'https:' + uploaderAvatar : 'https://rumble.com' + uploaderAvatar; - } - - // Extract views with better parsing - const viewsText = $('.rumbles-views, .video-views, .views-count, .video-info .views').first().text().trim(); - if (viewsText) { - const viewsMatch = viewsText.match(/([\d,]+(?:\.\d+)?)\s*(K|M|B)?/i); - if (viewsMatch) { - let num = parseFloat(viewsMatch[1].replace(/,/g, '')); - const multiplier = viewsMatch[2]?.toUpperCase(); - if (multiplier === 'K') num *= 1000; - else if (multiplier === 'M') num *= 1000000; - else if (multiplier === 'B') num *= 1000000000; - views = Math.floor(num); - } else { - // Try direct number parsing - const directMatch = viewsText.match(/(\d+(?:,\d+)*)/); - if (directMatch) { - views = parseInt(directMatch[1].replace(/,/g, '')); - } - } - } - - // Extract duration with improved parsing - const durationText = $('meta[property="video:duration"]').attr('content') || - $('video').attr('duration') || - $('.video-duration, .duration, .video-time').first().text().trim() || - $('.time-duration').text().trim(); - - if (durationText) { - if (!isNaN(durationText)) { - duration = parseInt(durationText); - } else { - // Parse various duration formats - const timeMatch = durationText.match(/(\d+):(\d+)(?::(\d+))?/); - if (timeMatch) { - const hours = parseInt(timeMatch[3] || '0'); - const minutes = parseInt(timeMatch[1]); - const seconds = parseInt(timeMatch[2]); - duration = hours * 3600 + minutes * 60 + seconds; - } else { - // Try HH:MM:SS format or MM:SS - const parts = durationText.split(':').map(p => parseInt(p.trim()) || 0); - if (parts.length === 3) { - duration = parts[0] * 3600 + parts[1] * 60 + parts[2]; - } else if (parts.length === 2) { - duration = parts[0] * 60 + parts[1]; - } - } - } - } - - // Extract upload date - uploadedDate = $('meta[property="article:published_time"]').attr('content') || - $('.upload-date, .published-date, .video-date').first().text().trim() || ''; - - // Try to parse relative dates - if (!uploadedDate || uploadedDate.includes('ago')) { - const relativeDate = $('.upload-date, .published-date').first().text().trim(); - if (relativeDate && relativeDate.includes('ago')) { - // Convert relative date to ISO string (simple conversion) - const now = new Date(); - if (relativeDate.includes('hour')) { - const hours = parseInt(relativeDate.match(/(\d+)/)?.[1] || '1'); - now.setHours(now.getHours() - hours); - uploadedDate = now.toISOString(); - } else if (relativeDate.includes('day')) { - const days = parseInt(relativeDate.match(/(\d+)/)?.[1] || '1'); - now.setDate(now.getDate() - days); - uploadedDate = now.toISOString(); - } - } - } - - // Extract description - description = $('meta[property="og:description"]').attr('content') || - $('.video-description, .description, .video-summary').first().text().trim() || ''; - - // Extract video ID from various sources - let extractedVideoId = videoId; - const videoIdMatch = html.match(/"video_id"\s*:\s*"([^"]+)"/) || - html.match(/video[_-]id["\s:]+([^"\s]+)/) || - html.match(/embed\/([^/?]+)/); - if (videoIdMatch && videoIdMatch[1]) { - extractedVideoId = videoIdMatch[1]; - } - - // Validate extracted data - const isValidVideo = title || thumbnail || uploaderName; - - return { - videoId: extractedVideoId, - title: title || 'Untitled Video', - thumbnail, - uploaderName: uploaderName || 'Unknown Uploader', - uploaderAvatar: uploaderAvatar || thumbnail, - views: Math.max(0, views), - duration: Math.max(0, duration), - uploadedDate: uploadedDate || new Date().toISOString(), - description, - url, - type: 'video', - scraped: true, - confidence: isValidVideo ? 'high' : 'low' - }; - } catch (error) { - console.error('Erreur scraping Rumble:', error.message); - - // Return minimal data for fallback with error info - return { - videoId, - title: 'Video unavailable', - thumbnail: '', - uploaderName: 'Unknown', - uploaderAvatar: '', - views: 0, - duration: 0, - uploadedDate: '', - description: '', - url, - type: 'video', - error: error.message, - scraped: false, - confidence: 'none' - }; - } -} - function guessContentTypeByExt(ext) { const e = String(ext || '').toLowerCase(); if (e === 'mp4' || e === 'm4v') return 'video/mp4'; diff --git a/server/providers/rumble.mjs b/server/providers/rumble.mjs index 85b5116..a02f48e 100644 --- a/server/providers/rumble.mjs +++ b/server/providers/rumble.mjs @@ -117,14 +117,32 @@ async function pythonFetch(url, { timeoutMs = 20_000 } = {}) { return null; } +/** + * Cooldown global après un échec total de fetch (phase « ménage Rumble »). + * Sans lui, chaque route (/browse, /shorts, /video, /search) repaie le cycle + * complet Node (12 s max) + spawns python (~20 s) pendant TOUT un blocage + * Cloudflare — c'était 12-20 s × N à chaque page chargée, alors que le + * negative cache ne couvrait que les clés de recherche de l'unifiée. + * ponytail: cooldown fixe 60 s, pas de backoff exponentiel — si le trafic + * monte : daemon python persistant + backoff. + */ +const FETCH_COOLDOWN_MS = Number(process.env.RUMBLE_FETCH_COOLDOWN_MS || 60_000); +let fetchCooldownUntil = 0; +export function armRumbleFetchCooldown(ms = FETCH_COOLDOWN_MS) { + fetchCooldownUntil = Date.now() + (Number(ms) > 0 ? Number(ms) : FETCH_COOLDOWN_MS); +} +export function resetRumbleFetchCooldown() { fetchCooldownUntil = 0; } + /** Node fetch first, python/curl_cffi fallback. */ async function fetchHtml(url) { + if (Date.now() < fetchCooldownUntil) return null; try { const r = await nodeFetch(url); if (r.status === 200 && !isChallenge(r.html)) return r; } catch { /* fallback */ } const py = await pythonFetch(url); if (py) return py; + armRumbleFetchCooldown(); // échec total : quasi certainement un blocage CF return null; } @@ -382,26 +400,350 @@ export function parseSearchHtml(html, { limit = 50 } = {}) { // durée ≤75 s) tranche. Le scraper HTML n'expose pas de dimensions. }); }); - // Phase 3.1 — surcouche JSON-LD : complète ce que le DOM n'expose pas - // (date de publication, vues, auteur) SANS jamais écraser une valeur réellement - // présente dans la liste. Le DOM reste la source primaire, le JSON-LD le filet. + patchItemsFromJsonLd(items, html); + return items; +} + +/* ================= Cœur partagé (routeur /api/rumble + recherche) ========= */ + +/** + * Erreur typée : un échec Cloudflare n'est JAMAIS un résultat vide (phase 3.4). + * `status` pilote le code HTTP côté routeur : 503 = « bloqué, à réessayer », + * 404 = « réellement introuvable » (le seul cas où le flux Shorts retire une + * vidéo, cf. watch-short.component.ts). + */ +function rumbleFail(code, status, { challenged = true } = {}) { + return Object.assign(new Error(code), { code, status, rumbleChallenged: challenged }); +} + +/** + * Surcouche JSON-LD partagée (phase 3.1) : complète ce que le DOM n'expose pas + * (date de publication, vues, auteur) SANS jamais écraser une valeur réellement + * présente dans la liste. Le DOM reste la source primaire, le JSON-LD le filet. + * @param {Array} items items au format Suggestion (mutés en place) + * @param {string} html page d'origine (pour y relire les blocs ld+json) + */ +function patchItemsFromJsonLd(items, html) { try { const ld = parseJsonLd(html); - if (ld.size > 0) { - for (const it of items) { - const patch = ld.get(urlKey(it.url)) || ld.get(urlKey(it.id)); - if (!patch) continue; - if (it.publishedAt === undefined && patch.publishedAt) it.publishedAt = patch.publishedAt; - if (it.views === undefined && patch.views) it.views = patch.views; - if (it.duration === undefined && patch.duration) it.duration = patch.duration; - if (it.thumbnail === undefined && patch.thumbnail) it.thumbnail = patch.thumbnail; - if (it.channelId === undefined && patch.channelId) it.channelId = patch.channelId; - } + if (ld.size === 0) return items; + for (const it of items) { + const patch = ld.get(urlKey(it.url)) || ld.get(urlKey(it.id)); + if (!patch) continue; + if (it.publishedAt === undefined && patch.publishedAt) it.publishedAt = patch.publishedAt; + if (it.views === undefined && patch.views) it.views = patch.views; + if (it.duration === undefined && patch.duration) it.duration = patch.duration; + if (it.thumbnail === undefined && patch.thumbnail) it.thumbnail = patch.thumbnail; + if (it.channelId === undefined && patch.channelId) it.channelId = patch.channelId; } } catch { /* la surcouche est un bonus, jamais un motif d'échec */ } return items; } +/* ------------------------- Utils: normalisation ID ------------------------ */ +/** + * Rumble expose plusieurs formes: + * - Page canoniques: https://rumble.com/v6siqxf-some-title.html + * - Ancienne forme: https://rumble.com/video/12345 + * - URL d’embed officielle: https://rumble.com/embed/v6siqxf/ + * - ID brut attendu: v6siqxf (toujours commence par 'v' + base62) + * + * Accepte ID ou URL et renvoie { id: 'vXXXX', urlCanonique, embedUrl } + * (id null + urlCanonique pour l'ancienne forme /video/123 : la page porte + * le vrai identifiant, c'est `extractVideoIdentity` qui le lira.) + */ +export function normalizeRumbleId(input) { + if (!input) return null; + + // 1) Si on nous donne déjà un ID "vXXXX" + const clean = String(input).trim(); + if (/^v[0-9A-Za-z]+$/.test(clean)) { + return { id: clean, urlCanonique: `https://rumble.com/${clean}`, embedUrl: `https://rumble.com/embed/${clean}/` }; + } + + // 2) Si on nous donne une URL + try { + const u = new URL(clean, 'https://rumble.com'); + // /embed/vXXXX/ + let m = /\/embed\/(v[0-9A-Za-z]+)/.exec(u.pathname); + if (!m) m = /\/(v[0-9A-Za-z]+)(?:[-/.]|$)/.exec(u.pathname); + if (!m) { + // ancienne forme /video/123 → on ne peut pas convertir de manière fiable + if (/\/video\/([0-9A-Za-z]+)/.test(u.pathname)) return { id: null, urlCanonique: u.href, embedUrl: null }; + return null; + } + const id = m[1]; + return { id, urlCanonique: `https://rumble.com/${id}`, embedUrl: `https://rumble.com/embed/${id}/` }; + } catch { + return null; + } +} + +/* ---------------------- Parsing robuste d’une PAGE vidéo ------------------ */ +/** + * Source d’autorité pour le vrai ID: le JS inline: + * Rumble("play", {..., "video":"vXXXX", ...}) + * Puis fallback: (souvent /embed/vXXXX/), + * puis / (contenant /vXXXX-...). + */ +function extractVideoIdentity($) { + // 1) Script "Rumble('play', {... "video":"vXXXX" ...})" — regex, jamais d'exécution. + const html = $.html() || ''; + const m = /Rumble\(\s*["']play["']\s*,\s*{[^}]*["']video["']\s*:\s*["'](v[0-9A-Za-z]+)["']/s.exec(html); + if (m && m[1]) { + const id = m[1]; + return { id, embedUrl: `https://rumble.com/embed/${id}/`, urlCanonique: `https://rumble.com/${id}` }; + } + + // 2) og:video → .../embed/vXXXX/... + let embed = $('meta[property="og:video"]').attr('content') + || $('meta[name="twitter:player"]').attr('content'); + if (embed) { + if (embed.startsWith('//')) embed = 'https:' + embed; + const mm = /\/embed\/(v[0-9A-Za-z]+)/.exec(embed); + if (mm) { + const id = mm[1]; + return { id, embedUrl: `https://rumble.com/embed/${id}/`, urlCanonique: `https://rumble.com/${id}` }; + } + } + + // 3) Canonical / og:url → .../vXXXX-... + const canon = $('link[rel="canonical"]').attr('href') + || $('meta[property="og:url"]').attr('content'); + if (canon) { + const mm = /\/(v[0-9A-Za-z]+)(?:[-/.]|$)/.exec(canon); + if (mm) { + const id = mm[1]; + return { id, embedUrl: `https://rumble.com/embed/${id}/`, urlCanonique: `https://rumble.com/${id}` }; + } + } + + return null; +} + +/* -------------------------- Scraper d’une vidéo -------------------------- */ +/** + * Accepte un id (vXXXX), une URL canonique ou l'ancienne forme /video/123. + * + * Contrat d'erreur (ménage Rumble) : + * - fetch total échoué / cooldown CF → throw 503 `rumble_cloudflare_challenge`. + * Un blocage ne doit JAMAIS se transformer en 404 : le 404 fait retirer la + * vidéo du flux Shorts (« introuvable de façon certaine »), alors qu'un + * challenge Cloudflare ne dit rien sur la vidéo. + * - page reçue mais identité introuvable → throw 404 `rumble_video_not_found`. + */ +export async function scrapeRumbleVideo(videoIdOrUrl) { + const norm = normalizeRumbleId(videoIdOrUrl); + const fetchUrl = norm?.urlCanonique || `https://rumble.com/${videoIdOrUrl}`; + const r = await fetchHtml(fetchUrl); + if (!r || isChallenge(r.html)) throw rumbleFail('rumble_cloudflare_challenge', 503); + + const $ = load(r.html); + const ident = extractVideoIdentity($); + if (!ident?.id) throw rumbleFail('rumble_video_not_found', 404, { challenged: false }); + + const title = + $('h1.video-title, .video-title h1').first().text().trim() + || $('meta[property="og:title"]').attr('content') || 'Untitled Video'; + + let thumbnail = $('meta[property="og:image"]').attr('content') || ''; + if (thumbnail && thumbnail.startsWith('//')) thumbnail = 'https:' + thumbnail; + + const uploaderName = + $('.media-by--a, .channel-name, a[href*="/c/"]').first().text().trim() || ''; + + // Vues : parseRumbleViews (K/M/B, espaces insécables). L'ancien parseur + // `.replace(/[^\d]/g,'')` lisait « 1,2K » en 12 ; undefined = absent, et le + // front conserve alors le compteur venu de la liste. + const viewsText = + $('.rumbles-views, .video-views, .media-view-count, [data-view-count]').first().text().trim() || ''; + const views = parseRumbleViews(viewsText); + + const duration = parseInt($('meta[property="video:duration"]').attr('content') || '', 10) || 0; + + const uploadedDate = + $('meta[property="article:published_time"]').attr('content') + || $('time[datetime]').attr('datetime') || ''; + + const description = + $('meta[property="og:description"]').attr('content') + || $('meta[name="description"]').attr('content') || ''; + + return { + videoId: ident.id, + title, + thumbnail, + uploaderName, + views, + duration, + uploadedDate, + description, + url: ident.urlCanonique, + embedUrl: ident.embedUrl, // toujours la forme officielle /embed/vXXXX/ + type: 'video', + }; +} + +/* ------------------- Scraper de liste (browse / shorts) ------------------ */ +/** + * Scan générique de cartes, pour les pages qui ne sont PAS une recherche + * (/videos?sort=… et le flux natif /shorts, dont les liens sont /shorts/vXXXX). + * Aucune structure de balises n'est supposée : on part des liens /vXXXX et on + * remonte au conteneur le plus proche — c'est ce qui rend ces pages lisibles + * quand le markup diffère de `li.video-listing-entry`. La recherche passe par + * `parseSearchHtml` (structurel + JSON-LD), via le registre. + * + * @param {string} html + * @param {{ mode?: 'browse'|'shorts' }} [opts] + * @returns {Array} items non bornés (le dédoublonnage et la limite sont + * appliqués par `scrapeRumbleList`, pour ne jamais couper avant dédup). + */ +export function parseCardsHtml(html, { mode = 'browse' } = {}) { + const $ = load(html); + const isShortsMode = mode === 'shorts'; + const found = []; + const linkSelector = isShortsMode + ? 'a[href*="/shorts/v"], a[href^="/v"], a[href^="/video/"]' + : 'a[href^="/v"], a[href^="/video/"]'; + $(linkSelector).each((_idx, el) => { + const href = ($(el).attr('href') || '').split('?')[0]; + // On préfère STRICTEMENT l’ID /vXXXX (aussi après /shorts/). + const m = /\/(v[0-9A-Za-z]+)(?:[-/.]|$)/.exec(href); + const id = m?.[1] || null; + // Fallback /video/123 : on ne convertit pas ici, /video/:id normalise par parse de page. + const isLegacy = !id && /^\/video\//.test(href); + if (!id && !isLegacy) return; + + const card = $(el).closest('li, article, .video-listing-entry, .video-item, .video-card, div'); + + // Titre : préférer le vrai titre de la carte (h3/h2) au texte brut du lien + // (le lien contient souvent des badges "N watching" / durées qui polluaient le titre). + const cardTitle = card.find('h3, h2, .video-item--title').first().text().replace(/\s+/g, ' ').trim(); + const linkTitle = ($(el).attr('title') || '').replace(/\s+/g, ' ').trim(); + const linkText = $(el).text().replace(/\s+/g, ' ').trim(); + let title = cardTitle || linkTitle || linkText; + // Filtrer les faux titres (badges viewers, durées seules, chaînes vides) + if (!title || /^\d+\s+watching$/i.test(title) || /^[0-9:.,\s]+$/.test(title) || title.length < 3) { + // Dernier recours : attribut alt de l'image de la carte + const alt = (card.find('img').first().attr('alt') || '').replace(/\s+/g, ' ').trim(); + if (alt && alt.length >= 3 && !/^\d+\s+watching$/i.test(alt)) title = alt; + else return; // ignorer cette carte (badge live, doublon de lien, …) + } + + const thumb = normalizeThumb(card.find('img').first().attr('data-src') || card.find('img').first().attr('src') || ''); + + // Durée : data-value (motif Rumble ), + // data-duration/time/aria-label/title/texte, puis motif HH:MM(:SS) dans le HTML de la carte. + const durationElement = card.find( + '.video-item--duration, .video-duration, .duration, .video-item__duration, ' + + '[data-duration], .videoDuration, .video-time, .time, ' + + '.video-card__duration, .media__duration, .thumb-time, ' + + '.video-listing-entry__duration, .video-item__duration, time' + ).first(); + const durationCandidates = []; + if (durationElement.length) { + durationCandidates.push( + durationElement.attr('data-value'), + durationElement.attr('data-duration'), + durationElement.attr('data-time'), + durationElement.attr('aria-label'), + durationElement.attr('title'), + durationElement.text()?.trim() + ); + } + try { + const htmlSnippet = card.html() || ''; + const match = />\s*([0-9]+:[0-9]{2}(?::[0-9]{2})?)\s* 0) { durationSeconds = parsed; break; } + } + + // Vues : parseRumbleViews gère « 1,2K », « 3 456 », les espaces insécables. + // L'ancien parseur du routeur faisait .replace(/[^\d]/g,'') → « 1,2K » = 12. + const viewsText = + card.find('.video-item--views, .rumbles-views, .views, .video-item__views, [data-views]').first() + .attr('data-views') || + card.find('.video-item--views, .rumbles-views, .views, .video-item__views, .video-views').first().text().trim(); + const views = parseRumbleViews(viewsText); + + const uploaderName = card.find('.video-item--channel, .channel-name, a[href^="/c/"], a[href^="/user/"]').first().text().replace(/\s+/g, ' ').trim() || undefined; + + // On renvoie TOUJOURS une URL canonique cohérente. + let videoId = null; + let url = null; + if (id) { + videoId = id; + url = `https://rumble.com/${id}`; + } else if (isLegacy) { + videoId = href.replace(/^\//, ''); // "video/123..." + url = `https://rumble.com/${videoId}`; + } + + found.push({ + videoId, + title, + thumbnail: thumb, + uploaderName, + views, + duration: durationSeconds, + // Indicateur natif pour le front. En mode "shorts", tout ce qui sort du + // flux rumble.com/shorts (vertical ≤ 90 s) est un Short, même si le badge + // de durée est absent du HTML (la durée reste la source de vérité). + // `== null` : une durée absente vaut undefined (contrat « jamais 0 ») et + // doit rester acceptée par le flux Shorts, comme l'ancien 0 le faisait. + isShort: isShortsMode ? (durationSeconds == null || durationSeconds <= 90) : (durationSeconds > 0 && durationSeconds <= 75), + url, + type: 'video', + }); + }); + + // Dédoublonnage (un lien peut apparaître plusieurs fois dans une carte). + const seen = new Set(); + const unique = []; + for (const it of found) { + const key = it.videoId; + if (!key || seen.has(key)) continue; + seen.add(key); + unique.push(it); + } + // Le front /api/rumble/* lit `id` en priorité (mapRumbleItemToVideo) ; on + // garde `videoId` en clé secondaire pour l'ancien merge des Shorts. + for (const it of unique) it.id = it.videoId; + + patchItemsFromJsonLd(unique, html); + return unique; +} + +/** + * Pages liste hors recherche : /videos?sort=… (browse) et /shorts. + * Envelope { items, total, page, limit, nextCursor } identique à l'ancien + * routeur. Un blocage CF → throw 503 (le front dégrade en `items: []`, + * le bandeau « bloqué temporairement » vient de /api/search). + */ +export async function scrapeRumbleList({ page = 1, limit = 24, sort = 'viral', mode = 'browse' } = {}) { + const pageNum = Math.max(1, Number(page) || 1); + const limitNum = Math.min(50, Math.max(1, Number(limit) || 24)); + const url = mode === 'shorts' + ? `https://rumble.com/shorts?page=${pageNum}` + : `https://rumble.com/videos?sort=${encodeURIComponent(String(sort))}&page=${pageNum}`; + const r = await fetchHtml(url); + if (!r || isChallenge(r.html)) throw rumbleFail('rumble_cloudflare_challenge', 503); + + const items = parseCardsHtml(r.html, { mode }); + const list = items.slice(0, limitNum); + return { + items: list, + total: items.length, + page: pageNum, + limit: limitNum, + nextCursor: list.length === limitNum ? String(pageNum + 1) : null, + }; +} + /* --------------------------------- handler -------------------------------- */ const handler = { diff --git a/server/rumble.mjs b/server/rumble.mjs index 7209767..faf25a1 100644 --- a/server/rumble.mjs +++ b/server/rumble.mjs @@ -1,15 +1,29 @@ import express from 'express'; -import * as cheerio from 'cheerio'; -import axios from 'axios'; import rateLimit from 'express-rate-limit'; -import { spawn } from 'node:child_process'; -import path from 'node:path'; -import fs from 'node:fs'; -import { fileURLToPath } from 'node:url'; +import { getProviderAdapter } from './providers/registry.mjs'; +import { + normalizeRumbleId, + scrapeRumbleList, + scrapeRumbleVideo, +} from './providers/rumble.mjs'; + +/** + * Routes HTTP Rumble — COUCHE MINCE uniquement. + * + * Tout le scraping (fetch Cloudflare à 3 niveaux, cookie jar __cf_bm, cooldown + * global, negative cache, parsing DOM/JSON-LD, vues K/M/B) vit dans + * `server/providers/rumble.mjs`, le cœur partagé avec la recherche unifiée : + * aucune logique de fetch ni de parse dans ce fichier (l'implémentation + * dupliquée — axios + spawn python + parseur de cartes, vues « 1,2K » lues en + * 12 — a été supprimée ici). `/search` passe par le registre de providers : + * cache SQLite, negative cache et channelRef sont partagés avec `/api/search`. + */ const router = express.Router(); /* ----------------------------- Rate limiting ----------------------------- */ +// 20/min : la valeur vivante historique (l'ancien rumbleLimiter 10/min +// d'index.mjs n'était jamais monté — supprimé avec le code mort). const rumbleLimiter = rateLimit({ windowMs: 60 * 1000, max: 20, @@ -20,6 +34,9 @@ const rumbleLimiter = rateLimit({ router.use(rumbleLimiter); /* --------------------------------- Cache -------------------------------- */ +// Cache positif court : les pages Rumble changent peu et chaque miss coûte un +// fetch (+ éventuellement un spawn python). Les ÉCHECS ne sont pas cachés ici : +// c'est le negative cache + le cooldown du cœur qui les portent. const cache = new Map(); const TTL_MS = 60 * 1000; // 60s @@ -36,483 +53,20 @@ function getCache(key) { return hit.data; } -/* ------------------------------- HTTP GET -------------------------------- */ /** - * Stratégie best-effort (cf. server/providers/rumble.mjs) : - * 1. axios/Node avec headers navigateur (rapide quand CF ne challenge pas) ; - * 2. helper python3 + curl_cffi (impersonation Chrome) — indispensable derrière - * Cloudflare (les pages /search/* retournent sinon 403 "Just a moment"). - * Retourne le HTML brut ou lève une erreur si tout est challengé. + * Contrat d'erreur : un blocage Cloudflare est un 503 typé, JAMAIS un 404 + * (le 404 ferait retirer la vidéo du flux Shorts — « introuvable de façon + * certaine ») et jamais un 200 vide sans diagnostic. Les erreurs sans statut + * (bug inattendu) restent en 500 typé. */ -const ROUTER_DIR = path.dirname(fileURLToPath(import.meta.url)); -const ROUTER_PY_HELPER = path.join(ROUTER_DIR, 'providers', 'rumble_fetch.py'); -const ROUTER_PY_HELPER_EXISTS = fs.existsSync(ROUTER_DIR) && fs.existsSync(ROUTER_PY_HELPER); - -function isChallengeHtml(html) { - return /Just a moment|challenge-platform|cf-chl/i.test(String(html || '').slice(0, 4000)); -} - -async function pythonFetchHtml(url, { timeoutMs = 20_000 } = {}) { - if (!ROUTER_PY_HELPER_EXISTS) return null; - for (const py of ['python3', 'python']) { - const result = await new Promise((resolve) => { - let settled = false; - const done = (v) => { if (!settled) { settled = true; resolve(v); } }; - let child; - try { - child = spawn(py, [ROUTER_PY_HELPER, url, String(Math.ceil(timeoutMs / 1000))], { - stdio: ['ignore', 'pipe', 'ignore'], - }); - } catch { return done(null); } - let stdout = ''; - let failed = false; - const timer = setTimeout(() => { failed = true; try { child.kill(); } catch {} done(null); }, timeoutMs + 5_000); - child.stdout.on('data', (d) => { stdout += d.toString(); }); - child.on('error', () => { clearTimeout(timer); done(null); }); - child.on('close', (code) => { - clearTimeout(timer); - if (failed) return; - done(code === 0 && stdout.length > 1000 ? stdout : null); - }); - }); - if (result) return result; - } - return null; -} - -async function httpGet(url) { - // Tentative 1 : axios/Node direct - try { - const resp = await axios.get(url, { - headers: { - // UA “desktop” moderne pour minimiser les anti-bot simples - 'User-Agent': - 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36', - 'Accept': - 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8', - 'Accept-Language': 'en-US,en;q=0.9', - 'sec-ch-ua': '"Chromium";v="131", "Not_A Brand";v="24"', - 'sec-ch-ua-mobile': '?0', - 'sec-ch-ua-platform': '"Windows"', - 'sec-fetch-dest': 'document', - 'sec-fetch-mode': 'navigate', - 'sec-fetch-site': 'none', - 'sec-fetch-user': '?1', - 'upgrade-insecure-requests': '1', - }, - timeout: 15000, - // Important: pas de redirects inter-domain hasardeux - maxRedirects: 3, - validateStatus: s => s >= 200 && s < 400 - }); - const html = resp.data; - if (typeof html === 'string' && html.length > 1000 && !isChallengeHtml(html)) return html; - } catch (e) { - // Statuts 403 Cloudflare → on bascule vers le helper python ci-dessous - const status = e?.response?.status; - if (status && status !== 403) throw e; - } - // Tentative 2 : python + curl_cffi (impersonation navigateur) - const pyHtml = await pythonFetchHtml(url); - if (pyHtml) return pyHtml; - const err = new Error('Request failed with status code 403'); - err.status = 403; - throw err; -} - -/* ------------------------- Utils: normalisation ID ------------------------ */ -/** - * Rumble expose plusieurs formes: - * - Page canoniques: https://rumble.com/v6siqxf-some-title.html - * - Ancienne forme: https://rumble.com/video/12345 - * - URL d’embed officielle: https://rumble.com/embed/v6siqxf/ - * - ID brut attendu: v6siqxf (toujours commence par 'v' + base62) - * - * Cette fonction accepte: ID ou URL et renvoie { id: 'vXXXX', urlCanonique, embedUrl } - */ -function normalizeRumbleId(input, { preferEmbed = true } = {}) { - if (!input) return null; - - let id = null; - let urlCanonique = null; - let embedUrl = null; - - // 1) Si on nous donne déjà un ID "vXXXX" - const clean = String(input).trim(); - const mIdOnly = /^v[0-9A-Za-z]+$/.exec(clean); - if (mIdOnly) { - id = clean; - urlCanonique = `https://rumble.com/${id}`; - embedUrl = `https://rumble.com/embed/${id}/`; - return { id, urlCanonique, embedUrl }; - } - - // 2) Si on nous donne une URL - try { - const u = new URL(clean, 'https://rumble.com'); - // /embed/vXXXX/ - let m = /\/embed\/(v[0-9A-Za-z]+)/.exec(u.pathname); - if (!m) m = /\/(v[0-9A-Za-z]+)(?:[-/.]|$)/.exec(u.pathname); - if (!m) { - // ancienne forme /video/123 → on ne sait pas convertir de manière fiable - const mOld = /\/video\/([0-9A-Za-z]+)/.exec(u.pathname); - if (mOld) { - // On garde l’URL telle quelle et laissera le parseur de la page extraire le vrai vID. - return { id: null, urlCanonique: u.href, embedUrl: null }; - } - return null; - } - id = m[1]; - urlCanonique = `https://rumble.com/${id}`; - embedUrl = `https://rumble.com/embed/${id}/`; - return { id, urlCanonique, embedUrl }; - } catch { - return null; - } -} - -/* ---------------------- Parsing robuste d’une PAGE vidéo ------------------ */ -/** - * Source d’autorité pour le vrai ID: le JS inline: - * Rumble("play", {..., "video":"vXXXX", ...}) - * On prend ensuite en fallback: (souvent /embed/vXXXX/) - * puis ou (contenant /vXXXX-...). - * - * NB: ce choix est basé sur l’observation publique: la valeur "video":"vXXXX" - * est exactement l’ID attendu par l’embed officiel. - */ -function extractVideoIdentity($) { - // 1) Script "Rumble('play', {... "video":"vXXXX" ...})" - // On évite d’exécuter quoi que ce soit; simple regex sur tout le HTML. - const html = $.html() || ''; - let m = /Rumble\(\s*["']play["']\s*,\s*{[^}]*["']video["']\s*:\s*["'](v[0-9A-Za-z]+)["']/s.exec(html); - if (m && m[1]) { - const id = m[1]; - return { - id, - embedUrl: `https://rumble.com/embed/${id}/`, - urlCanonique: `https://rumble.com/${id}` - }; - } - - // 2) og:video → .../embed/vXXXX/... - let embed = $('meta[property="og:video"]').attr('content') - || $('meta[name="twitter:player"]').attr('content'); - if (embed) { - if (embed.startsWith('//')) embed = 'https:' + embed; - const mm = /\/embed\/(v[0-9A-Za-z]+)/.exec(embed); - if (mm) { - const id = mm[1]; - return { id, embedUrl: `https://rumble.com/embed/${id}/`, urlCanonique: `https://rumble.com/${id}` }; - } - } - - // 3) Canonical / og:url → .../vXXXX-... - let canon = $('link[rel="canonical"]').attr('href') - || $('meta[property="og:url"]').attr('content'); - if (canon) { - const mm = /\/(v[0-9A-Za-z]+)(?:[-/.]|$)/.exec(canon); - if (mm) { - const id = mm[1]; - return { id, embedUrl: `https://rumble.com/embed/${id}/`, urlCanonique: `https://rumble.com/${id}` }; - } - } - - return null; -} - -/* -------------------------- Scraper d’une vidéo -------------------------- */ -async function scrapeRumbleVideo(videoIdOrUrl) { - try { - // Accepte /:videoId ou une URL complète. - let norm = normalizeRumbleId(videoIdOrUrl); - const fetchUrl = norm?.urlCanonique || `https://rumble.com/${videoIdOrUrl}`; - const html = await httpGet(fetchUrl); - const $ = cheerio.load(html); - - // Identité fiable (id + embed + canonique) - let ident = extractVideoIdentity($); - if (!ident) { - // dernier recours: ré-essayer avec la page telle quelle si on est venu via /video/123 - if (!norm?.id && norm?.urlCanonique) { - ident = extractVideoIdentity($); - } - } - if (!ident?.id) { - return { error: 'Unable to determine Rumble video ID', input: videoIdOrUrl }; - } - - // Métadonnées robustes - const title = - $('h1.video-title, .video-title h1').first().text().trim() - || $('meta[property="og:title"]').attr('content') || 'Untitled Video'; - - let thumbnail = $('meta[property="og:image"]').attr('content') || ''; - if (thumbnail && thumbnail.startsWith('//')) thumbnail = 'https:' + thumbnail; - - const uploaderName = - $('.media-by--a, .channel-name, a[href*="/c/"]').first().text().trim() || ''; - - const viewsText = - $('.rumbles-views, .video-views, .media-view-count, [data-view-count]').first().text().trim() || ''; - const views = parseInt(viewsText.replace(/[^\d]/g, ''), 10) || 0; - - const duration = parseInt($('meta[property="video:duration"]').attr('content') || '', 10) || 0; - - const uploadedDate = - $('meta[property="article:published_time"]').attr('content') - || $('time[datetime]').attr('datetime') || ''; - - const description = - $('meta[property="og:description"]').attr('content') - || $('meta[name="description"]').attr('content') || ''; - - // embedUrl final — toujours la forme officielle - const embedUrl = ident.embedUrl; - - return { - videoId: ident.id, - title, - thumbnail, - uploaderName, - views, - duration, - uploadedDate, - description, - url: ident.urlCanonique, - embedUrl, - type: 'video' - }; - } catch (e) { - const msg = (e && e.message) ? e.message : String(e); - return { error: `Scraping failed: ${msg}` }; - } -} - -/* ------------------ Scraper de liste (search / browse) ------------------ */ -function parseDurationToSeconds(text) { - if (!text) return 0; - // Garde-fou : les attributs datetime (dates de publication) ne sont jamais des durées. - if (/^\d{4}-\d{2}-\d{2}/.test(String(text).trim())) return 0; - - // Nettoyer le texte en supprimant les espaces et caractères non numériques inutiles - const cleanText = text.trim().replace(/\s+/g, ''); - - // Format hh:mm:ss - let m = cleanText.match(/^(\d+):(\d{2}):(\d{2})$/); - if (m) { - const h = parseInt(m[1], 10) || 0; - const mn = parseInt(m[2], 10) || 0; - const s = parseInt(m[3], 10) || 0; - return h * 3600 + mn * 60 + s; - } - - // Format mm:ss - m = cleanText.match(/^(\d+):(\d{2})$/); - if (m) { - const mn = parseInt(m[1], 10) || 0; - const s = parseInt(m[2], 10) || 0; - return mn * 60 + s; - } - - // Format avec unités (ex: 1h 30m 45s) - m = cleanText.match(/(\d+h)?(\d+m)?(\d+s)?/); - if (m) { - const hours = m[1] ? parseInt(m[1], 10) : 0; - const minutes = m[2] ? parseInt(m[2], 10) : 0; - const seconds = m[3] ? parseInt(m[3], 10) : 0; - return (hours * 3600) + (minutes * 60) + seconds; - } - - // Si on arrive ici, on essaie d'extraire tous les nombres et on suppose un format mmss - const numbers = cleanText.match(/\d+/g); - if (numbers && numbers.length > 0) { - // Si un seul nombre, on suppose que c'est en secondes - if (numbers.length === 1) { - return parseInt(numbers[0], 10) || 0; - } - // Si deux nombres, on suppose mm:ss - if (numbers.length === 2) { - return (parseInt(numbers[0], 10) * 60) + (parseInt(numbers[1], 10) || 0); - } - // Si trois nombres, on suppose hh:mm:ss - if (numbers.length >= 3) { - return (parseInt(numbers[0], 10) * 3600) + - (parseInt(numbers[1], 10) * 60) + - (parseInt(numbers[2], 10) || 0); - } - } - - return 0; // Par défaut si aucun format n'est reconnu -} - -async function scrapeRumbleList({ q, page = 1, limit = 24, sort = 'viral', mode = 'videos' }) { - try { - // Mode "shorts" : le flux natif rumble.com/shorts (vertical ≤ 90 s). - // Les liens sont de la forme /shorts/vXXXX (plus éventuellement ?tracking). - const isShortsMode = mode === 'shorts'; - const url = isShortsMode - ? `https://rumble.com/shorts?page=${page}` - : q - ? `https://rumble.com/search/video?q=${encodeURIComponent(q)}&page=${page}` - : `https://rumble.com/videos?sort=${encodeURIComponent(sort)}&page=${page}`; - - const html = await httpGet(url); - const $ = cheerio.load(html); - - const found = []; - // 1) Cartes "vidéos" standards (li/article/div) - const linkSelector = isShortsMode - ? 'a[href*="/shorts/v"], a[href^="/v"], a[href^="/video/"]' - : 'a[href^="/v"], a[href^="/video/"]'; - $(linkSelector).each((_, el) => { - const href = ($(el).attr('href') || '').split('?')[0]; - // On préfère STRICTEMENT l’ID /vXXXX (aussi après /shorts/). - let m = /\/(v[0-9A-Za-z]+)(?:[-/.]|$)/.exec(href); - let id = m?.[1] || null; - - // Fallback minimaliste pour /video/123 → on ne convertit pas ici; on laissera /video/:id passer au détails qui normalise par parse de la page. - const isLegacy = !id && /^\/video\//.test(href); - - if (!id && !isLegacy) return; - - const card = $(el).closest('li, article, .video-listing-entry, .video-item, .video-card, div'); - - // Titre : préférer le vrai titre de la carte (h3/h2) au texte brut du lien - // (le lien contient souvent des badges "N watching" / durées qui polluaient le titre). - const cardTitle = card.find('h3, h2, .video-item--title').first().text().replace(/\s+/g, ' ').trim(); - const linkTitle = ($(el).attr('title') || '').replace(/\s+/g, ' ').trim(); - const linkText = $(el).text().replace(/\s+/g, ' ').trim(); - let title = cardTitle || linkTitle || linkText; - // Filtrer les faux titres (badges viewers, durées seules, chaînes vides) - if (!title || /^\d+\s+watching$/i.test(title) || /^[0-9:.,\s]+$/.test(title) || title.length < 3) { - // Dernier recours : attribut alt de l'image de la carte - const alt = (card.find('img').first().attr('alt') || '').replace(/\s+/g, ' ').trim(); - if (alt && alt.length >= 3 && !/^\d+\s+watching$/i.test(alt)) title = alt; - else return; // ignorer cette carte (badge live, doublon de lien, …) - } - - // Thumb robuste: data-src > src - let thumb = - card.find('img').first().attr('data-src') - || card.find('img').first().attr('src') - || ''; - if (thumb && thumb.startsWith('//')) thumb = 'https:' + thumb; - - // Essayer plusieurs sélecteurs pour la durée, y compris les attributs data- - // Ajout de plus de sélecteurs spécifiques à Rumble pour la durée - const durationElement = card.find( - '.video-item--duration, .video-duration, .duration, .video-item__duration, ' + - '[data-duration], .videoDuration, .video-time, .time, ' + - '.video-card__duration, .media__duration, .thumb-time, ' + - '.video-listing-entry__duration, .video-item__duration, time' - ).first(); - - const durationCandidates = []; - if (durationElement.length) { - durationCandidates.push( - durationElement.attr('data-value'), // motif Rumble : - durationElement.attr('data-duration'), - durationElement.attr('data-time'), - durationElement.attr('aria-label'), - durationElement.attr('title'), - durationElement.text()?.trim() - ); - } - - // Chercher dans le HTML de la carte un motif HH:MM(:SS) - try { - const htmlSnippet = card.html() || ''; - const match = />\s*([0-9]+:[0-9]{2}(?::[0-9]{2})?)\s* 0) { - durationSeconds = parsed; - break; - } - } - - // Extraire les vues - const viewsText = - card.find('.video-item--views, .rumbles-views, .views, .video-item__views, [data-views]').first() - .attr('data-views') || - card.find('.video-item--views, .rumbles-views, .views, .video-item__views, .video-views').first().text().trim(); - - const views = parseInt((viewsText || '').replace(/[^\d]/g, ''), 10) || 0; - - // Important: on renvoie TOUJOURS une URL canonique cohérente - let url = null; - let videoId = null; - - if (id) { - videoId = id; - url = `https://rumble.com/${id}`; - } else if (isLegacy) { - // Laisse l’endpoint /video/:slug gérer la normalisation - videoId = href.replace(/^\//, ''); // "video/123..." - url = `https://rumble.com/${videoId}`; - } - - // Filtrage doublons par videoId (id ou "video/123...") - const key = videoId; - // Uploader : meilleur effort (nom de chaîne sous la carte) - const uploaderName = card.find('.video-item--channel, .channel-name, a[href^="/c/"], a[href^="/user/"]').first().text().replace(/\s+/g, ' ').trim() || ''; - found.push({ - videoId: key, - title, - thumbnail: thumb, - uploaderName, - views, - duration: durationSeconds, - // Indicateur natif pour le front. En mode "shorts", tout ce qui sort - // du flux rumble.com/shorts (vertical ≤ 90 s) est un Short, même si - // le badge de durée est absent du HTML. - // (Rumble n'expose pas de flag natif ; la durée reste la source de vérité.) - isShort: isShortsMode ? (durationSeconds <= 0 || durationSeconds <= 90) : (durationSeconds > 0 && durationSeconds <= 75), - uploadedDate: '', - url, - type: 'video' - }); - }); - - // De-dupe - const seen = new Set(); - const unique = []; - for (const it of found) { - if (!it.videoId) continue; - if (seen.has(it.videoId)) continue; - seen.add(it.videoId); - unique.push(it); - } - - // Limite + nextCursor (page-based) - const list = unique.slice(0, limit); - const nextCursor = list.length === limit ? String(Number(page) + 1) : null; - - return { - items: list, - total: unique.length, - page: Number(page), - limit: Number(limit), - nextCursor - }; - } catch (e) { - return { - items: [], - total: 0, - page: Number(page), - limit: Number(limit), - nextCursor: null, - error: (e && e.message) ? e.message : String(e) - }; - } +function sendError(res, e) { + const status = Number(e?.status) || 500; + return res.status(status).json({ error: e?.code || 'rumble_failed' }); } /* --------------------------------- Routes -------------------------------- */ + +// Tendances / classements : https://rumble.com/videos?sort=… router.get('/browse', async (req, res) => { const page = Math.max(1, parseInt(String(req.query.page || '1'), 10) || 1); const limit = Math.min(50, Math.max(1, parseInt(String(req.query.limit || '24'), 10) || 24)); @@ -522,16 +76,19 @@ router.get('/browse', async (req, res) => { const cached = getCache(key); if (cached) return res.json(cached); - const data = await scrapeRumbleList({ page, limit, sort }); - setCache(key, data); - return res.json(data); + try { + const data = await scrapeRumbleList({ page, limit, sort }); + setCache(key, data); + return res.json(data); + } catch (e) { + return sendError(res, e); + } }); -/* ---------------- Route "shorts" natifs (rumble.com/shorts) -------------- */ /** * GET /api/rumble/shorts?page=&limit= - * Scrape le flux natif des Shorts Rumble (vertical ≤ 90 s) : IDs /vXXXX - * propres, compatibles avec l'embed officiel /embed/vXXXX/. + * Flux natif des Shorts Rumble (rumble.com/shorts, vertical ≤ 90 s) : + * IDs /vXXXX propres, compatibles avec l'embed officiel /embed/vXXXX/. */ router.get('/shorts', async (req, res) => { const page = Math.max(1, parseInt(String(req.query.page || '1'), 10) || 1); @@ -541,11 +98,17 @@ router.get('/shorts', async (req, res) => { const cached = getCache(key); if (cached) return res.json(cached); - const data = await scrapeRumbleList({ page, limit, mode: 'shorts' }); - setCache(key, data); - return res.json(data); + try { + const data = await scrapeRumbleList({ page, limit, mode: 'shorts' }); + setCache(key, data); + return res.json(data); + } catch (e) { + return sendError(res, e); + } }); +// Recherche : même adaptateur (et donc même cache SQLite, negative cache et +// channelRef) que le groupe `ru` de la recherche unifiée. router.get('/search', async (req, res) => { const q = String(req.query.q || '').trim(); if (!q) return res.status(400).json({ error: 'Query parameter required' }); @@ -563,12 +126,25 @@ router.get('/search', async (req, res) => { const cached = getCache(key); if (cached) return res.json(cached); - const data = await scrapeRumbleList({ q, page, limit }); - setCache(key, data); - return res.json(data); + try { + const items = await getProviderAdapter('ru').search(q, { limit, page }); + const data = { + items, + total: items.length, + page, + limit, + nextCursor: items.length === limit ? String(page + 1) : null, + }; + setCache(key, data); + return res.json(data); + } catch (e) { + return sendError(res, e); + } }); -// Endpoint details. Accepte :videoId pouvant être "vXXXX" OU "video/123..." +// Détails d'une vidéo (résolution d'embed fiable pour /watch et /shorts). +// Accepte :videoId pouvant être "vXXXX" OU "video/123..." — l'ancien pattern +// :videoId(*) est conservé tel quel (comportement de route inchangé). router.get('/video/:videoId(*)', async (req, res) => { try { const raw = String(req.params.videoId); @@ -576,40 +152,36 @@ router.get('/video/:videoId(*)', async (req, res) => { const cached = getCache(key); if (cached) return res.json(cached); - // Normalise au maximum avant scrape const norm = normalizeRumbleId(raw) || { urlCanonique: `https://rumble.com/${raw}` }; const data = await scrapeRumbleVideo(norm.id || norm.urlCanonique); - if (data.error) return res.status(404).json(data); - setCache(key, data); return res.json(data); - } catch (error) { - return res.status(500).json({ error: 'Failed to scrape video' }); + } catch (e) { + return sendError(res, e); } }); -/* ----------------- Option: “prélecteur” sans pub (non-embed) -------------- */ /** - * On NE désactive PAS les pubs côté Rumble (pas de param officiel fiable). - * Mais on peut servir un “preplay”: - * - On affiche miniature/titre. - * - Au clic: (A) ouvrir dans Rumble (UX la plus propre), ou (B) injecter l’iframe - * officiellement (ce qui déclenchera leur logique pub). - * Cette route renvoie juste les meta nécessaires pour ce composant prélecteur. + * Option : « prélecteur » sans pub (non-embed). + * On NE désactive PAS les pubs côté Rumble (pas de paramètre officiel fiable) : + * cette route renvoie juste les métadonnées du composant prélecteur + * (miniature/titre + lien d'ouverture fournisseur, iframe en dernier recours). */ router.get('/video/:videoId/preplay', async (req, res) => { - const raw = String(req.params.videoId); - const norm = normalizeRumbleId(raw) || { urlCanonique: `https://rumble.com/${raw}` }; - const data = await scrapeRumbleVideo(norm.id || norm.urlCanonique); - if (data.error) return res.status(404).json(data); - const preplay = { - videoId: data.videoId, - title: data.title, - thumbnail: data.thumbnail, - rumbleUrl: data.url, // bouton "Ouvrir sur le site du fournisseur" - embedUrl: data.embedUrl // injection différée si l’utilisateur insiste pour lire ici - }; - res.json(preplay); + try { + const raw = String(req.params.videoId); + const norm = normalizeRumbleId(raw) || { urlCanonique: `https://rumble.com/${raw}` }; + const data = await scrapeRumbleVideo(norm.id || norm.urlCanonique); + return res.json({ + videoId: data.videoId, + title: data.title, + thumbnail: data.thumbnail, + rumbleUrl: data.url, // bouton "Ouvrir sur le site du fournisseur" + embedUrl: data.embedUrl // injection différée si l'utilisateur insiste + }); + } catch (e) { + return sendError(res, e); + } }); export default router; diff --git a/server/tests/rumble-ld.test.mjs b/server/tests/rumble-ld.test.mjs index 24304a6..0314760 100644 --- a/server/tests/rumble-ld.test.mjs +++ b/server/tests/rumble-ld.test.mjs @@ -1,4 +1,4 @@ -import { parseJsonLd, parseRumbleViews, parseSearchHtml, resetRumbleNegativeCache } from '../providers/rumble.mjs'; +import { parseJsonLd, parseRumbleViews, parseSearchHtml, resetRumbleNegativeCache, parseCardsHtml, scrapeRumbleList, scrapeRumbleVideo, armRumbleFetchCooldown, resetRumbleFetchCooldown } from '../providers/rumble.mjs'; import { parseRelativeDate } from '../providers/youtube-innertube.mjs'; import { itemDurationSec } from '../search-filters.mjs'; @@ -133,4 +133,52 @@ eq(itemDurationSec({ duration: 0, width: 1080, height: 1920 }), undefined, resetRumbleNegativeCache(); pass++; console.log(' V negative cache is resettable (tests stay hermetic)'); +// --- Cœur partagé (ménage Rumble) : cartes génériques + contrat 503/404 --- +// Formes d'URL réelles : /vXXXX-titre-slug.html (l'id s'arrête au slug) et +// /shorts/vXXXX pour le flux natif. +const cardsHtml = ` +
+ +

Titre A

1,2K views + +
+
+ +

Titre B

+
+
+

Titre C

+
+`; + +const browseCards = parseCardsHtml(cardsHtml, { mode: 'browse' }); +eq(browseCards.length, 2, 'cards browse: /vXXXX lu, /shorts/ ignoré'); +eq(browseCards[0].id, 'vAAA11', 'cards: id = vXXXX avant le slug -titre'); +eq(browseCards[0].views, 1200, 'cards: vues « 1,2K » -> 1200 (bug parseur legacy corrigé)'); +eq(browseCards[0].duration, 90, 'cards: durée depuis data-value'); +eq(browseCards[0].thumbnail, 'https://i.rumble.com/a.jpg', 'cards: vignette protocol-relative -> https'); +eq(browseCards[0].isShort, false, 'cards browse: 90 s > 75 -> pas un Short'); +eq(browseCards[1].views, 3456, 'cards: data-views numérique brut lu tel quel'); +eq(browseCards[1].duration, undefined, 'cards: durée absente -> undefined (jamais 0)'); + +const shortCards = parseCardsHtml(cardsHtml, { mode: 'shorts' }); +eq(shortCards.length, 3, 'cards shorts: lien /shorts/vXXXX aussi lu'); +eq(shortCards[2].id, 'vCCC33', 'cards shorts: id extrait après /shorts/'); +eq(shortCards[2].isShort, true, 'cards shorts: durée absente -> Short natif (ancien comportement 0)'); + +// Cooldown CF : fetch court-circuité => 503 typé, JAMAIS 404 (le 404 ferait +// retirer la vidéo du flux Shorts), et ce pour les deux chemins de scrape. +armRumbleFetchCooldown(60_000); +const expectFail = async (fn, status, code, msg) => { + let err = null; + try { await fn(); } catch (e) { err = e; } + if (!err) throw new Error(`${msg}: aucun throw — le cooldown n'a pas court-circuité le fetch`); + eq(err.status, status, `${msg}: status ${status}`); + eq(err.code, code, `${msg}: code ${code}`); +}; +await expectFail(() => scrapeRumbleVideo('v-cooltest'), 503, 'rumble_cloudflare_challenge', '/video en cooldown CF'); +await expectFail(() => scrapeRumbleList({}), 503, 'rumble_cloudflare_challenge', '/browse en cooldown CF'); +resetRumbleFetchCooldown(); +pass++; console.log(' V cooldown CF: /video et /browse échouent en 503 sans réseau'); + console.log(`\n rumble-ld: ${pass} assertions OK`); diff --git a/src/services/youtube-api.service.ts b/src/services/youtube-api.service.ts index 9c438aa..e339280 100644 --- a/src/services/youtube-api.service.ts +++ b/src/services/youtube-api.service.ts @@ -1361,7 +1361,7 @@ export class YoutubeApiService { const uploaderAvatar = i?.author_avatar || i?.uploader_avatar || i?.channel?.avatar || i?.uploaderAvatar || thumb || ''; const channelSlug = i?.channel?.slug || i?.channel_slug || (i?.channel?.url ? i.channel.url.replace(/^https?:\/\/rumble.com\//, '').replace(/\/$/, '') : '') || ''; const channelExternalId = channelSlug || (uploaderName ? uploaderName.replace(/\s+/g, '').toLowerCase() : ''); - const uploadedDate = i?.published_at || i?.created_at || i?.uploadedDate || ''; + const uploadedDate = i?.published_at || i?.created_at || i?.publishedAt || i?.uploadedDate || ''; const views = Number(i?.views || i?.view_count || 0); const duration = this.normalizeDuration( i?.duration ?? i?.durationSeconds ?? i?.duration_seconds ?? i?.video_duration ?? i?.length ?? i?.meta?.duration