Deux causes indépendantes faisaient échouer le job test de la CI (et seulement en CI : ni .env ni les mêmes ordres de chargement en local) : - test_agent_web_tools.mock_http patche http_client.shared_client pendant sa fenêtre d'exécution. Le premier import d'importers/url_fetch dans cette fenêtre fige la factory moquée dans le namespace du module → tous les appels fetch_url suivants du processus passaient par le handler mocké de l'autre test (« assert '…/post' == '…/page' » dans test_v56_import). url_fetch résout désormais le client à l'appel, et le mock_http restaure la vraie factory (référence figée au chargement du module) y compris sur url_fetch ; - test_agent.py posait RATE_LIMIT_ENABLED=false en env var, sans effet sur le singleton Settings déjà instancié : sous pytest -n auto, le worker dépassait le quota de 60 req/min et 8 tests recevaient des 429 (KeyError 'id' au passage). Le fixture désactive maintenant le limiter sur le singleton, comme les autres fichiers de tests.
102 lines
3.8 KiB
Python
102 lines
3.8 KiB
Python
"""FlowDeck — URL / web clipper importer (v5.6.0, Phase 5).
|
|
|
|
Fetches a web page and turns it into a page: a bookmark card (OG metadata)
|
|
followed by the article converted to FlowDeck blocks.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import ipaddress
|
|
import socket
|
|
from urllib.parse import urlparse
|
|
|
|
import httpx
|
|
|
|
from app.services import http_client
|
|
from app.services.export import markdown_to_blocks
|
|
from app.services.importers.base import ImportPage, ImportResult
|
|
from app.services.importers.html_notes import _html_to_markdown
|
|
|
|
_BLOCKED_HOSTS = {"localhost", "localhost.localdomain"}
|
|
_MAX_BYTES = 3_000_000
|
|
|
|
|
|
def _is_public_host(host: str) -> bool:
|
|
"""SSRF guard: reject loopback/private/link-local/reserved addresses."""
|
|
if not host or host.lower() in _BLOCKED_HOSTS:
|
|
return False
|
|
try:
|
|
infos = socket.getaddrinfo(host, None)
|
|
except OSError:
|
|
# gaierror, socket.timeout, herror… : le garde-fou doit TOUJOURS
|
|
# renvoyer un booléen, jamais lever — sinon l'appelant tombe dans son
|
|
# `except Exception` (bookmarks) au lieu d'un refus explicite.
|
|
return False
|
|
for info in infos:
|
|
address = info[4][0]
|
|
try:
|
|
ip = ipaddress.ip_address(address)
|
|
except ValueError:
|
|
return False
|
|
if (ip.is_private or ip.is_loopback or ip.is_link_local
|
|
or ip.is_reserved or ip.is_multicast or ip.is_unspecified):
|
|
return False
|
|
return True
|
|
|
|
|
|
def _validate_url(url: str) -> str:
|
|
parsed = urlparse(url.strip())
|
|
if parsed.scheme not in ("http", "https"):
|
|
raise ValueError("Seules les URLs http(s) sont autorisées")
|
|
if not parsed.hostname or not _is_public_host(parsed.hostname):
|
|
raise ValueError("Hôte non autorisé")
|
|
return url.strip()
|
|
|
|
|
|
def _bookmark_block(url: str, meta: dict) -> dict:
|
|
block = {"type": "bookmark", "url": url}
|
|
for key in ("title", "description", "image", "site_name"):
|
|
if meta.get(key):
|
|
block[key] = meta[key]
|
|
return block
|
|
|
|
|
|
async def fetch_url_result(url: str, *, transport: httpx.BaseTransport | None = None) -> ImportResult:
|
|
"""Fetch ``url`` and build a single-page ImportResult (raises on bad URL)."""
|
|
safe_url = _validate_url(url)
|
|
result = ImportResult(source="url")
|
|
try:
|
|
# Résolu à l'APPEL (et non capturé à l'import) : un import tardif de
|
|
# ce module pendant qu'un test patche `http_client.shared_client`
|
|
# figerait la factory moquée pour tout le reste du processus.
|
|
async with http_client.shared_client(
|
|
timeout=15, follow_redirects=True, transport=transport,
|
|
headers={"User-Agent": "FlowDeck-Importer/1.0"},
|
|
) as client:
|
|
response = await client.get(safe_url)
|
|
response.raise_for_status()
|
|
if response.url.host and not _is_public_host(response.url.host):
|
|
raise ValueError("Redirection vers un hôte non autorisé")
|
|
content_type = response.headers.get("content-type", "")
|
|
if "html" not in content_type.lower():
|
|
raise ValueError("La ressource n'est pas une page HTML")
|
|
body = response.text[:_MAX_BYTES]
|
|
except httpx.HTTPError as exc:
|
|
result.warn(f"Échec du téléchargement : {exc}")
|
|
return result.finalize()
|
|
|
|
from app.services.og_fetcher import parse_og
|
|
|
|
meta = parse_og(body, safe_url)
|
|
title = (meta.get("title") or urlparse(safe_url).hostname or "Page").strip()
|
|
markdown = _html_to_markdown(body)
|
|
blocks = [_bookmark_block(safe_url, meta)]
|
|
if markdown:
|
|
blocks.extend(markdown_to_blocks(markdown))
|
|
result.pages.append(ImportPage(
|
|
title=title[:200],
|
|
blocks=blocks,
|
|
source_path=safe_url,
|
|
external_id=safe_url,
|
|
))
|
|
return result.finalize()
|