- `app/services/http_client.py` : `async with shared_client(timeout=15) as client:` remplace les 49 créations `async with httpx.AsyncClient(` de 14 fichiers (gitea ×21, providers oidc/oauth ×11, calendar ×4, automations ×3…) — le pool de connexions est réutilisé au lieu d'être recréé à chaque appel. __aexit__ no-op (le client partagé ne se ferme pas à la sortie). - Cache par (boucle d'event, kwargs) en WeakKeyDictionary : un AsyncClient n'est JAMAIS partagé entre deux loops (piège des tests « Event loop is closed ») — une boucle par test = client propre collecté avec la boucle. Clé = kwargs triés, repr() pour les valeurs non hashables (`headers=` dict → TypeError rattrapé par la suite). - Laissés délibérément : github_adapter (transport MockTransport injecté), webhook_outbound (client « own_client » fermé par la fonction). - Tests : `test_http_client_shared_and_loop_scoped` (réutilisation mêmes kwargs / cloisonné kwargs / cloisonné loop) ; le stub des webhooks patche aussi la fabrique `http_client.httpx` + purge du cache (avant : webhook_outbound.httpx patché mais la fabrique partagée créait un vrai client → réseau réel dans les tests). suite **1091/1091** · ruff OK · docs à jour
96 lines
3.3 KiB
Python
96 lines
3.3 KiB
Python
"""FlowDeck — URL / web clipper importer (v5.6.0, Phase 5).
|
|
|
|
Fetches a web page and turns it into a page: a bookmark card (OG metadata)
|
|
followed by the article converted to FlowDeck blocks.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import ipaddress
|
|
import socket
|
|
from urllib.parse import urlparse
|
|
|
|
import httpx
|
|
|
|
from app.services.export import markdown_to_blocks
|
|
from app.services.http_client import shared_client
|
|
from app.services.importers.base import ImportPage, ImportResult
|
|
from app.services.importers.html_notes import _html_to_markdown
|
|
|
|
_BLOCKED_HOSTS = {"localhost", "localhost.localdomain"}
|
|
_MAX_BYTES = 3_000_000
|
|
|
|
|
|
def _is_public_host(host: str) -> bool:
|
|
"""SSRF guard: reject loopback/private/link-local/reserved addresses."""
|
|
if not host or host.lower() in _BLOCKED_HOSTS:
|
|
return False
|
|
try:
|
|
infos = socket.getaddrinfo(host, None)
|
|
except socket.gaierror:
|
|
return False
|
|
for info in infos:
|
|
address = info[4][0]
|
|
try:
|
|
ip = ipaddress.ip_address(address)
|
|
except ValueError:
|
|
return False
|
|
if (ip.is_private or ip.is_loopback or ip.is_link_local
|
|
or ip.is_reserved or ip.is_multicast or ip.is_unspecified):
|
|
return False
|
|
return True
|
|
|
|
|
|
def _validate_url(url: str) -> str:
|
|
parsed = urlparse(url.strip())
|
|
if parsed.scheme not in ("http", "https"):
|
|
raise ValueError("Seules les URLs http(s) sont autorisées")
|
|
if not parsed.hostname or not _is_public_host(parsed.hostname):
|
|
raise ValueError("Hôte non autorisé")
|
|
return url.strip()
|
|
|
|
|
|
def _bookmark_block(url: str, meta: dict) -> dict:
|
|
block = {"type": "bookmark", "url": url}
|
|
for key in ("title", "description", "image", "site_name"):
|
|
if meta.get(key):
|
|
block[key] = meta[key]
|
|
return block
|
|
|
|
|
|
async def fetch_url_result(url: str, *, transport: httpx.BaseTransport | None = None) -> ImportResult:
|
|
"""Fetch ``url`` and build a single-page ImportResult (raises on bad URL)."""
|
|
safe_url = _validate_url(url)
|
|
result = ImportResult(source="url")
|
|
try:
|
|
async with shared_client(
|
|
timeout=15, follow_redirects=True, transport=transport,
|
|
headers={"User-Agent": "FlowDeck-Importer/1.0"},
|
|
) as client:
|
|
response = await client.get(safe_url)
|
|
response.raise_for_status()
|
|
if response.url.host and not _is_public_host(response.url.host):
|
|
raise ValueError("Redirection vers un hôte non autorisé")
|
|
content_type = response.headers.get("content-type", "")
|
|
if "html" not in content_type.lower():
|
|
raise ValueError("La ressource n'est pas une page HTML")
|
|
body = response.text[:_MAX_BYTES]
|
|
except httpx.HTTPError as exc:
|
|
result.warn(f"Échec du téléchargement : {exc}")
|
|
return result.finalize()
|
|
|
|
from app.services.og_fetcher import parse_og
|
|
|
|
meta = parse_og(body, safe_url)
|
|
title = (meta.get("title") or urlparse(safe_url).hostname or "Page").strip()
|
|
markdown = _html_to_markdown(body)
|
|
blocks = [_bookmark_block(safe_url, meta)]
|
|
if markdown:
|
|
blocks.extend(markdown_to_blocks(markdown))
|
|
result.pages.append(ImportPage(
|
|
title=title[:200],
|
|
blocks=blocks,
|
|
source_path=safe_url,
|
|
external_id=safe_url,
|
|
))
|
|
return result.finalize()
|