Files
flowdeck/app/services/importers/url_fetch.py
T
bruno 3bb8e87ef2
FlowDeck CI / lint (push) Canceled after 0s
FlowDeck CI / test (push) Canceled after 0s
FlowDeck CI / docker (push) Canceled after 0s
fix: A42 terminé — client httpx partagé par boucle (v7.28.0)
- `app/services/http_client.py` : `async with shared_client(timeout=15)
  as client:` remplace les 49 créations `async with httpx.AsyncClient(`
  de 14 fichiers (gitea ×21, providers oidc/oauth ×11, calendar ×4,
  automations ×3…) — le pool de connexions est réutilisé au lieu d'être
  recréé à chaque appel. __aexit__ no-op (le client partagé ne se ferme
  pas à la sortie).
- Cache par (boucle d'event, kwargs) en WeakKeyDictionary : un
  AsyncClient n'est JAMAIS partagé entre deux loops (piège des tests
  « Event loop is closed ») — une boucle par test = client propre
  collecté avec la boucle. Clé = kwargs triés, repr() pour les valeurs
  non hashables (`headers=` dict → TypeError rattrapé par la suite).
- Laissés délibérément : github_adapter (transport MockTransport
  injecté), webhook_outbound (client « own_client » fermé par la
  fonction).
- Tests : `test_http_client_shared_and_loop_scoped` (réutilisation mêmes
  kwargs / cloisonné kwargs / cloisonné loop) ; le stub des webhooks
  patche aussi la fabrique `http_client.httpx` + purge du cache (avant :
  webhook_outbound.httpx patché mais la fabrique partagée créait un vrai
  client → réseau réel dans les tests).

suite **1091/1091** · ruff OK · docs à jour
2026-10-01 23:09:45 -04:00

96 lines
3.3 KiB
Python

"""FlowDeck — URL / web clipper importer (v5.6.0, Phase 5).
Fetches a web page and turns it into a page: a bookmark card (OG metadata)
followed by the article converted to FlowDeck blocks.
"""
from __future__ import annotations
import ipaddress
import socket
from urllib.parse import urlparse
import httpx
from app.services.export import markdown_to_blocks
from app.services.http_client import shared_client
from app.services.importers.base import ImportPage, ImportResult
from app.services.importers.html_notes import _html_to_markdown
_BLOCKED_HOSTS = {"localhost", "localhost.localdomain"}
_MAX_BYTES = 3_000_000
def _is_public_host(host: str) -> bool:
"""SSRF guard: reject loopback/private/link-local/reserved addresses."""
if not host or host.lower() in _BLOCKED_HOSTS:
return False
try:
infos = socket.getaddrinfo(host, None)
except socket.gaierror:
return False
for info in infos:
address = info[4][0]
try:
ip = ipaddress.ip_address(address)
except ValueError:
return False
if (ip.is_private or ip.is_loopback or ip.is_link_local
or ip.is_reserved or ip.is_multicast or ip.is_unspecified):
return False
return True
def _validate_url(url: str) -> str:
parsed = urlparse(url.strip())
if parsed.scheme not in ("http", "https"):
raise ValueError("Seules les URLs http(s) sont autorisées")
if not parsed.hostname or not _is_public_host(parsed.hostname):
raise ValueError("Hôte non autorisé")
return url.strip()
def _bookmark_block(url: str, meta: dict) -> dict:
block = {"type": "bookmark", "url": url}
for key in ("title", "description", "image", "site_name"):
if meta.get(key):
block[key] = meta[key]
return block
async def fetch_url_result(url: str, *, transport: httpx.BaseTransport | None = None) -> ImportResult:
"""Fetch ``url`` and build a single-page ImportResult (raises on bad URL)."""
safe_url = _validate_url(url)
result = ImportResult(source="url")
try:
async with shared_client(
timeout=15, follow_redirects=True, transport=transport,
headers={"User-Agent": "FlowDeck-Importer/1.0"},
) as client:
response = await client.get(safe_url)
response.raise_for_status()
if response.url.host and not _is_public_host(response.url.host):
raise ValueError("Redirection vers un hôte non autorisé")
content_type = response.headers.get("content-type", "")
if "html" not in content_type.lower():
raise ValueError("La ressource n'est pas une page HTML")
body = response.text[:_MAX_BYTES]
except httpx.HTTPError as exc:
result.warn(f"Échec du téléchargement : {exc}")
return result.finalize()
from app.services.og_fetcher import parse_og
meta = parse_og(body, safe_url)
title = (meta.get("title") or urlparse(safe_url).hostname or "Page").strip()
markdown = _html_to_markdown(body)
blocks = [_bookmark_block(safe_url, meta)]
if markdown:
blocks.extend(markdown_to_blocks(markdown))
result.pages.append(ImportPage(
title=title[:200],
blocks=blocks,
source_path=safe_url,
external_id=safe_url,
))
return result.finalize()