- `app/services/http_client.py` : `async with shared_client(timeout=15) as client:` remplace les 49 créations `async with httpx.AsyncClient(` de 14 fichiers (gitea ×21, providers oidc/oauth ×11, calendar ×4, automations ×3…) — le pool de connexions est réutilisé au lieu d'être recréé à chaque appel. __aexit__ no-op (le client partagé ne se ferme pas à la sortie). - Cache par (boucle d'event, kwargs) en WeakKeyDictionary : un AsyncClient n'est JAMAIS partagé entre deux loops (piège des tests « Event loop is closed ») — une boucle par test = client propre collecté avec la boucle. Clé = kwargs triés, repr() pour les valeurs non hashables (`headers=` dict → TypeError rattrapé par la suite). - Laissés délibérément : github_adapter (transport MockTransport injecté), webhook_outbound (client « own_client » fermé par la fonction). - Tests : `test_http_client_shared_and_loop_scoped` (réutilisation mêmes kwargs / cloisonné kwargs / cloisonné loop) ; le stub des webhooks patche aussi la fabrique `http_client.httpx` + purge du cache (avant : webhook_outbound.httpx patché mais la fabrique partagée créait un vrai client → réseau réel dans les tests). suite **1091/1091** · ruff OK · docs à jour
174 lines
6.2 KiB
Python
174 lines
6.2 KiB
Python
"""FlowDeck — Bookmark cards (v5.5.0): Open Graph metadata via httpx.
|
|
|
|
Fetches a URL server-side, extracts OG/Twitter meta tags (title, description,
|
|
image, site name, favicon) and returns a safe, compact payload used to render
|
|
Notion-style bookmark cards. Robust to missing tags, non-HTML bodies and
|
|
slow/unreachable hosts.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import html as htmlmod
|
|
import logging
|
|
import re
|
|
from urllib.parse import urljoin, urlparse
|
|
|
|
from app.services.http_client import shared_client
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_META_TAG_RE = re.compile(r"<meta\b[^>]*?>", re.I)
|
|
_ATTR_RE = re.compile(r"([A-Za-z_:][-A-Za-z0-9_:.]*)\s*=\s*[\"']([^\"']*)[\"']")
|
|
_TITLE_RE = re.compile(r"<title[^>]*>(.*?)</title>", re.I | re.S)
|
|
_FAVICON_RE = re.compile(r"<link\b[^>]*?>", re.I)
|
|
_ICON_REL = re.compile(r"\b(?:shortcut\s+)?icon\b", re.I)
|
|
|
|
# Property/name keys we look for, in priority order, mapped to our payload keys.
|
|
_OG_TITLE = ("og:title", "twitter:title", "title", "og:site_name")
|
|
_OG_DESC = ("og:description", "twitter:description", "description")
|
|
_OG_IMG = ("og:image", "twitter:image", "twitter:image:src", "image")
|
|
_OG_SITE = ("og:site_name", "twitter:site", "application-name")
|
|
|
|
|
|
def _attrs(tag: str) -> dict:
|
|
return {k.lower(): v for k, v in _ATTR_RE.findall(tag)}
|
|
|
|
|
|
def _extract_og(body: str) -> dict:
|
|
"""Parse all ``<meta>`` tags into a ``{key: content}`` dict.
|
|
|
|
Attributes may appear in any order (``content`` before or after
|
|
``property``/``name``), which the previous implementation mishandled.
|
|
First value wins so the most specific tag (top of document) is kept.
|
|
"""
|
|
props: dict[str, str] = {}
|
|
for tag in _META_TAG_RE.finditer(body[:400_000]):
|
|
attrs = _attrs(tag.group(0))
|
|
key = (attrs.get("property") or attrs.get("name") or attrs.get("itemprop") or "").lower()
|
|
content = attrs.get("content")
|
|
if key and content is not None and key not in props:
|
|
props[key] = content
|
|
return props
|
|
|
|
|
|
def _pick(props: dict, keys: tuple) -> str:
|
|
for k in keys:
|
|
v = props.get(k)
|
|
if v:
|
|
return v
|
|
return ""
|
|
|
|
|
|
def _title_of(props: dict, body: str) -> str:
|
|
t = _pick(props, _OG_TITLE)
|
|
if t:
|
|
return t
|
|
m = _TITLE_RE.search(body[:200_000])
|
|
return m.group(1).strip() if m else ""
|
|
|
|
|
|
def _site_name(url: str) -> str:
|
|
host = urlparse(url).netloc.replace("www.", "")
|
|
return host.split(".")[0].capitalize() if host else ""
|
|
|
|
|
|
def _favicon(body: str, base_url: str) -> str:
|
|
for tag in _FAVICON_RE.finditer(body):
|
|
attrs = _attrs(tag.group(0))
|
|
rel = attrs.get("rel", "")
|
|
href = attrs.get("href", "")
|
|
if href and _ICON_REL.search(rel):
|
|
return urljoin(base_url, htmlmod.unescape(href))
|
|
return ""
|
|
|
|
|
|
def parse_og(body: str, url: str) -> dict:
|
|
"""Pure HTML → bookmark payload (no network). ``url`` is the base URL."""
|
|
src = url.strip()
|
|
if not src.startswith(("http://", "https://")):
|
|
src = "https://" + src
|
|
props = _extract_og(body)
|
|
title = htmlmod.unescape(_title_of(props, body))
|
|
desc = htmlmod.unescape(_pick(props, _OG_DESC))
|
|
img = _pick(props, _OG_IMG)
|
|
site = htmlmod.unescape(_pick(props, _OG_SITE)) or _site_name(src)
|
|
|
|
def abs_url(u: str) -> str:
|
|
return urljoin(src, htmlmod.unescape(u)) if u else ""
|
|
|
|
return {
|
|
"url": src,
|
|
"title": title.strip()[:200] or urlparse(src).netloc or src,
|
|
"description": desc.strip()[:400],
|
|
"image": abs_url(img),
|
|
"site_name": site.strip()[:100],
|
|
"favicon": _favicon(body, src),
|
|
}
|
|
|
|
|
|
_MAX_REDIRECTS = 5
|
|
|
|
|
|
async def _get_checked(client, url: str, headers: dict):
|
|
"""GET avec re-vérification de l'hôte à CHAQUE saut de redirection (A12 SSRF).
|
|
|
|
`follow_redirects=True` laisserait une URL publique rediriger vers
|
|
169.254.169.254 / localhost — la garde doit donc tourner à chaque hop.
|
|
"""
|
|
from app.services.importers.url_fetch import _is_public_host
|
|
|
|
current = url
|
|
for _ in range(_MAX_REDIRECTS + 1):
|
|
parsed = urlparse(current)
|
|
if parsed.scheme not in ("http", "https") or not parsed.hostname or not _is_public_host(parsed.hostname):
|
|
raise ValueError(f"hôte non autorisé: {parsed.hostname!r}")
|
|
r = await client.get(current, headers=headers, follow_redirects=False)
|
|
if r.status_code in (301, 302, 303, 307, 308):
|
|
loc = r.headers.get("location")
|
|
if not loc:
|
|
return r
|
|
current = urljoin(current, loc)
|
|
continue
|
|
r.raise_for_status()
|
|
return r
|
|
raise ValueError("trop de redirections")
|
|
|
|
|
|
async def fetch_og_metadata(url: str, timeout: float = 6.0, transport=None) -> dict:
|
|
"""Fetch ``url`` and return {url, title, description, image, site_name,
|
|
favicon}. Empty strings are omitted. Never raises for network errors.
|
|
|
|
``transport`` is an optional ``httpx`` transport (used by tests to mock
|
|
HTTP without hitting the network).
|
|
"""
|
|
src = url.strip()
|
|
if not src.startswith(("http://", "https://")):
|
|
src = "https://" + src
|
|
base = {"url": src, "title": "", "description": "", "image": "", "site_name": "", "favicon": ""}
|
|
try:
|
|
|
|
headers = {
|
|
"User-Agent": "FlowDeck/5.5 bookmark-fetcher (+https://flowdeck.dracodev.net)",
|
|
"Accept": "text/html,application/xhtml+xml",
|
|
}
|
|
kwargs = {"timeout": timeout}
|
|
if transport is not None:
|
|
kwargs["transport"] = transport
|
|
async with shared_client(**kwargs) as client:
|
|
resp = await _get_checked(client, src, headers)
|
|
except ValueError:
|
|
# A12 : hôte privé/loopback ou trop de redirections → refus explicite.
|
|
raise
|
|
except Exception as exc: # noqa: BLE001 - network/parse failures are non-fatal
|
|
logger.debug("og fetch failed for %s: %s", src, exc)
|
|
base["title"] = urlparse(src).netloc or src
|
|
base["site_name"] = _site_name(src)
|
|
return base
|
|
|
|
ctype = (resp.headers.get("content-type") or "").lower()
|
|
if "text/html" not in ctype and "xhtml" not in ctype:
|
|
base["title"] = urlparse(src).netloc or src
|
|
base["site_name"] = _site_name(src)
|
|
return base
|
|
|
|
return parse_og(resp.text, src)
|