Files
flowdeck/app/services/og_fetcher.py
T
bruno 3bb8e87ef2
FlowDeck CI / lint (push) Canceled after 0s
FlowDeck CI / test (push) Canceled after 0s
FlowDeck CI / docker (push) Canceled after 0s
fix: A42 terminé — client httpx partagé par boucle (v7.28.0)
- `app/services/http_client.py` : `async with shared_client(timeout=15)
  as client:` remplace les 49 créations `async with httpx.AsyncClient(`
  de 14 fichiers (gitea ×21, providers oidc/oauth ×11, calendar ×4,
  automations ×3…) — le pool de connexions est réutilisé au lieu d'être
  recréé à chaque appel. __aexit__ no-op (le client partagé ne se ferme
  pas à la sortie).
- Cache par (boucle d'event, kwargs) en WeakKeyDictionary : un
  AsyncClient n'est JAMAIS partagé entre deux loops (piège des tests
  « Event loop is closed ») — une boucle par test = client propre
  collecté avec la boucle. Clé = kwargs triés, repr() pour les valeurs
  non hashables (`headers=` dict → TypeError rattrapé par la suite).
- Laissés délibérément : github_adapter (transport MockTransport
  injecté), webhook_outbound (client « own_client » fermé par la
  fonction).
- Tests : `test_http_client_shared_and_loop_scoped` (réutilisation mêmes
  kwargs / cloisonné kwargs / cloisonné loop) ; le stub des webhooks
  patche aussi la fabrique `http_client.httpx` + purge du cache (avant :
  webhook_outbound.httpx patché mais la fabrique partagée créait un vrai
  client → réseau réel dans les tests).

suite **1091/1091** · ruff OK · docs à jour
2026-10-01 23:09:45 -04:00

174 lines
6.2 KiB
Python

"""FlowDeck — Bookmark cards (v5.5.0): Open Graph metadata via httpx.
Fetches a URL server-side, extracts OG/Twitter meta tags (title, description,
image, site name, favicon) and returns a safe, compact payload used to render
Notion-style bookmark cards. Robust to missing tags, non-HTML bodies and
slow/unreachable hosts.
"""
from __future__ import annotations
import html as htmlmod
import logging
import re
from urllib.parse import urljoin, urlparse
from app.services.http_client import shared_client
logger = logging.getLogger(__name__)
_META_TAG_RE = re.compile(r"<meta\b[^>]*?>", re.I)
_ATTR_RE = re.compile(r"([A-Za-z_:][-A-Za-z0-9_:.]*)\s*=\s*[\"']([^\"']*)[\"']")
_TITLE_RE = re.compile(r"<title[^>]*>(.*?)</title>", re.I | re.S)
_FAVICON_RE = re.compile(r"<link\b[^>]*?>", re.I)
_ICON_REL = re.compile(r"\b(?:shortcut\s+)?icon\b", re.I)
# Property/name keys we look for, in priority order, mapped to our payload keys.
_OG_TITLE = ("og:title", "twitter:title", "title", "og:site_name")
_OG_DESC = ("og:description", "twitter:description", "description")
_OG_IMG = ("og:image", "twitter:image", "twitter:image:src", "image")
_OG_SITE = ("og:site_name", "twitter:site", "application-name")
def _attrs(tag: str) -> dict:
return {k.lower(): v for k, v in _ATTR_RE.findall(tag)}
def _extract_og(body: str) -> dict:
"""Parse all ``<meta>`` tags into a ``{key: content}`` dict.
Attributes may appear in any order (``content`` before or after
``property``/``name``), which the previous implementation mishandled.
First value wins so the most specific tag (top of document) is kept.
"""
props: dict[str, str] = {}
for tag in _META_TAG_RE.finditer(body[:400_000]):
attrs = _attrs(tag.group(0))
key = (attrs.get("property") or attrs.get("name") or attrs.get("itemprop") or "").lower()
content = attrs.get("content")
if key and content is not None and key not in props:
props[key] = content
return props
def _pick(props: dict, keys: tuple) -> str:
for k in keys:
v = props.get(k)
if v:
return v
return ""
def _title_of(props: dict, body: str) -> str:
t = _pick(props, _OG_TITLE)
if t:
return t
m = _TITLE_RE.search(body[:200_000])
return m.group(1).strip() if m else ""
def _site_name(url: str) -> str:
host = urlparse(url).netloc.replace("www.", "")
return host.split(".")[0].capitalize() if host else ""
def _favicon(body: str, base_url: str) -> str:
for tag in _FAVICON_RE.finditer(body):
attrs = _attrs(tag.group(0))
rel = attrs.get("rel", "")
href = attrs.get("href", "")
if href and _ICON_REL.search(rel):
return urljoin(base_url, htmlmod.unescape(href))
return ""
def parse_og(body: str, url: str) -> dict:
"""Pure HTML → bookmark payload (no network). ``url`` is the base URL."""
src = url.strip()
if not src.startswith(("http://", "https://")):
src = "https://" + src
props = _extract_og(body)
title = htmlmod.unescape(_title_of(props, body))
desc = htmlmod.unescape(_pick(props, _OG_DESC))
img = _pick(props, _OG_IMG)
site = htmlmod.unescape(_pick(props, _OG_SITE)) or _site_name(src)
def abs_url(u: str) -> str:
return urljoin(src, htmlmod.unescape(u)) if u else ""
return {
"url": src,
"title": title.strip()[:200] or urlparse(src).netloc or src,
"description": desc.strip()[:400],
"image": abs_url(img),
"site_name": site.strip()[:100],
"favicon": _favicon(body, src),
}
_MAX_REDIRECTS = 5
async def _get_checked(client, url: str, headers: dict):
"""GET avec re-vérification de l'hôte à CHAQUE saut de redirection (A12 SSRF).
`follow_redirects=True` laisserait une URL publique rediriger vers
169.254.169.254 / localhost — la garde doit donc tourner à chaque hop.
"""
from app.services.importers.url_fetch import _is_public_host
current = url
for _ in range(_MAX_REDIRECTS + 1):
parsed = urlparse(current)
if parsed.scheme not in ("http", "https") or not parsed.hostname or not _is_public_host(parsed.hostname):
raise ValueError(f"hôte non autorisé: {parsed.hostname!r}")
r = await client.get(current, headers=headers, follow_redirects=False)
if r.status_code in (301, 302, 303, 307, 308):
loc = r.headers.get("location")
if not loc:
return r
current = urljoin(current, loc)
continue
r.raise_for_status()
return r
raise ValueError("trop de redirections")
async def fetch_og_metadata(url: str, timeout: float = 6.0, transport=None) -> dict:
"""Fetch ``url`` and return {url, title, description, image, site_name,
favicon}. Empty strings are omitted. Never raises for network errors.
``transport`` is an optional ``httpx`` transport (used by tests to mock
HTTP without hitting the network).
"""
src = url.strip()
if not src.startswith(("http://", "https://")):
src = "https://" + src
base = {"url": src, "title": "", "description": "", "image": "", "site_name": "", "favicon": ""}
try:
headers = {
"User-Agent": "FlowDeck/5.5 bookmark-fetcher (+https://flowdeck.dracodev.net)",
"Accept": "text/html,application/xhtml+xml",
}
kwargs = {"timeout": timeout}
if transport is not None:
kwargs["transport"] = transport
async with shared_client(**kwargs) as client:
resp = await _get_checked(client, src, headers)
except ValueError:
# A12 : hôte privé/loopback ou trop de redirections → refus explicite.
raise
except Exception as exc: # noqa: BLE001 - network/parse failures are non-fatal
logger.debug("og fetch failed for %s: %s", src, exc)
base["title"] = urlparse(src).netloc or src
base["site_name"] = _site_name(src)
return base
ctype = (resp.headers.get("content-type") or "").lower()
if "text/html" not in ctype and "xhtml" not in ctype:
base["title"] = urlparse(src).netloc or src
base["site_name"] = _site_name(src)
return base
return parse_og(resp.text, src)