Files
flowdeck/app/services/og_fetcher.py
T
bruno 5a537f5dc3
FlowDeck CI / docker (push) Successful in 1m44s
FlowDeck CI / lint (push) Successful in 1m53s
FlowDeck CI / test (push) Successful in 20m45s
fix: A12–A24 — SSRF, auth routes legacy, uploads, N+1 et routes doublonnes (v7.3.3)
- A12 — `og_fetcher` : GET sans `follow_redirects`, `_is_public_host` revérifié à
  chaque saut (max 5) ; `POST /board/api/og/metadata` → 400 sur hôte privé/loopback
- A13 — router automations sous `Depends(_require_session)` (CRUD, run,
  press-button) + `created_by` sans fallback ; action `webhook` validée par
  `_is_public_host` avant POST (SSRF)
- A15 — webhooks sortants : `_require_admin` sur GET/POST/DELETE + `_is_public_host`
  sur l'URL en création
- A17 — router legacy `/api` sous `Depends(_require_session_or_bearer)` (session ou
  Bearer `/api/v1`), allowlist explicite `/api/health` + `/api/frontend-error`
- A22 — les 2 uploads locales : session exigée (`_require_user_id`) + `validate_upload`
  branché (taille + extension) + `FLOWDECK_DATA_DIR` au lieu de `/data` codé en dur
- A23 — N+1 : COUNT→`GROUP BY` (dashboard), cards→`executemany` (board sync),
  duplicata de propriétés→`executemany` + remap des ids par SELECT (collections)
- A24 — 2 routes écrasées supprimées : `GET /api/projects` (api.py) et
  `GET /workspace` (workspace.py) + test « aucun doublon méthode+chemin »
- Tests : +9 dans `tests/test_audit_p0_fixes.py` (SSRF, 401s, validate_upload,
  doublons de routes) ; tests OG sur hôtes résolubles (la garde fait du DNS)
- suite **1025/1025** · `ruff check app tests` OK
2026-09-30 23:12:20 -04:00

173 lines
6.1 KiB
Python

"""FlowDeck — Bookmark cards (v5.5.0): Open Graph metadata via httpx.
Fetches a URL server-side, extracts OG/Twitter meta tags (title, description,
image, site name, favicon) and returns a safe, compact payload used to render
Notion-style bookmark cards. Robust to missing tags, non-HTML bodies and
slow/unreachable hosts.
"""
from __future__ import annotations
import html as htmlmod
import logging
import re
from urllib.parse import urljoin, urlparse
logger = logging.getLogger(__name__)
_META_TAG_RE = re.compile(r"<meta\b[^>]*?>", re.I)
_ATTR_RE = re.compile(r"([A-Za-z_:][-A-Za-z0-9_:.]*)\s*=\s*[\"']([^\"']*)[\"']")
_TITLE_RE = re.compile(r"<title[^>]*>(.*?)</title>", re.I | re.S)
_FAVICON_RE = re.compile(r"<link\b[^>]*?>", re.I)
_ICON_REL = re.compile(r"\b(?:shortcut\s+)?icon\b", re.I)
# Property/name keys we look for, in priority order, mapped to our payload keys.
_OG_TITLE = ("og:title", "twitter:title", "title", "og:site_name")
_OG_DESC = ("og:description", "twitter:description", "description")
_OG_IMG = ("og:image", "twitter:image", "twitter:image:src", "image")
_OG_SITE = ("og:site_name", "twitter:site", "application-name")
def _attrs(tag: str) -> dict:
return {k.lower(): v for k, v in _ATTR_RE.findall(tag)}
def _extract_og(body: str) -> dict:
"""Parse all ``<meta>`` tags into a ``{key: content}`` dict.
Attributes may appear in any order (``content`` before or after
``property``/``name``), which the previous implementation mishandled.
First value wins so the most specific tag (top of document) is kept.
"""
props: dict[str, str] = {}
for tag in _META_TAG_RE.finditer(body[:400_000]):
attrs = _attrs(tag.group(0))
key = (attrs.get("property") or attrs.get("name") or attrs.get("itemprop") or "").lower()
content = attrs.get("content")
if key and content is not None and key not in props:
props[key] = content
return props
def _pick(props: dict, keys: tuple) -> str:
for k in keys:
v = props.get(k)
if v:
return v
return ""
def _title_of(props: dict, body: str) -> str:
t = _pick(props, _OG_TITLE)
if t:
return t
m = _TITLE_RE.search(body[:200_000])
return m.group(1).strip() if m else ""
def _site_name(url: str) -> str:
host = urlparse(url).netloc.replace("www.", "")
return host.split(".")[0].capitalize() if host else ""
def _favicon(body: str, base_url: str) -> str:
for tag in _FAVICON_RE.finditer(body):
attrs = _attrs(tag.group(0))
rel = attrs.get("rel", "")
href = attrs.get("href", "")
if href and _ICON_REL.search(rel):
return urljoin(base_url, htmlmod.unescape(href))
return ""
def parse_og(body: str, url: str) -> dict:
"""Pure HTML → bookmark payload (no network). ``url`` is the base URL."""
src = url.strip()
if not src.startswith(("http://", "https://")):
src = "https://" + src
props = _extract_og(body)
title = htmlmod.unescape(_title_of(props, body))
desc = htmlmod.unescape(_pick(props, _OG_DESC))
img = _pick(props, _OG_IMG)
site = htmlmod.unescape(_pick(props, _OG_SITE)) or _site_name(src)
def abs_url(u: str) -> str:
return urljoin(src, htmlmod.unescape(u)) if u else ""
return {
"url": src,
"title": title.strip()[:200] or urlparse(src).netloc or src,
"description": desc.strip()[:400],
"image": abs_url(img),
"site_name": site.strip()[:100],
"favicon": _favicon(body, src),
}
_MAX_REDIRECTS = 5
async def _get_checked(client, url: str, headers: dict):
"""GET avec re-vérification de l'hôte à CHAQUE saut de redirection (A12 SSRF).
`follow_redirects=True` laisserait une URL publique rediriger vers
169.254.169.254 / localhost — la garde doit donc tourner à chaque hop.
"""
from app.services.importers.url_fetch import _is_public_host
current = url
for _ in range(_MAX_REDIRECTS + 1):
parsed = urlparse(current)
if parsed.scheme not in ("http", "https") or not parsed.hostname or not _is_public_host(parsed.hostname):
raise ValueError(f"hôte non autorisé: {parsed.hostname!r}")
r = await client.get(current, headers=headers, follow_redirects=False)
if r.status_code in (301, 302, 303, 307, 308):
loc = r.headers.get("location")
if not loc:
return r
current = urljoin(current, loc)
continue
r.raise_for_status()
return r
raise ValueError("trop de redirections")
async def fetch_og_metadata(url: str, timeout: float = 6.0, transport=None) -> dict:
"""Fetch ``url`` and return {url, title, description, image, site_name,
favicon}. Empty strings are omitted. Never raises for network errors.
``transport`` is an optional ``httpx`` transport (used by tests to mock
HTTP without hitting the network).
"""
src = url.strip()
if not src.startswith(("http://", "https://")):
src = "https://" + src
base = {"url": src, "title": "", "description": "", "image": "", "site_name": "", "favicon": ""}
try:
import httpx
headers = {
"User-Agent": "FlowDeck/5.5 bookmark-fetcher (+https://flowdeck.dracodev.net)",
"Accept": "text/html,application/xhtml+xml",
}
kwargs = {"timeout": timeout}
if transport is not None:
kwargs["transport"] = transport
async with httpx.AsyncClient(**kwargs) as client:
resp = await _get_checked(client, src, headers)
except ValueError:
# A12 : hôte privé/loopback ou trop de redirections → refus explicite.
raise
except Exception as exc: # noqa: BLE001 - network/parse failures are non-fatal
logger.debug("og fetch failed for %s: %s", src, exc)
base["title"] = urlparse(src).netloc or src
base["site_name"] = _site_name(src)
return base
ctype = (resp.headers.get("content-type") or "").lower()
if "text/html" not in ctype and "xhtml" not in ctype:
base["title"] = urlparse(src).netloc or src
base["site_name"] = _site_name(src)
return base
return parse_og(resp.text, src)