Files
flowdeck/app/services/og_fetcher.py
T
bruno bdc15c7328
FlowDeck CI / lint (push) Successful in 44s
FlowDeck CI / test (push) Successful in 4m10s
FlowDeck CI / docker (push) Successful in 37s
feat(v5.5.0): Embeds & Media riche - universal embeds, bookmark cards, lightbox, inline previews, cover & icon
- embeds.py: provider detection/rewrite (YouTube, Vimeo, Figma, Maps, Docs,
  Loom, CodePen, Miro, Spotify, SoundCloud, Twitch, X/Twitter, Pinterest,
  Office) + resolve_embed/inline_kind/provider; POST /board/api/embed/resolve
- editor resolves pasted URLs and caches embed_src (persisted); renderer,
  public pages and MD/HTML/PDF exports prefer embed_src
- og_fetcher.py: robust meta parsing (any attribute order), favicon,
  injectable transport, network-safe fallback
- image lightbox with keyboard nav (arrows/Esc) in editor and public pages
- inline PDF/video/audio previews
- cover (URL or upload) & page icon endpoints
- fix broken editor API paths (/api/pages -> /board/api/pages) for cover,
  icon, versions, backlinks, import, move and OG metadata
- 47 tests in tests/test_v55.py; full suite 444 green; ruff clean
- version 5.11.2
2026-09-12 09:40:50 -04:00

143 lines
5.0 KiB
Python

"""FlowDeck — Bookmark cards (v5.5.0): Open Graph metadata via httpx.
Fetches a URL server-side, extracts OG/Twitter meta tags (title, description,
image, site name, favicon) and returns a safe, compact payload used to render
Notion-style bookmark cards. Robust to missing tags, non-HTML bodies and
slow/unreachable hosts.
"""
from __future__ import annotations
import html as htmlmod
import logging
import re
from urllib.parse import urljoin, urlparse
logger = logging.getLogger(__name__)
_META_TAG_RE = re.compile(r"<meta\b[^>]*?>", re.I)
_ATTR_RE = re.compile(r"([A-Za-z_:][-A-Za-z0-9_:.]*)\s*=\s*[\"']([^\"']*)[\"']")
_TITLE_RE = re.compile(r"<title[^>]*>(.*?)</title>", re.I | re.S)
_FAVICON_RE = re.compile(r"<link\b[^>]*?>", re.I)
_ICON_REL = re.compile(r"\b(?:shortcut\s+)?icon\b", re.I)
# Property/name keys we look for, in priority order, mapped to our payload keys.
_OG_TITLE = ("og:title", "twitter:title", "title", "og:site_name")
_OG_DESC = ("og:description", "twitter:description", "description")
_OG_IMG = ("og:image", "twitter:image", "twitter:image:src", "image")
_OG_SITE = ("og:site_name", "twitter:site", "application-name")
def _attrs(tag: str) -> dict:
return {k.lower(): v for k, v in _ATTR_RE.findall(tag)}
def _extract_og(body: str) -> dict:
"""Parse all ``<meta>`` tags into a ``{key: content}`` dict.
Attributes may appear in any order (``content`` before or after
``property``/``name``), which the previous implementation mishandled.
First value wins so the most specific tag (top of document) is kept.
"""
props: dict[str, str] = {}
for tag in _META_TAG_RE.finditer(body[:400_000]):
attrs = _attrs(tag.group(0))
key = (attrs.get("property") or attrs.get("name") or attrs.get("itemprop") or "").lower()
content = attrs.get("content")
if key and content is not None and key not in props:
props[key] = content
return props
def _pick(props: dict, keys: tuple) -> str:
for k in keys:
v = props.get(k)
if v:
return v
return ""
def _title_of(props: dict, body: str) -> str:
t = _pick(props, _OG_TITLE)
if t:
return t
m = _TITLE_RE.search(body[:200_000])
return m.group(1).strip() if m else ""
def _site_name(url: str) -> str:
host = urlparse(url).netloc.replace("www.", "")
return host.split(".")[0].capitalize() if host else ""
def _favicon(body: str, base_url: str) -> str:
for tag in _FAVICON_RE.finditer(body):
attrs = _attrs(tag.group(0))
rel = attrs.get("rel", "")
href = attrs.get("href", "")
if href and _ICON_REL.search(rel):
return urljoin(base_url, htmlmod.unescape(href))
return ""
def parse_og(body: str, url: str) -> dict:
"""Pure HTML → bookmark payload (no network). ``url`` is the base URL."""
src = url.strip()
if not src.startswith(("http://", "https://")):
src = "https://" + src
props = _extract_og(body)
title = htmlmod.unescape(_title_of(props, body))
desc = htmlmod.unescape(_pick(props, _OG_DESC))
img = _pick(props, _OG_IMG)
site = htmlmod.unescape(_pick(props, _OG_SITE)) or _site_name(src)
def abs_url(u: str) -> str:
return urljoin(src, htmlmod.unescape(u)) if u else ""
return {
"url": src,
"title": title.strip()[:200] or urlparse(src).netloc or src,
"description": desc.strip()[:400],
"image": abs_url(img),
"site_name": site.strip()[:100],
"favicon": _favicon(body, src),
}
async def fetch_og_metadata(url: str, timeout: float = 6.0, transport=None) -> dict:
"""Fetch ``url`` and return {url, title, description, image, site_name,
favicon}. Empty strings are omitted. Never raises for network errors.
``transport`` is an optional ``httpx`` transport (used by tests to mock
HTTP without hitting the network).
"""
src = url.strip()
if not src.startswith(("http://", "https://")):
src = "https://" + src
base = {"url": src, "title": "", "description": "", "image": "", "site_name": "", "favicon": ""}
try:
import httpx
headers = {
"User-Agent": "FlowDeck/5.5 bookmark-fetcher (+https://flowdeck.dracodev.net)",
"Accept": "text/html,application/xhtml+xml",
}
kwargs = {"follow_redirects": True, "timeout": timeout}
if transport is not None:
kwargs["transport"] = transport
async with httpx.AsyncClient(**kwargs) as client:
resp = await client.get(src, headers=headers)
resp.raise_for_status()
except Exception as exc: # noqa: BLE001 - network/parse failures are non-fatal
logger.debug("og fetch failed for %s: %s", src, exc)
base["title"] = urlparse(src).netloc or src
base["site_name"] = _site_name(src)
return base
ctype = (resp.headers.get("content-type") or "").lower()
if "text/html" not in ctype and "xhtml" not in ctype:
base["title"] = urlparse(src).netloc or src
base["site_name"] = _site_name(src)
return base
return parse_og(resp.text, src)