- embeds.py: provider detection/rewrite (YouTube, Vimeo, Figma, Maps, Docs, Loom, CodePen, Miro, Spotify, SoundCloud, Twitch, X/Twitter, Pinterest, Office) + resolve_embed/inline_kind/provider; POST /board/api/embed/resolve - editor resolves pasted URLs and caches embed_src (persisted); renderer, public pages and MD/HTML/PDF exports prefer embed_src - og_fetcher.py: robust meta parsing (any attribute order), favicon, injectable transport, network-safe fallback - image lightbox with keyboard nav (arrows/Esc) in editor and public pages - inline PDF/video/audio previews - cover (URL or upload) & page icon endpoints - fix broken editor API paths (/api/pages -> /board/api/pages) for cover, icon, versions, backlinks, import, move and OG metadata - 47 tests in tests/test_v55.py; full suite 444 green; ruff clean - version 5.11.2
143 lines
5.0 KiB
Python
143 lines
5.0 KiB
Python
"""FlowDeck — Bookmark cards (v5.5.0): Open Graph metadata via httpx.
|
|
|
|
Fetches a URL server-side, extracts OG/Twitter meta tags (title, description,
|
|
image, site name, favicon) and returns a safe, compact payload used to render
|
|
Notion-style bookmark cards. Robust to missing tags, non-HTML bodies and
|
|
slow/unreachable hosts.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import html as htmlmod
|
|
import logging
|
|
import re
|
|
from urllib.parse import urljoin, urlparse
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_META_TAG_RE = re.compile(r"<meta\b[^>]*?>", re.I)
|
|
_ATTR_RE = re.compile(r"([A-Za-z_:][-A-Za-z0-9_:.]*)\s*=\s*[\"']([^\"']*)[\"']")
|
|
_TITLE_RE = re.compile(r"<title[^>]*>(.*?)</title>", re.I | re.S)
|
|
_FAVICON_RE = re.compile(r"<link\b[^>]*?>", re.I)
|
|
_ICON_REL = re.compile(r"\b(?:shortcut\s+)?icon\b", re.I)
|
|
|
|
# Property/name keys we look for, in priority order, mapped to our payload keys.
|
|
_OG_TITLE = ("og:title", "twitter:title", "title", "og:site_name")
|
|
_OG_DESC = ("og:description", "twitter:description", "description")
|
|
_OG_IMG = ("og:image", "twitter:image", "twitter:image:src", "image")
|
|
_OG_SITE = ("og:site_name", "twitter:site", "application-name")
|
|
|
|
|
|
def _attrs(tag: str) -> dict:
|
|
return {k.lower(): v for k, v in _ATTR_RE.findall(tag)}
|
|
|
|
|
|
def _extract_og(body: str) -> dict:
|
|
"""Parse all ``<meta>`` tags into a ``{key: content}`` dict.
|
|
|
|
Attributes may appear in any order (``content`` before or after
|
|
``property``/``name``), which the previous implementation mishandled.
|
|
First value wins so the most specific tag (top of document) is kept.
|
|
"""
|
|
props: dict[str, str] = {}
|
|
for tag in _META_TAG_RE.finditer(body[:400_000]):
|
|
attrs = _attrs(tag.group(0))
|
|
key = (attrs.get("property") or attrs.get("name") or attrs.get("itemprop") or "").lower()
|
|
content = attrs.get("content")
|
|
if key and content is not None and key not in props:
|
|
props[key] = content
|
|
return props
|
|
|
|
|
|
def _pick(props: dict, keys: tuple) -> str:
|
|
for k in keys:
|
|
v = props.get(k)
|
|
if v:
|
|
return v
|
|
return ""
|
|
|
|
|
|
def _title_of(props: dict, body: str) -> str:
|
|
t = _pick(props, _OG_TITLE)
|
|
if t:
|
|
return t
|
|
m = _TITLE_RE.search(body[:200_000])
|
|
return m.group(1).strip() if m else ""
|
|
|
|
|
|
def _site_name(url: str) -> str:
|
|
host = urlparse(url).netloc.replace("www.", "")
|
|
return host.split(".")[0].capitalize() if host else ""
|
|
|
|
|
|
def _favicon(body: str, base_url: str) -> str:
|
|
for tag in _FAVICON_RE.finditer(body):
|
|
attrs = _attrs(tag.group(0))
|
|
rel = attrs.get("rel", "")
|
|
href = attrs.get("href", "")
|
|
if href and _ICON_REL.search(rel):
|
|
return urljoin(base_url, htmlmod.unescape(href))
|
|
return ""
|
|
|
|
|
|
def parse_og(body: str, url: str) -> dict:
|
|
"""Pure HTML → bookmark payload (no network). ``url`` is the base URL."""
|
|
src = url.strip()
|
|
if not src.startswith(("http://", "https://")):
|
|
src = "https://" + src
|
|
props = _extract_og(body)
|
|
title = htmlmod.unescape(_title_of(props, body))
|
|
desc = htmlmod.unescape(_pick(props, _OG_DESC))
|
|
img = _pick(props, _OG_IMG)
|
|
site = htmlmod.unescape(_pick(props, _OG_SITE)) or _site_name(src)
|
|
|
|
def abs_url(u: str) -> str:
|
|
return urljoin(src, htmlmod.unescape(u)) if u else ""
|
|
|
|
return {
|
|
"url": src,
|
|
"title": title.strip()[:200] or urlparse(src).netloc or src,
|
|
"description": desc.strip()[:400],
|
|
"image": abs_url(img),
|
|
"site_name": site.strip()[:100],
|
|
"favicon": _favicon(body, src),
|
|
}
|
|
|
|
|
|
async def fetch_og_metadata(url: str, timeout: float = 6.0, transport=None) -> dict:
|
|
"""Fetch ``url`` and return {url, title, description, image, site_name,
|
|
favicon}. Empty strings are omitted. Never raises for network errors.
|
|
|
|
``transport`` is an optional ``httpx`` transport (used by tests to mock
|
|
HTTP without hitting the network).
|
|
"""
|
|
src = url.strip()
|
|
if not src.startswith(("http://", "https://")):
|
|
src = "https://" + src
|
|
base = {"url": src, "title": "", "description": "", "image": "", "site_name": "", "favicon": ""}
|
|
try:
|
|
import httpx
|
|
|
|
headers = {
|
|
"User-Agent": "FlowDeck/5.5 bookmark-fetcher (+https://flowdeck.dracodev.net)",
|
|
"Accept": "text/html,application/xhtml+xml",
|
|
}
|
|
kwargs = {"follow_redirects": True, "timeout": timeout}
|
|
if transport is not None:
|
|
kwargs["transport"] = transport
|
|
async with httpx.AsyncClient(**kwargs) as client:
|
|
resp = await client.get(src, headers=headers)
|
|
resp.raise_for_status()
|
|
except Exception as exc: # noqa: BLE001 - network/parse failures are non-fatal
|
|
logger.debug("og fetch failed for %s: %s", src, exc)
|
|
base["title"] = urlparse(src).netloc or src
|
|
base["site_name"] = _site_name(src)
|
|
return base
|
|
|
|
ctype = (resp.headers.get("content-type") or "").lower()
|
|
if "text/html" not in ctype and "xhtml" not in ctype:
|
|
base["title"] = urlparse(src).netloc or src
|
|
base["site_name"] = _site_name(src)
|
|
return base
|
|
|
|
return parse_og(resp.text, src)
|