- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
95 lines
3.3 KiB
Python
95 lines
3.3 KiB
Python
"""FlowDeck — URL / web clipper importer (v5.6.0, Phase 5).
|
|
|
|
Fetches a web page and turns it into a page: a bookmark card (OG metadata)
|
|
followed by the article converted to FlowDeck blocks.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import ipaddress
|
|
import socket
|
|
from urllib.parse import urlparse
|
|
|
|
import httpx
|
|
|
|
from app.services.export import markdown_to_blocks
|
|
from app.services.importers.base import ImportPage, ImportResult
|
|
from app.services.importers.html_notes import _html_to_markdown
|
|
|
|
_BLOCKED_HOSTS = {"localhost", "localhost.localdomain"}
|
|
_MAX_BYTES = 3_000_000
|
|
|
|
|
|
def _is_public_host(host: str) -> bool:
|
|
"""SSRF guard: reject loopback/private/link-local/reserved addresses."""
|
|
if not host or host.lower() in _BLOCKED_HOSTS:
|
|
return False
|
|
try:
|
|
infos = socket.getaddrinfo(host, None)
|
|
except socket.gaierror:
|
|
return False
|
|
for info in infos:
|
|
address = info[4][0]
|
|
try:
|
|
ip = ipaddress.ip_address(address)
|
|
except ValueError:
|
|
return False
|
|
if (ip.is_private or ip.is_loopback or ip.is_link_local
|
|
or ip.is_reserved or ip.is_multicast or ip.is_unspecified):
|
|
return False
|
|
return True
|
|
|
|
|
|
def _validate_url(url: str) -> str:
|
|
parsed = urlparse(url.strip())
|
|
if parsed.scheme not in ("http", "https"):
|
|
raise ValueError("Seules les URLs http(s) sont autorisées")
|
|
if not parsed.hostname or not _is_public_host(parsed.hostname):
|
|
raise ValueError("Hôte non autorisé")
|
|
return url.strip()
|
|
|
|
|
|
def _bookmark_block(url: str, meta: dict) -> dict:
|
|
block = {"type": "bookmark", "url": url}
|
|
for key in ("title", "description", "image", "site_name"):
|
|
if meta.get(key):
|
|
block[key] = meta[key]
|
|
return block
|
|
|
|
|
|
async def fetch_url_result(url: str, *, transport: httpx.BaseTransport | None = None) -> ImportResult:
|
|
"""Fetch ``url`` and build a single-page ImportResult (raises on bad URL)."""
|
|
safe_url = _validate_url(url)
|
|
result = ImportResult(source="url")
|
|
try:
|
|
async with httpx.AsyncClient(
|
|
timeout=15, follow_redirects=True, transport=transport,
|
|
headers={"User-Agent": "FlowDeck-Importer/1.0"},
|
|
) as client:
|
|
response = await client.get(safe_url)
|
|
response.raise_for_status()
|
|
if response.url.host and not _is_public_host(response.url.host):
|
|
raise ValueError("Redirection vers un hôte non autorisé")
|
|
content_type = response.headers.get("content-type", "")
|
|
if "html" not in content_type.lower():
|
|
raise ValueError("La ressource n'est pas une page HTML")
|
|
body = response.text[:_MAX_BYTES]
|
|
except httpx.HTTPError as exc:
|
|
result.warn(f"Échec du téléchargement : {exc}")
|
|
return result.finalize()
|
|
|
|
from app.services.og_fetcher import parse_og
|
|
|
|
meta = parse_og(body, safe_url)
|
|
title = (meta.get("title") or urlparse(safe_url).hostname or "Page").strip()
|
|
markdown = _html_to_markdown(body)
|
|
blocks = [_bookmark_block(safe_url, meta)]
|
|
if markdown:
|
|
blocks.extend(markdown_to_blocks(markdown))
|
|
result.pages.append(ImportPage(
|
|
title=title[:200],
|
|
blocks=blocks,
|
|
source_path=safe_url,
|
|
external_id=safe_url,
|
|
))
|
|
return result.finalize()
|