- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
136 lines
5.1 KiB
Python
136 lines
5.1 KiB
Python
"""FlowDeck — Notion export importer (v5.6.0, Phase 1, amélioration v5.4.0).
|
|
|
|
Imports a Notion "Export as Markdown & CSV" ``.zip``: complete page hierarchy,
|
|
databases (``.csv``) turned into FlowDeck collections, and image attachments.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
import io
|
|
import re
|
|
import zipfile
|
|
from urllib.parse import unquote
|
|
|
|
from app.services.importers._common import split_frontmatter
|
|
from app.services.importers.base import (
|
|
ImportAttachment,
|
|
Importer,
|
|
ImportPage,
|
|
ImportResult,
|
|
decode_text,
|
|
register_importer,
|
|
)
|
|
from app.services.importers.tabular import rows_to_collection
|
|
|
|
_HASH_RE = re.compile(r"\s+[0-9a-f]{32}$")
|
|
_MD_LINK_RE = re.compile(r"\]\(([^)]+)\.md\)")
|
|
|
|
|
|
def _clean_name(name: str) -> str:
|
|
base = unquote(name.replace("\\", "/").rsplit("/", 1)[-1])
|
|
base = re.sub(r"\.(md|csv|markdown)$", "", base, flags=re.IGNORECASE)
|
|
return _HASH_RE.sub("", base).strip() or "Untitled"
|
|
|
|
|
|
def _strip_hash_link(match: re.Match) -> str:
|
|
target = unquote(match.group(1)).strip()
|
|
return f"]({_HASH_RE.sub('', target).strip() or target})"
|
|
|
|
|
|
@register_importer
|
|
class NotionImporter(Importer):
|
|
source_id = "notion"
|
|
label = "Notion (export .zip)"
|
|
description = "Export Notion Markdown & CSV : hiérarchie, databases → collections, images."
|
|
extensions = (".zip",)
|
|
order = 20
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
if not filename.lower().endswith(".zip"):
|
|
return False
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError):
|
|
return False
|
|
names = [n for n in zf.namelist() if not n.endswith("/")]
|
|
md = [n for n in names if n.lower().endswith(".md")]
|
|
csvs = [n for n in names if n.lower().endswith(".csv")]
|
|
if not md:
|
|
return False
|
|
if csvs:
|
|
return True
|
|
return any(_HASH_RE.search(unquote(n.rsplit("/", 1)[-1])) for n in md)
|
|
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
result = ImportResult(source=self.source_id)
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError) as exc:
|
|
result.warn(f"Archive invalide : {exc}")
|
|
return result.finalize()
|
|
|
|
names = [n for n in zf.namelist() if not n.endswith("/")]
|
|
md_names = [n for n in names if n.lower().endswith(".md")]
|
|
csv_names = [n for n in names if n.lower().endswith(".csv")]
|
|
|
|
pages_by_title: dict[str, ImportPage] = {}
|
|
for name in sorted(md_names, key=lambda n: (n.count("/"), n.lower())):
|
|
try:
|
|
text = decode_text(zf.read(name))
|
|
except Exception as exc: # noqa: BLE001
|
|
result.warn(f"Lecture impossible : {name} ({exc})")
|
|
continue
|
|
meta, body = split_frontmatter(text)
|
|
title = _clean_name(name)
|
|
body = _MD_LINK_RE.sub(_strip_hash_link, body)
|
|
first_line = body.lstrip().splitlines()[0] if body.strip() else ""
|
|
if first_line.strip().startswith("# ") and first_line.strip()[2:].strip() == title:
|
|
body = "\n".join(body.lstrip().splitlines()[1:]).lstrip("\n")
|
|
parts = unquote(name).replace("\\", "/").split("/")
|
|
page = ImportPage(
|
|
title=title,
|
|
markdown=body,
|
|
source_path=name,
|
|
parent_path="/".join(parts[:-1]),
|
|
properties={k: v for k, v in meta.items() if k != "title"},
|
|
external_id=name,
|
|
)
|
|
pages_by_title.setdefault(title, page)
|
|
result.pages.append(page)
|
|
|
|
for name in sorted(csv_names, key=lambda n: (n.count("/"), n.lower())):
|
|
try:
|
|
text = decode_text(zf.read(name))
|
|
except Exception as exc: # noqa: BLE001
|
|
result.warn(f"Lecture impossible : {name} ({exc})")
|
|
continue
|
|
reader = csv.DictReader(io.StringIO(text))
|
|
headers = [h for h in (reader.fieldnames or []) if h is not None]
|
|
rows = [dict(r) for r in reader]
|
|
title = _clean_name(name)
|
|
spec_page = rows_to_collection(title, headers, rows)
|
|
parts = unquote(name).replace("\\", "/").split("/")
|
|
existing = pages_by_title.get(title)
|
|
if existing is not None:
|
|
existing.collection = spec_page.collection
|
|
existing.source_path = existing.source_path or name
|
|
else:
|
|
spec_page.parent_path = "/".join(parts[:-1])
|
|
spec_page.source_path = name
|
|
spec_page.external_id = name
|
|
result.pages.append(spec_page)
|
|
|
|
for name in names:
|
|
low = name.lower()
|
|
if low.endswith((".md", ".csv")):
|
|
continue
|
|
try:
|
|
result.attachments.append(ImportAttachment(
|
|
source_path=name,
|
|
filename=unquote(name).replace("\\", "/").rsplit("/", 1)[-1],
|
|
data=zf.read(name),
|
|
))
|
|
except Exception: # noqa: BLE001
|
|
continue
|
|
return result.finalize()
|