- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
148 lines
4.5 KiB
Python
148 lines
4.5 KiB
Python
"""FlowDeck — unified import framework (v5.6.0, Phase 0).
|
|
|
|
Defines the normalized data model shared by every importer and the registry
|
|
used to auto-detect a source. An :class:`Importer` turns an uploaded file into
|
|
an :class:`ImportResult` (pages, attachments, warnings, stats) which the
|
|
pipeline (:mod:`app.services.importers.pipeline`) persists into FlowDeck.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from abc import ABC, abstractmethod
|
|
from dataclasses import dataclass, field
|
|
from typing import Any
|
|
|
|
|
|
def decode_text(data: bytes) -> str:
|
|
"""Best-effort decode of uploaded bytes (BOM aware, latin-1 fallback)."""
|
|
for enc in ("utf-8-sig", "utf-8", "utf-16", "latin-1"):
|
|
try:
|
|
return data.decode(enc)
|
|
except (UnicodeDecodeError, UnicodeError):
|
|
continue
|
|
return data.decode("utf-8", errors="replace")
|
|
|
|
|
|
@dataclass
|
|
class ImportAttachment:
|
|
"""A binary asset extracted from an archive/vault."""
|
|
|
|
source_path: str
|
|
filename: str
|
|
data: bytes = b""
|
|
mime: str = ""
|
|
|
|
|
|
@dataclass
|
|
class ImportPage:
|
|
"""One page to create. ``markdown`` is converted to blocks by the pipeline
|
|
unless ``blocks`` is already provided. ``collection`` marks a database
|
|
(Notion database, Excel sheet…) whose rows become ``collection_pages``."""
|
|
|
|
title: str = "Untitled"
|
|
markdown: str = ""
|
|
blocks: list[dict] = field(default_factory=list)
|
|
source_path: str = ""
|
|
parent_path: str = ""
|
|
properties: dict[str, Any] = field(default_factory=dict)
|
|
collection: dict[str, Any] | None = None
|
|
external_id: str = ""
|
|
page_id: int | None = None
|
|
|
|
|
|
@dataclass
|
|
class ImportResult:
|
|
"""Normalized output of any importer."""
|
|
|
|
source: str = ""
|
|
pages: list[ImportPage] = field(default_factory=list)
|
|
attachments: list[ImportAttachment] = field(default_factory=list)
|
|
warnings: list[str] = field(default_factory=list)
|
|
stats: dict[str, Any] = field(default_factory=dict)
|
|
|
|
def warn(self, message: str) -> None:
|
|
if message and message not in self.warnings:
|
|
self.warnings.append(message)
|
|
|
|
def finalize(self) -> ImportResult:
|
|
self.stats.setdefault("pages", len(self.pages))
|
|
self.stats.setdefault("collections", sum(1 for p in self.pages if p.collection))
|
|
self.stats.setdefault("attachments", len(self.attachments))
|
|
self.stats.setdefault("warnings", len(self.warnings))
|
|
return self
|
|
|
|
|
|
class Importer(ABC):
|
|
"""Base class for a source importer."""
|
|
|
|
source_id: str = ""
|
|
label: str = ""
|
|
description: str = ""
|
|
extensions: tuple[str, ...] = ()
|
|
order: int = 100
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
"""Return True when this importer recognizes the uploaded file."""
|
|
return False
|
|
|
|
@abstractmethod
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
"""Parse the upload into a normalized :class:`ImportResult`."""
|
|
|
|
def info(self) -> dict[str, Any]:
|
|
return {
|
|
"source_id": self.source_id,
|
|
"label": self.label,
|
|
"description": self.description,
|
|
"extensions": list(self.extensions),
|
|
}
|
|
|
|
|
|
REGISTRY: list[Importer] = []
|
|
|
|
|
|
def register_importer(cls: type[Importer]) -> type[Importer]:
|
|
"""Class decorator registering an importer instance."""
|
|
REGISTRY.append(cls())
|
|
REGISTRY.sort(key=lambda i: i.order)
|
|
return cls
|
|
|
|
|
|
def all_importers() -> list[Importer]:
|
|
return list(REGISTRY)
|
|
|
|
|
|
def get_importer(source_id: str) -> Importer | None:
|
|
for imp in REGISTRY:
|
|
if imp.source_id == source_id:
|
|
return imp
|
|
return None
|
|
|
|
|
|
def detect_importer(filename: str, data: bytes) -> Importer | None:
|
|
"""First importer that recognizes the file, else None."""
|
|
for imp in REGISTRY:
|
|
try:
|
|
if imp.detect(filename, data):
|
|
return imp
|
|
except Exception: # noqa: BLE001 - a broken detector must not break detection
|
|
continue
|
|
return None
|
|
|
|
|
|
def list_sources() -> list[dict[str, Any]]:
|
|
return [imp.info() for imp in REGISTRY]
|
|
|
|
|
|
def make_collection(name: str, schema: list[dict], rows: list[dict],
|
|
*, source_path: str = "", external_id: str = "") -> ImportPage:
|
|
"""Build an ImportPage carrying a collection spec (database import).
|
|
|
|
``rows`` entries are ``{"title": str, "properties": {name: value}}``.
|
|
"""
|
|
return ImportPage(
|
|
title=name or "Imported database",
|
|
collection={"name": name or "Imported database", "schema": schema, "rows": rows},
|
|
source_path=source_path or name,
|
|
external_id=external_id or source_path or name,
|
|
)
|