- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
88 lines
3.0 KiB
Python
88 lines
3.0 KiB
Python
"""FlowDeck — generic Markdown / text importer (v5.6.0, Phase 1)."""
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
import zipfile
|
|
|
|
from app.services.importers.base import (
|
|
Importer,
|
|
ImportPage,
|
|
ImportResult,
|
|
decode_text,
|
|
register_importer,
|
|
)
|
|
|
|
_MD_EXTS = (".md", ".markdown", ".txt", ".mdx")
|
|
|
|
|
|
def _title_from_name(name: str) -> str:
|
|
base = name.replace("\\", "/").rsplit("/", 1)[-1]
|
|
for ext in (".markdown", ".markdown", ".mdx", ".md", ".txt"):
|
|
if base.lower().endswith(ext):
|
|
base = base[: -len(ext)]
|
|
break
|
|
return base.strip() or "Untitled"
|
|
|
|
|
|
@register_importer
|
|
class MarkdownImporter(Importer):
|
|
source_id = "markdown"
|
|
label = "Markdown / texte"
|
|
description = "Fichiers .md/.markdown/.txt ou archive .zip de fichiers Markdown."
|
|
extensions = (".md", ".markdown", ".txt", ".zip")
|
|
order = 90
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
low = filename.lower()
|
|
if low.endswith(_MD_EXTS):
|
|
return True
|
|
if low.endswith(".zip"):
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError):
|
|
return False
|
|
names = [n for n in zf.namelist() if not n.endswith("/")]
|
|
return bool(names) and all(n.lower().endswith(_MD_EXTS) for n in names)
|
|
return False
|
|
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
result = ImportResult(source=self.source_id)
|
|
if filename.lower().endswith(".zip"):
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError) as exc:
|
|
result.warn(f"Archive invalide : {exc}")
|
|
return result.finalize()
|
|
entries = sorted(
|
|
(n for n in zf.namelist()
|
|
if not n.endswith("/") and n.lower().endswith(_MD_EXTS)),
|
|
key=lambda n: (n.count("/"), n.lower()),
|
|
)
|
|
if not entries:
|
|
result.warn("Aucun fichier Markdown trouvé dans l'archive")
|
|
return result.finalize()
|
|
for name in entries:
|
|
try:
|
|
text = decode_text(zf.read(name))
|
|
except Exception as exc: # noqa: BLE001
|
|
result.warn(f"Lecture impossible : {name} ({exc})")
|
|
continue
|
|
parts = name.replace("\\", "/").split("/")
|
|
result.pages.append(ImportPage(
|
|
title=_title_from_name(name),
|
|
markdown=text,
|
|
source_path=name,
|
|
parent_path="/".join(parts[:-1]),
|
|
external_id=name,
|
|
))
|
|
return result.finalize()
|
|
|
|
text = decode_text(data)
|
|
result.pages.append(ImportPage(
|
|
title=_title_from_name(filename),
|
|
markdown=text,
|
|
source_path=filename,
|
|
external_id=filename,
|
|
))
|
|
return result.finalize()
|