- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
119 lines
3.9 KiB
Python
119 lines
3.9 KiB
Python
"""FlowDeck — Obsidian vault importer (v5.6.0, Phase 1).
|
|
|
|
Imports a vault exported as a ``.zip``: Markdown notes (with YAML frontmatter),
|
|
the folder hierarchy, ``[[wikilinks]]``/``![[embeds]]`` and binary attachments.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
import zipfile
|
|
|
|
from app.services.importers._common import (
|
|
coerce_tags,
|
|
convert_wikilinks,
|
|
normalize_title,
|
|
split_frontmatter,
|
|
)
|
|
from app.services.importers.base import (
|
|
ImportAttachment,
|
|
Importer,
|
|
ImportPage,
|
|
ImportResult,
|
|
decode_text,
|
|
register_importer,
|
|
)
|
|
|
|
_SKIP_DIRS = (".obsidian/", ".trash/", ".git/", ".DS_Store")
|
|
|
|
|
|
def _mime_for(name: str) -> str:
|
|
import mimetypes
|
|
|
|
return mimetypes.guess_type(name)[0] or "application/octet-stream"
|
|
|
|
|
|
@register_importer
|
|
class ObsidianImporter(Importer):
|
|
source_id = "obsidian"
|
|
label = "Obsidian (vault .zip)"
|
|
description = "Vault Obsidian : notes Markdown, frontmatter YAML, wikilinks, pièces jointes."
|
|
extensions = (".zip",)
|
|
order = 10
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
if not filename.lower().endswith(".zip"):
|
|
return False
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError):
|
|
return False
|
|
names = zf.namelist()
|
|
if any("/.obsidian/" in n or n.startswith(".obsidian/") for n in names):
|
|
return True
|
|
# Heuristic: mostly-markdown archive containing wikilinks.
|
|
md = [n for n in names if n.lower().endswith(".md")]
|
|
if not md:
|
|
return False
|
|
for n in md[:20]:
|
|
try:
|
|
if "[[" in decode_text(zf.read(n)):
|
|
return True
|
|
except Exception: # noqa: BLE001
|
|
continue
|
|
return False
|
|
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
result = ImportResult(source=self.source_id)
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError) as exc:
|
|
result.warn(f"Archive invalide : {exc}")
|
|
return result.finalize()
|
|
|
|
names = [n for n in zf.namelist() if not n.endswith("/")]
|
|
notes = [n for n in names if n.lower().endswith(".md")]
|
|
assets = [n for n in names if not n.lower().endswith(".md")]
|
|
|
|
for name in assets:
|
|
clean = name.replace("\\", "/")
|
|
if any(part in clean for part in _SKIP_DIRS) or clean.split("/")[-1].startswith("."):
|
|
continue
|
|
try:
|
|
payload = zf.read(name)
|
|
except Exception: # noqa: BLE001
|
|
continue
|
|
result.attachments.append(ImportAttachment(
|
|
source_path=name,
|
|
filename=clean.rsplit("/", 1)[-1],
|
|
data=payload,
|
|
mime=_mime_for(name),
|
|
))
|
|
|
|
for name in sorted(notes, key=lambda n: (n.count("/"), n.lower())):
|
|
clean = name.replace("\\", "/")
|
|
if any(part in clean for part in _SKIP_DIRS):
|
|
continue
|
|
try:
|
|
text = decode_text(zf.read(name))
|
|
except Exception as exc: # noqa: BLE001
|
|
result.warn(f"Lecture impossible : {name} ({exc})")
|
|
continue
|
|
meta, body = split_frontmatter(text)
|
|
body = convert_wikilinks(body)
|
|
parts = clean.split("/")
|
|
title = normalize_title(meta.get("title")) or parts[-1][:-3]
|
|
props = dict(meta)
|
|
props.pop("title", None)
|
|
tags = coerce_tags(meta.get("tags"))
|
|
if tags:
|
|
props["tags"] = tags
|
|
result.pages.append(ImportPage(
|
|
title=title or "Untitled",
|
|
markdown=body,
|
|
source_path=clean,
|
|
parent_path="/".join(parts[:-1]),
|
|
properties=props,
|
|
external_id=clean,
|
|
))
|
|
return result.finalize()
|