- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
165 lines
5.8 KiB
Python
165 lines
5.8 KiB
Python
"""FlowDeck — Logseq / Roam Research outliner importer (v5.6.0, Phase 1)."""
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
import re
|
|
import zipfile
|
|
|
|
from app.services.importers._common import (
|
|
coerce_tags,
|
|
convert_wikilinks,
|
|
normalize_title,
|
|
split_frontmatter,
|
|
)
|
|
from app.services.importers.base import (
|
|
ImportAttachment,
|
|
Importer,
|
|
ImportPage,
|
|
ImportResult,
|
|
decode_text,
|
|
register_importer,
|
|
)
|
|
|
|
_LOGSEQ_MARKERS = ("property::", "logseq/", "journals/")
|
|
_ROAM_MARKERS = ("{{[[TODO]]}}", "{{[[DONE]]}}", "{{[[query]]}}")
|
|
_PROP_RE = re.compile(r"^\s*([a-zA-Z][\w-]*)::\s*(.*)$")
|
|
|
|
|
|
def _journal_title(name: str) -> str:
|
|
m = re.match(r"^(\d{4})[_-](\d{2})[_-](\d{2})", name)
|
|
if m:
|
|
return f"{m.group(1)}-{m.group(2)}-{m.group(3)}"
|
|
return name
|
|
|
|
|
|
def _clean_outline(text: str) -> tuple[dict, str]:
|
|
meta, body = split_frontmatter(text)
|
|
lines_out: list[str] = []
|
|
for line in body.splitlines():
|
|
m = _PROP_RE.match(line)
|
|
if m and line.lstrip().startswith("-"):
|
|
continue
|
|
# Logseq properties appear as bare ``key:: value`` lines too.
|
|
m2 = _PROP_RE.match(line)
|
|
if m2 and not line.lstrip().startswith(("-", "*", "#", "|")):
|
|
key = m2.group(1)
|
|
if key not in meta:
|
|
meta[key] = m2.group(2).strip()
|
|
continue
|
|
lines_out.append(line)
|
|
body = "\n".join(lines_out)
|
|
# Roam task markers → GFM checkboxes.
|
|
body = body.replace("{{[[TODO]]}}", "[ ] ").replace("{{[[DONE]]}}", "[x] ")
|
|
# Block references ((uuid)) → plain anchors.
|
|
body = re.sub(r"\(\(([0-9a-fA-F-]{6,})\)\)", r"[[\1]]", body)
|
|
body = convert_wikilinks(body)
|
|
return meta, body
|
|
|
|
|
|
class _OutlineBase(Importer):
|
|
markers: tuple[str, ...] = ()
|
|
property_syntax = False
|
|
source_id = "outline"
|
|
label = "Outliner"
|
|
description = ""
|
|
order = 30
|
|
|
|
def _text_matches(self, text: str) -> bool:
|
|
if any(m in text for m in self.markers if not m.endswith("/")):
|
|
return True
|
|
return bool(self.property_syntax and re.search(r"^\s*[a-zA-Z][\w-]*::", text, re.M))
|
|
|
|
def _looks_like(self, filename: str, data: bytes) -> bool:
|
|
low = filename.lower()
|
|
if low.endswith((".md", ".markdown", ".txt")):
|
|
return self._text_matches(decode_text(data))
|
|
if low.endswith(".zip"):
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError):
|
|
return False
|
|
names = zf.namelist()
|
|
if any(m in name for m in self.markers if m.endswith("/") for name in names):
|
|
return True
|
|
for n in [x for x in names if x.lower().endswith(".md")][:10]:
|
|
try:
|
|
if self._text_matches(decode_text(zf.read(n))):
|
|
return True
|
|
except Exception: # noqa: BLE001
|
|
continue
|
|
return False
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
return self._looks_like(filename, data)
|
|
|
|
def _emit(self, result: ImportResult, name: str, text: str, is_journal: bool) -> None:
|
|
meta, body = _clean_outline(text)
|
|
parts = name.replace("\\", "/").split("/")
|
|
raw_title = parts[-1].rsplit(".", 1)[0]
|
|
title = normalize_title(meta.get("title")) or (
|
|
_journal_title(raw_title) if is_journal else raw_title
|
|
)
|
|
props = {k: v for k, v in meta.items() if k != "title"}
|
|
tags = coerce_tags(meta.get("tags"))
|
|
if tags:
|
|
props["tags"] = tags
|
|
result.pages.append(ImportPage(
|
|
title=title or "Untitled",
|
|
markdown=body,
|
|
source_path=name,
|
|
parent_path="/".join(parts[:-1]),
|
|
properties=props,
|
|
external_id=name,
|
|
))
|
|
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
result = ImportResult(source=self.source_id)
|
|
if filename.lower().endswith(".zip"):
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError) as exc:
|
|
result.warn(f"Archive invalide : {exc}")
|
|
return result.finalize()
|
|
names = [n for n in zf.namelist() if not n.endswith("/")]
|
|
for name in [n for n in names if not n.lower().endswith(".md")]:
|
|
try:
|
|
result.attachments.append(ImportAttachment(
|
|
source_path=name,
|
|
filename=name.replace("\\", "/").rsplit("/", 1)[-1],
|
|
data=zf.read(name),
|
|
))
|
|
except Exception: # noqa: BLE001
|
|
continue
|
|
for name in sorted(
|
|
(n for n in names if n.lower().endswith(".md")),
|
|
key=lambda n: (n.count("/"), n.lower()),
|
|
):
|
|
try:
|
|
text = decode_text(zf.read(name))
|
|
except Exception as exc: # noqa: BLE001
|
|
result.warn(f"Lecture impossible : {name} ({exc})")
|
|
continue
|
|
self._emit(result, name, text, is_journal="journal" in name.lower())
|
|
return result.finalize()
|
|
|
|
text = decode_text(data)
|
|
self._emit(result, filename, text, is_journal=False)
|
|
return result.finalize()
|
|
|
|
|
|
@register_importer
|
|
class LogseqImporter(_OutlineBase):
|
|
source_id = "logseq"
|
|
label = "Logseq"
|
|
description = "Outliner Logseq : pages/journal, propriétés `key:: value`, block refs."
|
|
markers = ("property::", "logseq/", "journals/")
|
|
property_syntax = True
|
|
|
|
|
|
@register_importer
|
|
class RoamImporter(_OutlineBase):
|
|
source_id = "roam"
|
|
label = "Roam Research"
|
|
description = "Outliner Roam : `{{[[TODO]]}}`, block refs, wikilinks."
|
|
markers = _ROAM_MARKERS
|