Files
flowdeck/app/services/importers/outline.py
T
bruno 3b00cbc371 feat(import): v5.6.0 unified data import (phases 0-5)
- unified importer framework (app/services/importers/): normalized model,
  registry, common pipeline (hierarchy, attachments, collections, dedup),
  async jobs, dry-run preview, column->type mapping
- Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/
  OneNote), Google Keep, generic Markdown
- Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON
- Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders
- Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics,
  OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones)
- Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume,
  forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue,
  Notion relation resolution, exportable JSON reports
- /import wizard, API /api/import/*, migration 9 (import_items, import_jobs)
- fix: property values stored by property id (correct DB view rendering)
- deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf
- 43 import tests; full suite 491 green; ruff clean
- bump version 5.11.5
2026-09-13 10:43:20 -04:00

165 lines
5.8 KiB
Python

"""FlowDeck — Logseq / Roam Research outliner importer (v5.6.0, Phase 1)."""
from __future__ import annotations
import io
import re
import zipfile
from app.services.importers._common import (
coerce_tags,
convert_wikilinks,
normalize_title,
split_frontmatter,
)
from app.services.importers.base import (
ImportAttachment,
Importer,
ImportPage,
ImportResult,
decode_text,
register_importer,
)
_LOGSEQ_MARKERS = ("property::", "logseq/", "journals/")
_ROAM_MARKERS = ("{{[[TODO]]}}", "{{[[DONE]]}}", "{{[[query]]}}")
_PROP_RE = re.compile(r"^\s*([a-zA-Z][\w-]*)::\s*(.*)$")
def _journal_title(name: str) -> str:
m = re.match(r"^(\d{4})[_-](\d{2})[_-](\d{2})", name)
if m:
return f"{m.group(1)}-{m.group(2)}-{m.group(3)}"
return name
def _clean_outline(text: str) -> tuple[dict, str]:
meta, body = split_frontmatter(text)
lines_out: list[str] = []
for line in body.splitlines():
m = _PROP_RE.match(line)
if m and line.lstrip().startswith("-"):
continue
# Logseq properties appear as bare ``key:: value`` lines too.
m2 = _PROP_RE.match(line)
if m2 and not line.lstrip().startswith(("-", "*", "#", "|")):
key = m2.group(1)
if key not in meta:
meta[key] = m2.group(2).strip()
continue
lines_out.append(line)
body = "\n".join(lines_out)
# Roam task markers → GFM checkboxes.
body = body.replace("{{[[TODO]]}}", "[ ] ").replace("{{[[DONE]]}}", "[x] ")
# Block references ((uuid)) → plain anchors.
body = re.sub(r"\(\(([0-9a-fA-F-]{6,})\)\)", r"[[\1]]", body)
body = convert_wikilinks(body)
return meta, body
class _OutlineBase(Importer):
markers: tuple[str, ...] = ()
property_syntax = False
source_id = "outline"
label = "Outliner"
description = ""
order = 30
def _text_matches(self, text: str) -> bool:
if any(m in text for m in self.markers if not m.endswith("/")):
return True
return bool(self.property_syntax and re.search(r"^\s*[a-zA-Z][\w-]*::", text, re.M))
def _looks_like(self, filename: str, data: bytes) -> bool:
low = filename.lower()
if low.endswith((".md", ".markdown", ".txt")):
return self._text_matches(decode_text(data))
if low.endswith(".zip"):
try:
zf = zipfile.ZipFile(io.BytesIO(data))
except (zipfile.BadZipFile, OSError):
return False
names = zf.namelist()
if any(m in name for m in self.markers if m.endswith("/") for name in names):
return True
for n in [x for x in names if x.lower().endswith(".md")][:10]:
try:
if self._text_matches(decode_text(zf.read(n))):
return True
except Exception: # noqa: BLE001
continue
return False
def detect(self, filename: str, data: bytes) -> bool:
return self._looks_like(filename, data)
def _emit(self, result: ImportResult, name: str, text: str, is_journal: bool) -> None:
meta, body = _clean_outline(text)
parts = name.replace("\\", "/").split("/")
raw_title = parts[-1].rsplit(".", 1)[0]
title = normalize_title(meta.get("title")) or (
_journal_title(raw_title) if is_journal else raw_title
)
props = {k: v for k, v in meta.items() if k != "title"}
tags = coerce_tags(meta.get("tags"))
if tags:
props["tags"] = tags
result.pages.append(ImportPage(
title=title or "Untitled",
markdown=body,
source_path=name,
parent_path="/".join(parts[:-1]),
properties=props,
external_id=name,
))
def parse(self, filename: str, data: bytes) -> ImportResult:
result = ImportResult(source=self.source_id)
if filename.lower().endswith(".zip"):
try:
zf = zipfile.ZipFile(io.BytesIO(data))
except (zipfile.BadZipFile, OSError) as exc:
result.warn(f"Archive invalide : {exc}")
return result.finalize()
names = [n for n in zf.namelist() if not n.endswith("/")]
for name in [n for n in names if not n.lower().endswith(".md")]:
try:
result.attachments.append(ImportAttachment(
source_path=name,
filename=name.replace("\\", "/").rsplit("/", 1)[-1],
data=zf.read(name),
))
except Exception: # noqa: BLE001
continue
for name in sorted(
(n for n in names if n.lower().endswith(".md")),
key=lambda n: (n.count("/"), n.lower()),
):
try:
text = decode_text(zf.read(name))
except Exception as exc: # noqa: BLE001
result.warn(f"Lecture impossible : {name} ({exc})")
continue
self._emit(result, name, text, is_journal="journal" in name.lower())
return result.finalize()
text = decode_text(data)
self._emit(result, filename, text, is_journal=False)
return result.finalize()
@register_importer
class LogseqImporter(_OutlineBase):
source_id = "logseq"
label = "Logseq"
description = "Outliner Logseq : pages/journal, propriétés `key:: value`, block refs."
markers = ("property::", "logseq/", "journals/")
property_syntax = True
@register_importer
class RoamImporter(_OutlineBase):
source_id = "roam"
label = "Roam Research"
description = "Outliner Roam : `{{[[TODO]]}}`, block refs, wikilinks."
markers = _ROAM_MARKERS