- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
107 lines
3.5 KiB
Python
107 lines
3.5 KiB
Python
"""FlowDeck — shared helpers for note importers (frontmatter, wikilinks)."""
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from typing import Any
|
|
|
|
try: # PyYAML ships transitively via uvicorn[standard]
|
|
import yaml
|
|
except Exception: # pragma: no cover - fallback parser below
|
|
yaml = None
|
|
|
|
|
|
_FRONTMATTER_RE = re.compile(r"^\ufeff?---\s*\n(.*?)\n---\s*\n?", re.DOTALL)
|
|
_WIKILINK_RE = re.compile(r"(!?)\[\[([^\]|#]+)(?:#[^\]|]+)?(?:\|([^\]]+))?\]\]")
|
|
|
|
|
|
def split_frontmatter(text: str) -> tuple[dict[str, Any], str]:
|
|
"""Split YAML frontmatter from the body. Returns ``(metadata, body)``."""
|
|
m = _FRONTMATTER_RE.match(text)
|
|
if not m:
|
|
return {}, text
|
|
raw = m.group(1)
|
|
body = text[m.end():]
|
|
if yaml is not None:
|
|
try:
|
|
meta = yaml.safe_load(raw)
|
|
if isinstance(meta, dict):
|
|
return meta, body
|
|
except Exception: # noqa: BLE001 - fall back to the simple parser
|
|
pass
|
|
return _simple_yaml(raw), body
|
|
|
|
|
|
def _simple_yaml(raw: str) -> dict[str, Any]:
|
|
"""Minimal YAML subset parser (scalars, inline lists, block lists)."""
|
|
meta: dict[str, Any] = {}
|
|
current: str | None = None
|
|
for line in raw.splitlines():
|
|
if not line.strip() or line.lstrip().startswith("#"):
|
|
continue
|
|
if line.lstrip().startswith("- ") and current:
|
|
meta.setdefault(current, [])
|
|
if isinstance(meta[current], list):
|
|
meta[current].append(_scalar(line.lstrip()[2:].strip()))
|
|
continue
|
|
if ":" in line:
|
|
key, _, value = line.partition(":")
|
|
key = key.strip()
|
|
value = value.strip()
|
|
current = key
|
|
if not value:
|
|
meta[key] = []
|
|
elif value.startswith("[") and value.endswith("]"):
|
|
inner = value[1:-1].strip()
|
|
meta[key] = [_scalar(v.strip()) for v in inner.split(",") if v.strip()] if inner else []
|
|
else:
|
|
meta[key] = _scalar(value)
|
|
return meta
|
|
|
|
|
|
def _scalar(value: str) -> Any:
|
|
v = value.strip().strip('"').strip("'")
|
|
if v.lower() in ("true", "false"):
|
|
return v.lower() == "true"
|
|
if re.fullmatch(r"-?\d+", v):
|
|
return int(v)
|
|
if re.fullmatch(r"-?\d+\.\d+", v):
|
|
return float(v)
|
|
return v
|
|
|
|
|
|
def convert_wikilinks(text: str, *, embeds: bool = True) -> str:
|
|
"""Turn Obsidian/Logseq ``[[link]]`` into Markdown links and ``![[img]]``
|
|
into Markdown images so the block converter can render them."""
|
|
|
|
def repl(m: re.Match) -> str:
|
|
bang, target, alias = m.group(1), m.group(2).strip(), m.group(3)
|
|
label = (alias or target).strip()
|
|
if bang == "!":
|
|
return f"" if embeds else label
|
|
return f"[{label}]({target})"
|
|
|
|
return _WIKILINK_RE.sub(repl, text)
|
|
|
|
|
|
def normalize_title(value: Any) -> str:
|
|
return str(value).strip() if value is not None else ""
|
|
|
|
|
|
def coerce_tags(value: Any) -> list[str]:
|
|
if value is None:
|
|
return []
|
|
if isinstance(value, list):
|
|
return [str(v).strip().lstrip("#") for v in value if str(v).strip()]
|
|
if isinstance(value, str):
|
|
parts = re.split(r"[,\s]+", value)
|
|
return [p.strip().lstrip("#") for p in parts if p.strip()]
|
|
return [str(value)]
|
|
|
|
|
|
def strip_markdown(text: str) -> str:
|
|
text = re.sub(r"`{1,3}([^`]*)`{1,3}", r"\1", text)
|
|
text = re.sub(r"!\[[^\]]*\]\([^)]*\)", "", text)
|
|
text = re.sub(r"\[([^\]]*)\]\([^)]*\)", r"\1", text)
|
|
text = re.sub(r"[*_~#>]+", "", text)
|
|
return text.strip()
|