Files
flowdeck/app/services/importers/_common.py
T
bruno 3b00cbc371 feat(import): v5.6.0 unified data import (phases 0-5)
- unified importer framework (app/services/importers/): normalized model,
  registry, common pipeline (hierarchy, attachments, collections, dedup),
  async jobs, dry-run preview, column->type mapping
- Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/
  OneNote), Google Keep, generic Markdown
- Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON
- Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders
- Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics,
  OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones)
- Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume,
  forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue,
  Notion relation resolution, exportable JSON reports
- /import wizard, API /api/import/*, migration 9 (import_items, import_jobs)
- fix: property values stored by property id (correct DB view rendering)
- deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf
- 43 import tests; full suite 491 green; ruff clean
- bump version 5.11.5
2026-09-13 10:43:20 -04:00

107 lines
3.5 KiB
Python

"""FlowDeck — shared helpers for note importers (frontmatter, wikilinks)."""
from __future__ import annotations
import re
from typing import Any
try: # PyYAML ships transitively via uvicorn[standard]
import yaml
except Exception: # pragma: no cover - fallback parser below
yaml = None
_FRONTMATTER_RE = re.compile(r"^\ufeff?---\s*\n(.*?)\n---\s*\n?", re.DOTALL)
_WIKILINK_RE = re.compile(r"(!?)\[\[([^\]|#]+)(?:#[^\]|]+)?(?:\|([^\]]+))?\]\]")
def split_frontmatter(text: str) -> tuple[dict[str, Any], str]:
"""Split YAML frontmatter from the body. Returns ``(metadata, body)``."""
m = _FRONTMATTER_RE.match(text)
if not m:
return {}, text
raw = m.group(1)
body = text[m.end():]
if yaml is not None:
try:
meta = yaml.safe_load(raw)
if isinstance(meta, dict):
return meta, body
except Exception: # noqa: BLE001 - fall back to the simple parser
pass
return _simple_yaml(raw), body
def _simple_yaml(raw: str) -> dict[str, Any]:
"""Minimal YAML subset parser (scalars, inline lists, block lists)."""
meta: dict[str, Any] = {}
current: str | None = None
for line in raw.splitlines():
if not line.strip() or line.lstrip().startswith("#"):
continue
if line.lstrip().startswith("- ") and current:
meta.setdefault(current, [])
if isinstance(meta[current], list):
meta[current].append(_scalar(line.lstrip()[2:].strip()))
continue
if ":" in line:
key, _, value = line.partition(":")
key = key.strip()
value = value.strip()
current = key
if not value:
meta[key] = []
elif value.startswith("[") and value.endswith("]"):
inner = value[1:-1].strip()
meta[key] = [_scalar(v.strip()) for v in inner.split(",") if v.strip()] if inner else []
else:
meta[key] = _scalar(value)
return meta
def _scalar(value: str) -> Any:
v = value.strip().strip('"').strip("'")
if v.lower() in ("true", "false"):
return v.lower() == "true"
if re.fullmatch(r"-?\d+", v):
return int(v)
if re.fullmatch(r"-?\d+\.\d+", v):
return float(v)
return v
def convert_wikilinks(text: str, *, embeds: bool = True) -> str:
"""Turn Obsidian/Logseq ``[[link]]`` into Markdown links and ``![[img]]``
into Markdown images so the block converter can render them."""
def repl(m: re.Match) -> str:
bang, target, alias = m.group(1), m.group(2).strip(), m.group(3)
label = (alias or target).strip()
if bang == "!":
return f"![{label}]({target})" if embeds else label
return f"[{label}]({target})"
return _WIKILINK_RE.sub(repl, text)
def normalize_title(value: Any) -> str:
return str(value).strip() if value is not None else ""
def coerce_tags(value: Any) -> list[str]:
if value is None:
return []
if isinstance(value, list):
return [str(v).strip().lstrip("#") for v in value if str(v).strip()]
if isinstance(value, str):
parts = re.split(r"[,\s]+", value)
return [p.strip().lstrip("#") for p in parts if p.strip()]
return [str(value)]
def strip_markdown(text: str) -> str:
text = re.sub(r"`{1,3}([^`]*)`{1,3}", r"\1", text)
text = re.sub(r"!\[[^\]]*\]\([^)]*\)", "", text)
text = re.sub(r"\[([^\]]*)\]\([^)]*\)", r"\1", text)
text = re.sub(r"[*_~#>]+", "", text)
return text.strip()