- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
301 lines
11 KiB
Python
301 lines
11 KiB
Python
"""FlowDeck — HTML notes & Google Keep importer (v5.6.0, Phase 1).
|
|
|
|
Covers HTML exports from Apple Notes, Bear, Ulysses and OneNote, plus the
|
|
Google Takeout ``Keep`` JSON/HTML format. HTML is converted to Markdown and then
|
|
to FlowDeck blocks by the pipeline.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
import json
|
|
import re
|
|
import zipfile
|
|
|
|
from bs4 import BeautifulSoup, NavigableString, Tag
|
|
|
|
from app.services.importers._common import coerce_tags, normalize_title
|
|
from app.services.importers.base import (
|
|
ImportAttachment,
|
|
Importer,
|
|
ImportPage,
|
|
ImportResult,
|
|
decode_text,
|
|
register_importer,
|
|
)
|
|
|
|
_HTML_EXTS = (".html", ".htm")
|
|
|
|
|
|
def _inline(node: Tag) -> str:
|
|
out: list[str] = []
|
|
for child in node.children:
|
|
if isinstance(child, NavigableString):
|
|
out.append(str(child))
|
|
elif isinstance(child, Tag):
|
|
name = child.name.lower()
|
|
if name in ("strong", "b"):
|
|
out.append(f"**{_inline(child).strip()}**")
|
|
elif name in ("em", "i"):
|
|
out.append(f"*{_inline(child).strip()}*")
|
|
elif name == "code":
|
|
out.append(f"`{child.get_text()}`")
|
|
elif name == "br":
|
|
out.append("\n")
|
|
elif name == "a":
|
|
href = child.get("href", "")
|
|
label = _inline(child).strip() or href
|
|
out.append(f"[{label}]({href})" if href else label)
|
|
elif name == "img":
|
|
src = child.get("src", "")
|
|
alt = child.get("alt", "")
|
|
out.append(f"" if src else "")
|
|
elif name in ("del", "s", "strike"):
|
|
out.append(f"~~{_inline(child).strip()}~~")
|
|
else:
|
|
out.append(_inline(child))
|
|
return re.sub(r"[ \t]+", " ", "".join(out))
|
|
|
|
|
|
def _table(node: Tag) -> str:
|
|
rows: list[list[str]] = []
|
|
for tr in node.find_all("tr"):
|
|
cells = tr.find_all(["th", "td"])
|
|
rows.append([_inline(c).strip().replace("|", "\\|") for c in cells])
|
|
if not rows:
|
|
return ""
|
|
width = max(len(r) for r in rows)
|
|
rows = [r + [""] * (width - len(r)) for r in rows]
|
|
header = "| " + " | ".join(rows[0]) + " |"
|
|
sep = "| " + " | ".join(["---"] * width) + " |"
|
|
body = "\n".join("| " + " | ".join(r) + " |" for r in rows[1:])
|
|
return "\n".join(x for x in (header, sep, body) if x)
|
|
|
|
|
|
def _block(node: Tag, depth: int = 0) -> str:
|
|
name = node.name.lower()
|
|
if name in ("h1", "h2", "h3", "h4", "h5", "h6"):
|
|
return "#" * int(name[1]) + " " + _inline(node).strip()
|
|
if name == "p":
|
|
return _inline(node).strip()
|
|
if name in ("ul", "ol"):
|
|
lines = []
|
|
for i, li in enumerate(node.find_all("li", recursive=False)):
|
|
marker = f"{i + 1}." if name == "ol" else "-"
|
|
text = _inline(li).strip()
|
|
lines.append(f"{' ' * depth}{marker} {text}")
|
|
return "\n".join(lines)
|
|
if name == "blockquote":
|
|
return "\n".join(f"> {ln}" for ln in _inline(node).strip().splitlines())
|
|
if name == "pre":
|
|
code = node.get_text()
|
|
lang = ""
|
|
cls = " ".join(node.get("class", [])) if node.get("class") else ""
|
|
m = re.search(r"(?:language|lang)-([\w+-]+)", cls)
|
|
if m:
|
|
lang = m.group(1)
|
|
return f"```{lang}\n{code.rstrip()}\n```"
|
|
if name == "hr":
|
|
return "---"
|
|
if name == "table":
|
|
return _table(node)
|
|
if name == "img":
|
|
src = node.get("src", "")
|
|
return f"" if src else ""
|
|
if name in ("div", "section", "article", "body", "main", "html", "span", "font", "center"):
|
|
inner = "\n\n".join(
|
|
_block(c, depth) for c in node.children if isinstance(c, Tag)
|
|
).strip()
|
|
if inner:
|
|
return inner
|
|
text = _inline(node).strip()
|
|
return text
|
|
return _inline(node).strip()
|
|
|
|
|
|
def _html_to_markdown(html: str) -> str:
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
for tag in soup(["script", "style", "head", "nav", "footer"]):
|
|
tag.decompose()
|
|
root = soup.body or soup
|
|
blocks = [_block(c) for c in root.children if isinstance(c, Tag)]
|
|
md = "\n\n".join(b for b in blocks if b and b.strip())
|
|
return re.sub(r"\n{3,}", "\n\n", md).strip()
|
|
|
|
|
|
def _title_from_html(html: str, fallback: str) -> str:
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
if soup.title and soup.title.string:
|
|
return soup.title.string.strip()
|
|
h1 = soup.find(["h1", "h2"])
|
|
if h1:
|
|
return h1.get_text().strip()
|
|
return fallback
|
|
|
|
|
|
@register_importer
|
|
class HtmlNotesImporter(Importer):
|
|
source_id = "html_notes"
|
|
label = "HTML (Apple Notes, Bear, Ulysses, OneNote)"
|
|
description = "Fichiers HTML ou archive .zip (notes exportées en HTML)."
|
|
extensions = (".html", ".htm", ".zip")
|
|
order = 50
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
low = filename.lower()
|
|
if low.endswith(_HTML_EXTS):
|
|
return True
|
|
if low.endswith(".zip"):
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError):
|
|
return False
|
|
names = [n for n in zf.namelist() if not n.endswith("/")]
|
|
return any(n.lower().endswith(_HTML_EXTS) for n in names)
|
|
return False
|
|
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
result = ImportResult(source=self.source_id)
|
|
entries: list[tuple[str, bytes]] = []
|
|
if filename.lower().endswith(".zip"):
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError) as exc:
|
|
result.warn(f"Archive invalide : {exc}")
|
|
return result.finalize()
|
|
for name in zf.namelist():
|
|
if name.endswith("/"):
|
|
continue
|
|
clean = name.replace("\\", "/")
|
|
if clean.lower().endswith(_HTML_EXTS):
|
|
entries.append((clean, zf.read(name)))
|
|
else:
|
|
result.attachments.append(ImportAttachment(
|
|
source_path=clean,
|
|
filename=clean.rsplit("/", 1)[-1],
|
|
data=zf.read(name),
|
|
))
|
|
else:
|
|
entries.append((filename, data))
|
|
|
|
for name, payload in entries:
|
|
html = decode_text(payload)
|
|
fallback = name.replace("\\", "/").rsplit("/", 1)[-1].rsplit(".", 1)[0]
|
|
parts = name.replace("\\", "/").split("/")
|
|
result.pages.append(ImportPage(
|
|
title=_title_from_html(html, fallback) or "Untitled",
|
|
markdown=_html_to_markdown(html),
|
|
source_path=name,
|
|
parent_path="/".join(parts[:-1]),
|
|
external_id=name,
|
|
))
|
|
return result.finalize()
|
|
|
|
|
|
@register_importer
|
|
class GoogleKeepImporter(Importer):
|
|
source_id = "google_keep"
|
|
label = "Google Keep (Takeout)"
|
|
description = "Export Google Takeout : Keep/*.json (notes, listes, labels, pièces jointes)."
|
|
extensions = (".json", ".zip")
|
|
order = 40
|
|
|
|
def _is_keep_json(self, data: bytes) -> bool:
|
|
try:
|
|
obj = json.loads(decode_text(data))
|
|
except Exception: # noqa: BLE001
|
|
return False
|
|
return isinstance(obj, dict) and any(
|
|
k in obj for k in ("textContent", "listContent", "isTrashed", "color")
|
|
)
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
low = filename.lower()
|
|
if low.endswith(".json"):
|
|
return self._is_keep_json(data)
|
|
if low.endswith(".zip"):
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError):
|
|
return False
|
|
for n in zf.namelist():
|
|
if n.lower().endswith(".json") and "keep" in n.lower():
|
|
try:
|
|
if self._is_keep_json(zf.read(n)):
|
|
return True
|
|
except Exception: # noqa: BLE001
|
|
continue
|
|
return False
|
|
|
|
def _page_from_keep(self, obj: dict, name: str) -> ImportPage | None:
|
|
if obj.get("isTrashed"):
|
|
return None
|
|
title = normalize_title(obj.get("title"))
|
|
lines: list[str] = []
|
|
for item in obj.get("listContent") or []:
|
|
mark = "x" if item.get("isChecked") else " "
|
|
lines.append(f"- [{mark}] {item.get('text', '')}")
|
|
if obj.get("textContent"):
|
|
lines.insert(0, obj["textContent"])
|
|
body = "\n\n".join(lines)
|
|
if not title:
|
|
first = next((ln for ln in body.splitlines() if ln.strip()), "")
|
|
first = re.sub(r"^[-*+]\s*(\[[ xX]\]\s*)?", "", first).strip()
|
|
title = first[:60] or "Note"
|
|
labels = coerce_tags(obj.get("labels"))
|
|
props = {"tags": labels} if labels else {}
|
|
return ImportPage(
|
|
title=title,
|
|
markdown=body,
|
|
source_path=name,
|
|
parent_path="",
|
|
properties=props,
|
|
external_id=name,
|
|
)
|
|
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
result = ImportResult(source=self.source_id)
|
|
if filename.lower().endswith(".zip"):
|
|
try:
|
|
zf = zipfile.ZipFile(io.BytesIO(data))
|
|
except (zipfile.BadZipFile, OSError) as exc:
|
|
result.warn(f"Archive invalide : {exc}")
|
|
return result.finalize()
|
|
for name in zf.namelist():
|
|
if name.endswith("/"):
|
|
continue
|
|
clean = name.replace("\\", "/")
|
|
if clean.lower().endswith(".json") and "keep" in clean.lower():
|
|
try:
|
|
obj = json.loads(decode_text(zf.read(name)))
|
|
except Exception: # noqa: BLE001
|
|
continue
|
|
if not isinstance(obj, dict):
|
|
continue
|
|
page = self._page_from_keep(obj, clean)
|
|
if page:
|
|
result.pages.append(page)
|
|
elif "/keep/" in clean.lower() and not clean.lower().endswith(".json"):
|
|
result.attachments.append(ImportAttachment(
|
|
source_path=clean,
|
|
filename=clean.rsplit("/", 1)[-1],
|
|
data=zf.read(name),
|
|
))
|
|
return result.finalize()
|
|
|
|
try:
|
|
obj = json.loads(decode_text(data))
|
|
except Exception as exc: # noqa: BLE001
|
|
result.warn(f"JSON invalide : {exc}")
|
|
return result.finalize()
|
|
if isinstance(obj, list):
|
|
for i, item in enumerate(obj):
|
|
if isinstance(item, dict):
|
|
page = self._page_from_keep(item, f"{filename}#{i}")
|
|
if page:
|
|
result.pages.append(page)
|
|
else:
|
|
page = self._page_from_keep(obj, filename)
|
|
if page:
|
|
result.pages.append(page)
|
|
return result.finalize()
|