Files
flowdeck/app/services/importers/html_notes.py
T
bruno 3b00cbc371 feat(import): v5.6.0 unified data import (phases 0-5)
- unified importer framework (app/services/importers/): normalized model,
  registry, common pipeline (hierarchy, attachments, collections, dedup),
  async jobs, dry-run preview, column->type mapping
- Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/
  OneNote), Google Keep, generic Markdown
- Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON
- Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders
- Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics,
  OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones)
- Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume,
  forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue,
  Notion relation resolution, exportable JSON reports
- /import wizard, API /api/import/*, migration 9 (import_items, import_jobs)
- fix: property values stored by property id (correct DB view rendering)
- deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf
- 43 import tests; full suite 491 green; ruff clean
- bump version 5.11.5
2026-09-13 10:43:20 -04:00

301 lines
11 KiB
Python

"""FlowDeck — HTML notes & Google Keep importer (v5.6.0, Phase 1).
Covers HTML exports from Apple Notes, Bear, Ulysses and OneNote, plus the
Google Takeout ``Keep`` JSON/HTML format. HTML is converted to Markdown and then
to FlowDeck blocks by the pipeline.
"""
from __future__ import annotations
import io
import json
import re
import zipfile
from bs4 import BeautifulSoup, NavigableString, Tag
from app.services.importers._common import coerce_tags, normalize_title
from app.services.importers.base import (
ImportAttachment,
Importer,
ImportPage,
ImportResult,
decode_text,
register_importer,
)
_HTML_EXTS = (".html", ".htm")
def _inline(node: Tag) -> str:
out: list[str] = []
for child in node.children:
if isinstance(child, NavigableString):
out.append(str(child))
elif isinstance(child, Tag):
name = child.name.lower()
if name in ("strong", "b"):
out.append(f"**{_inline(child).strip()}**")
elif name in ("em", "i"):
out.append(f"*{_inline(child).strip()}*")
elif name == "code":
out.append(f"`{child.get_text()}`")
elif name == "br":
out.append("\n")
elif name == "a":
href = child.get("href", "")
label = _inline(child).strip() or href
out.append(f"[{label}]({href})" if href else label)
elif name == "img":
src = child.get("src", "")
alt = child.get("alt", "")
out.append(f"![{alt}]({src})" if src else "")
elif name in ("del", "s", "strike"):
out.append(f"~~{_inline(child).strip()}~~")
else:
out.append(_inline(child))
return re.sub(r"[ \t]+", " ", "".join(out))
def _table(node: Tag) -> str:
rows: list[list[str]] = []
for tr in node.find_all("tr"):
cells = tr.find_all(["th", "td"])
rows.append([_inline(c).strip().replace("|", "\\|") for c in cells])
if not rows:
return ""
width = max(len(r) for r in rows)
rows = [r + [""] * (width - len(r)) for r in rows]
header = "| " + " | ".join(rows[0]) + " |"
sep = "| " + " | ".join(["---"] * width) + " |"
body = "\n".join("| " + " | ".join(r) + " |" for r in rows[1:])
return "\n".join(x for x in (header, sep, body) if x)
def _block(node: Tag, depth: int = 0) -> str:
name = node.name.lower()
if name in ("h1", "h2", "h3", "h4", "h5", "h6"):
return "#" * int(name[1]) + " " + _inline(node).strip()
if name == "p":
return _inline(node).strip()
if name in ("ul", "ol"):
lines = []
for i, li in enumerate(node.find_all("li", recursive=False)):
marker = f"{i + 1}." if name == "ol" else "-"
text = _inline(li).strip()
lines.append(f"{' ' * depth}{marker} {text}")
return "\n".join(lines)
if name == "blockquote":
return "\n".join(f"> {ln}" for ln in _inline(node).strip().splitlines())
if name == "pre":
code = node.get_text()
lang = ""
cls = " ".join(node.get("class", [])) if node.get("class") else ""
m = re.search(r"(?:language|lang)-([\w+-]+)", cls)
if m:
lang = m.group(1)
return f"```{lang}\n{code.rstrip()}\n```"
if name == "hr":
return "---"
if name == "table":
return _table(node)
if name == "img":
src = node.get("src", "")
return f"![{node.get('alt', '')}]({src})" if src else ""
if name in ("div", "section", "article", "body", "main", "html", "span", "font", "center"):
inner = "\n\n".join(
_block(c, depth) for c in node.children if isinstance(c, Tag)
).strip()
if inner:
return inner
text = _inline(node).strip()
return text
return _inline(node).strip()
def _html_to_markdown(html: str) -> str:
soup = BeautifulSoup(html, "html.parser")
for tag in soup(["script", "style", "head", "nav", "footer"]):
tag.decompose()
root = soup.body or soup
blocks = [_block(c) for c in root.children if isinstance(c, Tag)]
md = "\n\n".join(b for b in blocks if b and b.strip())
return re.sub(r"\n{3,}", "\n\n", md).strip()
def _title_from_html(html: str, fallback: str) -> str:
soup = BeautifulSoup(html, "html.parser")
if soup.title and soup.title.string:
return soup.title.string.strip()
h1 = soup.find(["h1", "h2"])
if h1:
return h1.get_text().strip()
return fallback
@register_importer
class HtmlNotesImporter(Importer):
source_id = "html_notes"
label = "HTML (Apple Notes, Bear, Ulysses, OneNote)"
description = "Fichiers HTML ou archive .zip (notes exportées en HTML)."
extensions = (".html", ".htm", ".zip")
order = 50
def detect(self, filename: str, data: bytes) -> bool:
low = filename.lower()
if low.endswith(_HTML_EXTS):
return True
if low.endswith(".zip"):
try:
zf = zipfile.ZipFile(io.BytesIO(data))
except (zipfile.BadZipFile, OSError):
return False
names = [n for n in zf.namelist() if not n.endswith("/")]
return any(n.lower().endswith(_HTML_EXTS) for n in names)
return False
def parse(self, filename: str, data: bytes) -> ImportResult:
result = ImportResult(source=self.source_id)
entries: list[tuple[str, bytes]] = []
if filename.lower().endswith(".zip"):
try:
zf = zipfile.ZipFile(io.BytesIO(data))
except (zipfile.BadZipFile, OSError) as exc:
result.warn(f"Archive invalide : {exc}")
return result.finalize()
for name in zf.namelist():
if name.endswith("/"):
continue
clean = name.replace("\\", "/")
if clean.lower().endswith(_HTML_EXTS):
entries.append((clean, zf.read(name)))
else:
result.attachments.append(ImportAttachment(
source_path=clean,
filename=clean.rsplit("/", 1)[-1],
data=zf.read(name),
))
else:
entries.append((filename, data))
for name, payload in entries:
html = decode_text(payload)
fallback = name.replace("\\", "/").rsplit("/", 1)[-1].rsplit(".", 1)[0]
parts = name.replace("\\", "/").split("/")
result.pages.append(ImportPage(
title=_title_from_html(html, fallback) or "Untitled",
markdown=_html_to_markdown(html),
source_path=name,
parent_path="/".join(parts[:-1]),
external_id=name,
))
return result.finalize()
@register_importer
class GoogleKeepImporter(Importer):
source_id = "google_keep"
label = "Google Keep (Takeout)"
description = "Export Google Takeout : Keep/*.json (notes, listes, labels, pièces jointes)."
extensions = (".json", ".zip")
order = 40
def _is_keep_json(self, data: bytes) -> bool:
try:
obj = json.loads(decode_text(data))
except Exception: # noqa: BLE001
return False
return isinstance(obj, dict) and any(
k in obj for k in ("textContent", "listContent", "isTrashed", "color")
)
def detect(self, filename: str, data: bytes) -> bool:
low = filename.lower()
if low.endswith(".json"):
return self._is_keep_json(data)
if low.endswith(".zip"):
try:
zf = zipfile.ZipFile(io.BytesIO(data))
except (zipfile.BadZipFile, OSError):
return False
for n in zf.namelist():
if n.lower().endswith(".json") and "keep" in n.lower():
try:
if self._is_keep_json(zf.read(n)):
return True
except Exception: # noqa: BLE001
continue
return False
def _page_from_keep(self, obj: dict, name: str) -> ImportPage | None:
if obj.get("isTrashed"):
return None
title = normalize_title(obj.get("title"))
lines: list[str] = []
for item in obj.get("listContent") or []:
mark = "x" if item.get("isChecked") else " "
lines.append(f"- [{mark}] {item.get('text', '')}")
if obj.get("textContent"):
lines.insert(0, obj["textContent"])
body = "\n\n".join(lines)
if not title:
first = next((ln for ln in body.splitlines() if ln.strip()), "")
first = re.sub(r"^[-*+]\s*(\[[ xX]\]\s*)?", "", first).strip()
title = first[:60] or "Note"
labels = coerce_tags(obj.get("labels"))
props = {"tags": labels} if labels else {}
return ImportPage(
title=title,
markdown=body,
source_path=name,
parent_path="",
properties=props,
external_id=name,
)
def parse(self, filename: str, data: bytes) -> ImportResult:
result = ImportResult(source=self.source_id)
if filename.lower().endswith(".zip"):
try:
zf = zipfile.ZipFile(io.BytesIO(data))
except (zipfile.BadZipFile, OSError) as exc:
result.warn(f"Archive invalide : {exc}")
return result.finalize()
for name in zf.namelist():
if name.endswith("/"):
continue
clean = name.replace("\\", "/")
if clean.lower().endswith(".json") and "keep" in clean.lower():
try:
obj = json.loads(decode_text(zf.read(name)))
except Exception: # noqa: BLE001
continue
if not isinstance(obj, dict):
continue
page = self._page_from_keep(obj, clean)
if page:
result.pages.append(page)
elif "/keep/" in clean.lower() and not clean.lower().endswith(".json"):
result.attachments.append(ImportAttachment(
source_path=clean,
filename=clean.rsplit("/", 1)[-1],
data=zf.read(name),
))
return result.finalize()
try:
obj = json.loads(decode_text(data))
except Exception as exc: # noqa: BLE001
result.warn(f"JSON invalide : {exc}")
return result.finalize()
if isinstance(obj, list):
for i, item in enumerate(obj):
if isinstance(item, dict):
page = self._page_from_keep(item, f"{filename}#{i}")
if page:
result.pages.append(page)
else:
page = self._page_from_keep(obj, filename)
if page:
result.pages.append(page)
return result.finalize()