"""FlowDeck — HTML notes & Google Keep importer (v5.6.0, Phase 1). Covers HTML exports from Apple Notes, Bear, Ulysses and OneNote, plus the Google Takeout ``Keep`` JSON/HTML format. HTML is converted to Markdown and then to FlowDeck blocks by the pipeline. """ from __future__ import annotations import io import json import re import zipfile from bs4 import BeautifulSoup, NavigableString, Tag from app.services.importers._common import coerce_tags, normalize_title from app.services.importers.base import ( ImportAttachment, Importer, ImportPage, ImportResult, decode_text, register_importer, ) _HTML_EXTS = (".html", ".htm") def _inline(node: Tag) -> str: out: list[str] = [] for child in node.children: if isinstance(child, NavigableString): out.append(str(child)) elif isinstance(child, Tag): name = child.name.lower() if name in ("strong", "b"): out.append(f"**{_inline(child).strip()}**") elif name in ("em", "i"): out.append(f"*{_inline(child).strip()}*") elif name == "code": out.append(f"`{child.get_text()}`") elif name == "br": out.append("\n") elif name == "a": href = child.get("href", "") label = _inline(child).strip() or href out.append(f"[{label}]({href})" if href else label) elif name == "img": src = child.get("src", "") alt = child.get("alt", "") out.append(f"![{alt}]({src})" if src else "") elif name in ("del", "s", "strike"): out.append(f"~~{_inline(child).strip()}~~") else: out.append(_inline(child)) return re.sub(r"[ \t]+", " ", "".join(out)) def _table(node: Tag) -> str: rows: list[list[str]] = [] for tr in node.find_all("tr"): cells = tr.find_all(["th", "td"]) rows.append([_inline(c).strip().replace("|", "\\|") for c in cells]) if not rows: return "" width = max(len(r) for r in rows) rows = [r + [""] * (width - len(r)) for r in rows] header = "| " + " | ".join(rows[0]) + " |" sep = "| " + " | ".join(["---"] * width) + " |" body = "\n".join("| " + " | ".join(r) + " |" for r in rows[1:]) return "\n".join(x for x in (header, sep, body) if x) def _block(node: Tag, depth: int = 0) -> str: name = node.name.lower() if name in ("h1", "h2", "h3", "h4", "h5", "h6"): return "#" * int(name[1]) + " " + _inline(node).strip() if name == "p": return _inline(node).strip() if name in ("ul", "ol"): lines = [] for i, li in enumerate(node.find_all("li", recursive=False)): marker = f"{i + 1}." if name == "ol" else "-" text = _inline(li).strip() lines.append(f"{' ' * depth}{marker} {text}") return "\n".join(lines) if name == "blockquote": return "\n".join(f"> {ln}" for ln in _inline(node).strip().splitlines()) if name == "pre": code = node.get_text() lang = "" cls = " ".join(node.get("class", [])) if node.get("class") else "" m = re.search(r"(?:language|lang)-([\w+-]+)", cls) if m: lang = m.group(1) return f"```{lang}\n{code.rstrip()}\n```" if name == "hr": return "---" if name == "table": return _table(node) if name == "img": src = node.get("src", "") return f"![{node.get('alt', '')}]({src})" if src else "" if name in ("div", "section", "article", "body", "main", "html", "span", "font", "center"): inner = "\n\n".join( _block(c, depth) for c in node.children if isinstance(c, Tag) ).strip() if inner: return inner text = _inline(node).strip() return text return _inline(node).strip() def _html_to_markdown(html: str) -> str: soup = BeautifulSoup(html, "html.parser") for tag in soup(["script", "style", "head", "nav", "footer"]): tag.decompose() root = soup.body or soup blocks = [_block(c) for c in root.children if isinstance(c, Tag)] md = "\n\n".join(b for b in blocks if b and b.strip()) return re.sub(r"\n{3,}", "\n\n", md).strip() def _title_from_html(html: str, fallback: str) -> str: soup = BeautifulSoup(html, "html.parser") if soup.title and soup.title.string: return soup.title.string.strip() h1 = soup.find(["h1", "h2"]) if h1: return h1.get_text().strip() return fallback @register_importer class HtmlNotesImporter(Importer): source_id = "html_notes" label = "HTML (Apple Notes, Bear, Ulysses, OneNote)" description = "Fichiers HTML ou archive .zip (notes exportées en HTML)." extensions = (".html", ".htm", ".zip") order = 50 def detect(self, filename: str, data: bytes) -> bool: low = filename.lower() if low.endswith(_HTML_EXTS): return True if low.endswith(".zip"): try: zf = zipfile.ZipFile(io.BytesIO(data)) except (zipfile.BadZipFile, OSError): return False names = [n for n in zf.namelist() if not n.endswith("/")] return any(n.lower().endswith(_HTML_EXTS) for n in names) return False def parse(self, filename: str, data: bytes) -> ImportResult: result = ImportResult(source=self.source_id) entries: list[tuple[str, bytes]] = [] if filename.lower().endswith(".zip"): try: zf = zipfile.ZipFile(io.BytesIO(data)) except (zipfile.BadZipFile, OSError) as exc: result.warn(f"Archive invalide : {exc}") return result.finalize() for name in zf.namelist(): if name.endswith("/"): continue clean = name.replace("\\", "/") if clean.lower().endswith(_HTML_EXTS): entries.append((clean, zf.read(name))) else: result.attachments.append(ImportAttachment( source_path=clean, filename=clean.rsplit("/", 1)[-1], data=zf.read(name), )) else: entries.append((filename, data)) for name, payload in entries: html = decode_text(payload) fallback = name.replace("\\", "/").rsplit("/", 1)[-1].rsplit(".", 1)[0] parts = name.replace("\\", "/").split("/") result.pages.append(ImportPage( title=_title_from_html(html, fallback) or "Untitled", markdown=_html_to_markdown(html), source_path=name, parent_path="/".join(parts[:-1]), external_id=name, )) return result.finalize() @register_importer class GoogleKeepImporter(Importer): source_id = "google_keep" label = "Google Keep (Takeout)" description = "Export Google Takeout : Keep/*.json (notes, listes, labels, pièces jointes)." extensions = (".json", ".zip") order = 40 def _is_keep_json(self, data: bytes) -> bool: try: obj = json.loads(decode_text(data)) except Exception: # noqa: BLE001 return False return isinstance(obj, dict) and any( k in obj for k in ("textContent", "listContent", "isTrashed", "color") ) def detect(self, filename: str, data: bytes) -> bool: low = filename.lower() if low.endswith(".json"): return self._is_keep_json(data) if low.endswith(".zip"): try: zf = zipfile.ZipFile(io.BytesIO(data)) except (zipfile.BadZipFile, OSError): return False for n in zf.namelist(): if n.lower().endswith(".json") and "keep" in n.lower(): try: if self._is_keep_json(zf.read(n)): return True except Exception: # noqa: BLE001 continue return False def _page_from_keep(self, obj: dict, name: str) -> ImportPage | None: if obj.get("isTrashed"): return None title = normalize_title(obj.get("title")) lines: list[str] = [] for item in obj.get("listContent") or []: mark = "x" if item.get("isChecked") else " " lines.append(f"- [{mark}] {item.get('text', '')}") if obj.get("textContent"): lines.insert(0, obj["textContent"]) body = "\n\n".join(lines) if not title: first = next((ln for ln in body.splitlines() if ln.strip()), "") first = re.sub(r"^[-*+]\s*(\[[ xX]\]\s*)?", "", first).strip() title = first[:60] or "Note" labels = coerce_tags(obj.get("labels")) props = {"tags": labels} if labels else {} return ImportPage( title=title, markdown=body, source_path=name, parent_path="", properties=props, external_id=name, ) def parse(self, filename: str, data: bytes) -> ImportResult: result = ImportResult(source=self.source_id) if filename.lower().endswith(".zip"): try: zf = zipfile.ZipFile(io.BytesIO(data)) except (zipfile.BadZipFile, OSError) as exc: result.warn(f"Archive invalide : {exc}") return result.finalize() for name in zf.namelist(): if name.endswith("/"): continue clean = name.replace("\\", "/") if clean.lower().endswith(".json") and "keep" in clean.lower(): try: obj = json.loads(decode_text(zf.read(name))) except Exception: # noqa: BLE001 continue if not isinstance(obj, dict): continue page = self._page_from_keep(obj, clean) if page: result.pages.append(page) elif "/keep/" in clean.lower() and not clean.lower().endswith(".json"): result.attachments.append(ImportAttachment( source_path=clean, filename=clean.rsplit("/", 1)[-1], data=zf.read(name), )) return result.finalize() try: obj = json.loads(decode_text(data)) except Exception as exc: # noqa: BLE001 result.warn(f"JSON invalide : {exc}") return result.finalize() if isinstance(obj, list): for i, item in enumerate(obj): if isinstance(item, dict): page = self._page_from_keep(item, f"{filename}#{i}") if page: result.pages.append(page) else: page = self._page_from_keep(obj, filename) if page: result.pages.append(page) return result.finalize()