Files
ObsiGate/backend/export.py
T
bruno 240fd8586e
CI / lint (push) Successful in 1m1s
CI / security (push) Successful in 40s
CI / test (push) Successful in 1m17s
CI / build (push) Successful in 38s
CI / e2e (push) Successful in 10m55s
Desktop Build / build-windows (push) Canceled after 0s
Desktop Build / build-linux (push) Canceled after 0s
fix: corriger 33 erreurs mypy (CI bloquant) + lien README (BUG-003, BUG-004)
BUG-003: annotations de types, gardes None sur get_user(), PdfReader: Any et import PROVIDERS manquant (bug latent main.py:4523). Etape mypy du CI rendue bloquante (etait advisory).

BUG-004: lien README.md -> docs/CONTRIBUTING.md corrige (+ DELIVERY_WORKFLOW.md), arbre projet mis a jour, parite README.fr.md.

Verifie: mypy 0 erreur, ruff OK, pytest 728 passed / 5 skipped, frontend OK, liens md OK.
2026-09-11 14:57:55 -04:00

432 lines
15 KiB
Python

"""
Export utilities for ObsiGate — standalone HTML, Markdown bundle (.zip) and ePub.
Each function converts vault content (markdown notes) into a self-contained
downloadable artifact. Rendering reuses ``mistune`` (already a core dependency)
so no extra markdown engine is required. The ePub generator builds a minimal
but valid EPUB 3 container by hand (``zipfile`` only) so there is no hard
dependency on ``ebooklib``.
All functions raise :class:`ExportError` on any failure (missing file, binary
content, unreadable source, ...) so callers can translate to HTTP errors.
"""
from __future__ import annotations
import base64
import datetime
import html as html_mod
import io
import logging
import mimetypes
import re
import unicodedata
import zipfile
from pathlib import Path
import frontmatter
import mistune
logger = logging.getLogger("obsigate.export")
class ExportError(Exception):
"""Raised when an export operation cannot be completed."""
# Cached mistune renderer (singleton) — mirrors backend.main's configuration.
_markdown = mistune.create_markdown(
escape=False,
plugins=["table", "strikethrough", "footnotes", "task_lists"],
)
# File extensions considered "safe" text files for markdown-like exports.
_MARKDOWN_EXTS = {".md", ".markdown", ".mdown", ".mkd", ".txt"}
_HTML_CSS = """
body {
font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, Helvetica, Arial, sans-serif;
max-width: 760px;
margin: 40px auto;
padding: 0 24px;
line-height: 1.7;
color: #1a1a2e;
font-size: 16px;
}
h1 { font-size: 28px; border-bottom: 2px solid #333; padding-bottom: 8px; margin-top: 0; }
h2 { font-size: 22px; margin-top: 28px; border-bottom: 1px solid #ddd; padding-bottom: 4px; }
h3 { font-size: 18px; margin-top: 24px; }
h4, h5, h6 { font-size: 16px; margin-top: 20px; }
p { margin: 10px 0; }
pre {
background: #f5f5f5;
border: 1px solid #e0e0e0;
border-radius: 6px;
padding: 12px 16px;
font-size: 14px;
overflow-x: auto;
white-space: pre-wrap;
word-wrap: break-word;
}
code { font-size: 14px; background: #f5f5f5; padding: 2px 5px; border-radius: 3px; }
pre code { background: none; padding: 0; }
a { color: #4f46e5; text-decoration: none; }
a:hover { text-decoration: underline; }
img { max-width: 100%; border-radius: 4px; }
blockquote { border-left: 3px solid #ccc; padding-left: 14px; color: #555; margin: 12px 0; }
table { border-collapse: collapse; width: 100%; margin: 12px 0; }
th, td { border: 1px solid #ddd; padding: 6px 10px; text-align: left; font-size: 14px; }
th { background: #f0f0f0; font-weight: 600; }
ul, ol { margin: 8px 0; padding-left: 24px; }
li { margin: 2px 0; }
hr { border: none; border-top: 1px solid #ddd; margin: 20px 0; }
nav.export-nav {
border: 1px solid #e0e0e0;
border-radius: 8px;
padding: 14px 18px;
margin: 16px 0 28px;
background: #fafafa;
}
nav.export-nav h2 { border: none; margin: 0 0 8px; font-size: 15px; text-transform: uppercase; letter-spacing: .04em; }
nav.export-nav ul { margin: 0; padding-left: 20px; }
.wikilink-missing { color: #c0392b; }
"""
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
def _slugify(text: str) -> str:
"""URL-safe slug from arbitrary text (Unicode-aware)."""
text = html_mod.unescape(re.sub(r"<[^>]+>", "", text))
text = text.lower()
text = unicodedata.normalize("NFD", text)
text = "".join(ch for ch in text if not unicodedata.combining(ch))
cleaned = []
for ch in text:
if unicodedata.category(ch).startswith(("L", "N")) or ch in (" ", "-"):
cleaned.append(ch)
text = re.sub(r"\s+", "-", "".join(cleaned))
text = re.sub(r"-+", "-", text)
return text.strip("-") or "document"
def _resolve(vault_path: Path, file_path: Path | str) -> Path:
"""Resolve *file_path* against *vault_path*, accepting absolute or relative.
Returns an absolute :class:`Path`. Does not require the file to exist.
"""
file_path = Path(file_path)
if file_path.is_absolute():
return file_path
return (Path(vault_path) / file_path).resolve()
def _read_text(path: Path) -> str:
"""Read a text file, raising :class:`ExportError` on missing/binary content."""
if not path.exists() or not path.is_file():
raise ExportError(f"File not found: {path}")
try:
raw = path.read_bytes()
except OSError as e:
raise ExportError(f"Cannot read file {path}: {e}") from e
if b"\x00" in raw[:4096]:
raise ExportError(f"File appears to be binary: {path}")
return raw.decode("utf-8", errors="replace")
def _strip_frontmatter(text: str) -> tuple[str, dict]:
"""Split frontmatter from a markdown document.
Returns ``(body, metadata)``. Falls back gracefully when no frontmatter.
"""
try:
post = frontmatter.loads(text)
return post.content, dict(post.metadata or {})
except Exception:
return text, {}
def _safe_name(name: str) -> str:
"""ASCII-safe, filename-safe download name."""
cleaned = "".join(
c for c in name if c.isascii() and (c.isalnum() or c in " _-.")
).strip()
return cleaned or "document"
def _collect_markdown_files(vault_path: Path) -> list[Path]:
"""List all markdown files in the vault, sorted by relative path."""
vault_path = Path(vault_path)
results: list[Path] = []
if not vault_path.is_dir():
return results
for p in sorted(vault_path.rglob("*")):
if p.is_file() and p.suffix.lower() in _MARKDOWN_EXTS:
results.append(p)
return results
# ---------------------------------------------------------------------------
# Markdown → HTML (shared by HTML and ePub exporters)
# ---------------------------------------------------------------------------
def _inline_images(md: str, file_dir: Path, vault_path: Path) -> str:
"""Replace image references with base64 data URIs.
Supports standard markdown ``![alt](src)`` and Obsidian embeds
``![[image.png]]`` / ``![[image.png|300]]``. Sources are resolved relative
to the note's directory, then the vault root.
"""
candidates = [file_dir, vault_path]
def _to_data_uri(src: str) -> str:
src = src.strip().strip("<>")
if src.startswith(("http://", "https://", "data:")):
return src # leave remote/data URLs untouched
src_path = None
for base in candidates:
candidate = (base / src).resolve()
if candidate.is_file():
src_path = candidate
break
if src_path is None:
return src # unresolved — leave as-is
try:
data = src_path.read_bytes()
mime = mimetypes.guess_type(str(src_path))[0] or "application/octet-stream"
b64 = base64.b64encode(data).decode("ascii")
return f"data:{mime};base64,{b64}"
except OSError:
return src
# Obsidian embeds: ![[path]] or ![[path|size]]
def _embed(match):
inner = match.group(1).strip()
target = inner.split("|", 1)[0].strip()
return f"![{target}]({_to_data_uri(target)})"
md = re.sub(r"!\[\[([^\]]+)\]\]", _embed, md)
# Standard markdown images
def _img(match):
alt = match.group(1)
src = match.group(2)
return f"![{alt}]({_to_data_uri(src)})"
md = re.sub(r"!\[([^\]]*)\]\(([^)]+)\)", _img, md)
return md
def _convert_wikilinks(md: str, vault_path: Path, current: Path) -> str:
"""Convert ``[[Note]]`` / ``[[Note|display]]`` to cross-file HTML links."""
md_files = {p.stem.lower(): p for p in _collect_markdown_files(vault_path)}
def _replace(match):
target = match.group(1).strip()
display = (match.group(2) or target).strip()
# Same-document anchor
if target.startswith("#"):
return f'<a href="#{_slugify(target[1:])}">{html_mod.escape(display)}</a>'
# Resolve to another file — link to its exported .html sibling
key = Path(target).stem.lower()
if key in md_files:
href = f"{_slugify(md_files[key].stem)}.html"
return f'<a href="{href}">{html_mod.escape(display)}</a>'
return f'<span class="wikilink-missing">{html_mod.escape(display)}</span>'
return re.sub(r"\[\[([^\]|]+)(?:\|([^\]]+))?\]\]", _replace, md)
def _render_body(md: str, file_dir: Path, vault_path: Path, current: Path) -> str:
"""Render raw markdown to an HTML fragment (images inlined, wikilinks resolved)."""
md = _inline_images(md, file_dir, vault_path)
md = _convert_wikilinks(md, vault_path, current)
return _markdown(md)
def _build_nav(vault_path: Path, current: Path) -> str:
"""Build a navigation block linking to every other markdown note in the vault."""
items = []
for p in _collect_markdown_files(vault_path):
if p.resolve() == Path(current).resolve():
continue
stem = p.stem
items.append(f'<li><a href="{_slugify(stem)}.html">{html_mod.escape(stem)}</a></li>')
if not items:
return ""
return (
'<nav class="export-nav"><h2>Notes</h2><ul>'
+ "".join(items)
+ "</ul></nav>"
)
# ---------------------------------------------------------------------------
# Public exporters
# ---------------------------------------------------------------------------
def export_html(vault_path: Path, file_path: Path) -> bytes:
"""Convert a markdown note into a standalone HTML document.
Images are inlined as base64 data URIs, wikilinks become cross-file links,
and a navigation block lists every other note in the vault.
Returns:
UTF-8 encoded HTML as bytes.
"""
vault_path = Path(vault_path)
path = _resolve(vault_path, file_path)
raw = _read_text(path)
body, metadata = _strip_frontmatter(raw)
title = str(metadata.get("title") or path.stem)
rendered = _render_body(body, path.parent, vault_path, path)
nav = _build_nav(vault_path, path)
html_doc = f"""<!DOCTYPE html>
<html lang="fr">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<meta name="generator" content="ObsiGate">
<title>{html_mod.escape(title)}</title>
<style>{_HTML_CSS}</style>
</head>
<body>
<header><h1>{html_mod.escape(title)}</h1></header>
{nav}
{rendered}
</body>
</html>"""
return html_doc.encode("utf-8")
def export_md_bundle(vault_path: Path, directory_path: Path) -> bytes:
"""Zip a directory (or single file) of markdown, preserving structure.
Returns:
ZIP archive as bytes.
"""
vault_path = Path(vault_path)
path = _resolve(vault_path, directory_path)
if not path.exists():
raise ExportError(f"Path not found: {path}")
buffer = io.BytesIO()
try:
with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as zf:
if path.is_dir():
base = path
for p in sorted(path.rglob("*")):
if p.is_file():
arcname = p.relative_to(base).as_posix()
zf.write(p, arcname=arcname)
# Avoid emitting an empty (invalid) zip for an empty directory
if not any(p.is_file() for p in path.rglob("*")):
zf.writestr(".empty", "")
else:
zf.write(path, arcname=path.name)
except OSError as e:
raise ExportError(f"Cannot create bundle: {e}") from e
return buffer.getvalue()
def export_epub(vault_path: Path, file_path: Path) -> bytes:
"""Convert a markdown note into a minimal, valid EPUB 3 container.
Uses only ``zipfile`` + ``mistune`` (no ``ebooklib`` dependency). Images
are inlined as base64 so the ePub is fully self-contained.
Returns:
EPUB archive as bytes.
"""
vault_path = Path(vault_path)
path = _resolve(vault_path, file_path)
raw = _read_text(path)
body, metadata = _strip_frontmatter(raw)
title = str(metadata.get("title") or path.stem)
author = str(metadata.get("author") or "ObsiGate")
rendered = _render_body(body, path.parent, vault_path, path)
uid = f"obsigate-{_slugify(title)}-{int(datetime.datetime.now(tz=datetime.timezone.utc).timestamp())}"
xhtml = f"""<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE html>
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="fr" lang="fr">
<head>
<title>{html_mod.escape(title)}</title>
<style>{_HTML_CSS}</style>
</head>
<body>
<h1>{html_mod.escape(title)}</h1>
{rendered}
</body>
</html>"""
# EPUB requires a valid XHTML-ish body; strip our HTML5-only doctype is not
# needed since we emit XML directly above. Escape any bare ampersands in
# text nodes outside tags is delegated to mistune's own escaping.
content_opf = f"""<?xml version="1.0" encoding="utf-8"?>
<package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="bookid">
<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">
<dc:identifier id="bookid">{html_mod.escape(uid)}</dc:identifier>
<dc:title>{html_mod.escape(title)}</dc:title>
<dc:creator>{html_mod.escape(author)}</dc:creator>
<dc:language>fr</dc:language>
<meta property="dcterms:modified">{datetime.datetime.now(datetime.timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ')}</meta>
</metadata>
<manifest>
<item id="ncx" href="toc.ncx" media-type="application/x-dtbncx+xml"/>
<item id="chapter1" href="chapter1.xhtml" media-type="application/xhtml+xml"/>
</manifest>
<spine toc="ncx">
<itemref idref="chapter1"/>
</spine>
</package>"""
toc_ncx = f"""<?xml version="1.0" encoding="utf-8"?>
<ncx xmlns="http://www.daisy.org/z3986/2005/ncx/" version="2005-1">
<head>
<meta name="dtb:uid" content="{html_mod.escape(uid)}"/>
</head>
<docTitle><text>{html_mod.escape(title)}</text></docTitle>
<navMap>
<navPoint id="navpoint-1" playOrder="1">
<navLabel><text>{html_mod.escape(title)}</text></navLabel>
<content src="chapter1.xhtml"/>
</navPoint>
</navMap>
</ncx>"""
container_xml = """<?xml version="1.0" encoding="utf-8"?>
<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
<rootfiles>
<rootfile full-path="OEBPS/content.opf" media-type="application/oebps-package+xml"/>
</rootfiles>
</container>"""
buffer = io.BytesIO()
try:
with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as zf:
# `mimetype` must be the first entry, stored (uncompressed).
zf.writestr(
zipfile.ZipInfo("mimetype"),
"application/epub+zip",
compress_type=zipfile.ZIP_STORED,
)
zf.writestr("META-INF/container.xml", container_xml)
zf.writestr("OEBPS/content.opf", content_opf)
zf.writestr("OEBPS/toc.ncx", toc_ncx)
zf.writestr("OEBPS/chapter1.xhtml", xhtml)
except OSError as e:
raise ExportError(f"Cannot create ePub: {e}") from e
return buffer.getvalue()