230 lines
8.4 KiB
Python
230 lines
8.4 KiB
Python
"""Markdown rendering pipeline (ROADMAP #85, tranche 9).
|
|
|
|
Helpers extraits de :mod:`backend.main` sans changement de comportement :
|
|
slugification des headings, IDs d'ancrage, rendu mistune singleton,
|
|
wikilinks, normalisation des sauts de ligne et pipeline complet
|
|
:func:`_render_markdown` (rendu + sanitizer XSS BUG-021).
|
|
|
|
Les noms gardent leur préfixe ``_`` d'origine pour un déplacement
|
|
strictement verbatim (tests et routers pointent ici désormais).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import html as html_mod
|
|
import re
|
|
import unicodedata
|
|
from pathlib import Path
|
|
from typing import cast
|
|
|
|
import mistune
|
|
|
|
from backend.image_processor import preprocess_images
|
|
from backend.indexer import find_file_in_index, get_vault_data
|
|
from backend.secret_redactor import redact_with_placeholders, restore_masks
|
|
from backend.services.sanitizer import sanitize_html
|
|
|
|
|
|
def _heading_slugify(text: str) -> str:
|
|
"""Generate a URL-safe slug from heading text.
|
|
|
|
Matches the JavaScript slugify algorithm exactly using
|
|
Unicode-aware character classification:
|
|
1. Strip HTML tags (e.g. wikilink spans rendered inside headings)
|
|
2. Decode HTML entities (e.g. ``&`` → ``&``)
|
|
3. Lowercase
|
|
4. NFD normalize + strip combining marks
|
|
5. Keep only Unicode letters, numbers, spaces, hyphens
|
|
6. Replace spaces with hyphens, collapse multiple hyphens
|
|
|
|
Args:
|
|
text: The heading text content (may contain inline HTML).
|
|
|
|
Returns:
|
|
A URL-safe slug string.
|
|
"""
|
|
# Strip any inline HTML so it does not pollute the slug
|
|
text = re.sub(r"<[^>]+>", "", text)
|
|
# Decode HTML entities so & becomes & before slugification
|
|
text = html_mod.unescape(text)
|
|
text = text.lower()
|
|
text = unicodedata.normalize("NFD", text)
|
|
text = "".join(ch for ch in text if not unicodedata.combining(ch))
|
|
# Unicode-aware: keep letters (L*), numbers (N*), spaces, and hyphens
|
|
cleaned = []
|
|
for ch in text:
|
|
cat = unicodedata.category(ch)
|
|
if cat.startswith('L') or cat.startswith('N') or ch in (' ', '-'):
|
|
cleaned.append(ch)
|
|
text = "".join(cleaned)
|
|
text = re.sub(r"\s+", "-", text)
|
|
text = re.sub(r"-+", "-", text)
|
|
result = text.strip("-")
|
|
return result if result else "heading"
|
|
|
|
|
|
def _add_heading_ids(html: str) -> str:
|
|
"""Post-process rendered HTML to add IDs to heading tags.
|
|
|
|
Adds an ``id`` attribute to every ``<h1>`` through ``<h6>`` tag
|
|
using a slug generated from the heading's text content.
|
|
Duplicate slugs get a ``-2``, ``-3``, etc. suffix.
|
|
|
|
Args:
|
|
html: Rendered HTML string.
|
|
|
|
Returns:
|
|
HTML with heading IDs injected.
|
|
"""
|
|
used_ids: dict[str, int] = {}
|
|
|
|
def _replace_heading(match):
|
|
tag = match.group(1)
|
|
content = match.group(2)
|
|
slug = _heading_slugify(content)
|
|
count = used_ids.get(slug, 0)
|
|
used_ids[slug] = count + 1
|
|
if count > 0:
|
|
slug = f"{slug}-{count + 1}"
|
|
return f'<{tag} id="{slug}">{content}</{tag}>'
|
|
|
|
# Match h1-h6 tags with text content (no existing id attribute)
|
|
return re.sub(
|
|
r'<(h[1-6])>([^<]*(?:<(?!/?h[1-6])[^<]*)*)</h[1-6]>',
|
|
_replace_heading,
|
|
html,
|
|
)
|
|
|
|
|
|
# Cached mistune renderer — avoids re-creating on every request
|
|
_markdown_renderer = mistune.create_markdown(
|
|
escape=False,
|
|
plugins=["table", "strikethrough", "footnotes", "task_lists"],
|
|
)
|
|
|
|
|
|
def _convert_wikilinks(content: str, current_vault: str) -> str:
|
|
"""Convert ``[[wikilinks]]`` and ``[[target|display]]`` to clickable HTML.
|
|
|
|
Supports:
|
|
- Internal file links: ``[[My Note]]`` / ``[[My Note|display]]``
|
|
- Same-document anchors: ``[[#Heading]]`` / ``[[#Heading|display]]``
|
|
|
|
Resolved file links get a ``data-vault`` / ``data-path`` attribute pair.
|
|
Anchor links target the slugified heading ID in the current document.
|
|
Unresolved links are rendered as ``<span class="wikilink-missing">``.
|
|
|
|
Args:
|
|
content: Markdown string potentially containing wikilinks.
|
|
current_vault: Active vault name for resolution priority.
|
|
|
|
Returns:
|
|
Markdown string with wikilinks replaced by HTML anchors.
|
|
"""
|
|
def _replace(match):
|
|
target = match.group(1).strip()
|
|
display = match.group(2).strip() if match.group(2) else target
|
|
|
|
# Same-document anchor link: [[#Heading|display]]
|
|
if target.startswith("#"):
|
|
anchor_text = target[1:].strip()
|
|
anchor_slug = _heading_slugify(anchor_text)
|
|
link_display = display if display != target else anchor_text
|
|
return f'<a class="wikilink-anchor" href="#{anchor_slug}">{link_display}</a>'
|
|
|
|
found = find_file_in_index(target, current_vault)
|
|
if found:
|
|
return (
|
|
f'<a class="wikilink" href="#" '
|
|
f'data-vault="{found["vault"]}" '
|
|
f'data-path="{found["path"]}">{display}</a>'
|
|
)
|
|
return f'<span class="wikilink-missing">{display}</span>'
|
|
|
|
pattern = r'\[\[([^\]|]+)(?:\|([^\]]+))?\]\]'
|
|
return re.sub(pattern, _replace, content)
|
|
|
|
|
|
def _normalize_line_breaks(text: str) -> str:
|
|
"""Convert single newlines to hard breaks (matching Obsidian default behavior).
|
|
|
|
In standard Markdown, a single ``\\n`` is a "soft break" — it renders as a space,
|
|
not a visible line break. Obsidian defaults to treating single newlines as hard
|
|
breaks (equivalent to ``<br>``). This function pre-processes the Markdown source
|
|
so that mistune renders standalone lines on separate rows, while still honouring
|
|
blank lines as paragraph separators.
|
|
|
|
Fenced code blocks (`` ``` ``) are left untouched so their internal newlines are
|
|
preserved verbatim.
|
|
"""
|
|
parts = re.split(r"(```[\s\S]*?```)", text)
|
|
for i, part in enumerate(parts):
|
|
if part.startswith("```"):
|
|
continue # Protect fenced code blocks
|
|
# Single \n (not preceded or followed by another \n) → two spaces + \n
|
|
parts[i] = re.sub(r"(?<!\n)\n(?!\n)", " \n", part)
|
|
return "".join(parts)
|
|
|
|
|
|
def _render_markdown(
|
|
raw_md: str,
|
|
vault_name: str,
|
|
current_file_path: Path | None = None,
|
|
*,
|
|
click_to_copy: bool = False,
|
|
) -> str:
|
|
"""Render a markdown string to HTML with wikilink and image support.
|
|
|
|
Uses the cached singleton mistune renderer for performance.
|
|
|
|
Args:
|
|
raw_md: Raw markdown text (frontmatter already stripped).
|
|
vault_name: Current vault for wikilink resolution context.
|
|
current_file_path: Absolute path to the current markdown file.
|
|
click_to_copy: Restore masked secrets as clickable badges carrying
|
|
the real value (authenticated app preview, feature #188).
|
|
Public shares and PDF exports keep plain labels: the secret
|
|
never reaches their HTML.
|
|
|
|
Returns:
|
|
HTML string.
|
|
"""
|
|
# Get vault data for image resolution
|
|
vault_data = get_vault_data(vault_name)
|
|
vault_root = Path(vault_data["path"]) if vault_data else None
|
|
attachments_path = vault_data.get("config", {}).get("attachmentsPath") if vault_data else None
|
|
|
|
# Redact secrets before rendering (P0 security). Placeholders survive
|
|
# the markdown conversion (fenced code blocks included) and are turned
|
|
# back into visible masks — clickable badges when click_to_copy — right
|
|
# after the HTML is produced (feature #188).
|
|
raw_md, secret_entries = redact_with_placeholders(
|
|
raw_md, str(current_file_path) if current_file_path else ""
|
|
)
|
|
|
|
# Preprocess images first
|
|
if vault_root:
|
|
raw_md = preprocess_images(raw_md, vault_name, vault_root, current_file_path, attachments_path)
|
|
|
|
# Convert wikilinks
|
|
converted = _convert_wikilinks(raw_md, vault_name)
|
|
|
|
# Normalize line breaks to match Obsidian behavior (single \n → hard break)
|
|
converted = _normalize_line_breaks(converted)
|
|
|
|
# mistune 3.3 types `Markdown.__call__` as `str | list[...]` (les
|
|
# renderers HTML renvoient toujours `str` à l'exécution).
|
|
rendered = cast(str, _markdown_renderer(converted))
|
|
|
|
# Restore secret masks (plain labels, or clickable badges carrying the
|
|
# real value on the authenticated app preview — feature #188).
|
|
rendered = restore_masks(rendered, secret_entries, click_to_copy=click_to_copy)
|
|
|
|
# Add heading IDs for TOC navigation
|
|
rendered = _add_heading_ids(rendered)
|
|
|
|
# Sanitize: raw HTML in vault content must never reach the DOM (BUG-021).
|
|
rendered = sanitize_html(rendered)
|
|
|
|
return rendered
|