fix(export): v4.7.2 contenu des documents absent des exports MD/HTML/PDF
Le service d'export ne lisait que les pages content_format='blocks'. Les docs stockees autrement sortaient avec le seul titre: - content_format='file' (.md/code uploades): content=JSON meta, le vrai texte est sur disque (/data/uploads/workspace_*) -> n'etait jamais lu. - content_format='markdown': HTML/PDF enveloppait chaque ligne en <p> (headings et listes aplatis). Resolution de la vraie source pour les 3 formats (app/services/export.py): - lit le fichier upload sur disque pour les pages file (repertoire via FLOWDECK_DATA_DIR, defaut /data), - rend les pages markdown / fichiers .md en blocs (headings, listes, code, quote, todo) pour un HTML/PDF riche, - fichiers texte non-markdown -> bloc de code, - binaires (PDF/images) ignores. 195/195 tests (5 nouveaux v4.7.2). Verifie en reel via HTTP sur README.md et l'arborescence Base de Connaissances (contenu complet dans les 3 formats).
This commit is contained in:
+2
-2
@@ -39,13 +39,13 @@ async def lifespan(_app: FastAPI):
|
||||
(admin_hash,)
|
||||
)
|
||||
conn.commit()
|
||||
logger.info("FlowDeck v4.7.1 started on port %d", settings.app_port)
|
||||
logger.info("FlowDeck v4.7.2 started on port %d", settings.app_port)
|
||||
yield
|
||||
|
||||
|
||||
app = FastAPI(
|
||||
title="FlowDeck",
|
||||
version="4.7.1",
|
||||
version="4.7.2",
|
||||
docs_url="/docs" if settings.log_level == "DEBUG" else None,
|
||||
redoc_url=None,
|
||||
lifespan=lifespan,
|
||||
|
||||
+253
-16
@@ -1,16 +1,26 @@
|
||||
"""FlowDeck — Export service (v4.7.0).
|
||||
"""FlowDeck — Export service (v4.7.2).
|
||||
|
||||
Four types of export, all generated server-side:
|
||||
- Markdown (``page_to_markdown``): title + blocks + récursif sous-pages
|
||||
- HTML (``page_to_standalone_html``): document autonome (styles inline)
|
||||
- PDF (``page_to_pdf_bytes``): convertit un HTML print-friendly
|
||||
- Site (``build_static_site``): site statique multi-pages (zip)
|
||||
|
||||
Supports the three ways a page's content can be stored:
|
||||
- content_format == "blocks" -> JSON list of blocks in ``content``
|
||||
- content_format == "markdown" -> raw Markdown in ``content``
|
||||
- content_format == "file" -> ``content`` is JSON metadata; the real text
|
||||
lives in an uploaded file on disk (uploads/workspace_*). We read it back so
|
||||
an exported document carries its actual content, not just its title.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
from urllib.parse import quote
|
||||
|
||||
from app.db import get_conn
|
||||
@@ -54,6 +64,242 @@ def _block_text(b: dict) -> str:
|
||||
return _text(b.get("content"), escape=False)
|
||||
|
||||
|
||||
# ── Source resolution: read real textual content for ANY page type ──
|
||||
|
||||
# Extensions whose content is plain text / code / markdown (textual exportable).
|
||||
_TEXTUAL_EXTS = {
|
||||
"md", "markdown", "txt", "log", "text",
|
||||
"py", "js", "ts", "jsx", "tsx", "html", "htm", "css", "json", "xml",
|
||||
"yaml", "yml", "toml", "ini", "cfg", "conf", "env", "sh", "bash", "zsh",
|
||||
"ps1", "bat", "cmd", "rb", "go", "rs", "java", "c", "cpp", "h", "hpp",
|
||||
"php", "swift", "kt", "scala", "sql", "r", "vue", "svelte", "astro",
|
||||
"properties", "gitignore", "dockerfile", "makefile",
|
||||
}
|
||||
_CODE_LANG = {
|
||||
"py": "python", "js": "javascript", "ts": "typescript", "jsx": "javascript",
|
||||
"tsx": "typescript", "html": "html", "htm": "html", "css": "css",
|
||||
"json": "json", "xml": "xml", "yaml": "yaml", "yml": "yaml",
|
||||
"toml": "toml", "ini": "ini", "cfg": "ini", "conf": "ini", "env": "ini",
|
||||
"sh": "bash", "bash": "bash", "zsh": "bash", "ps1": "powershell",
|
||||
"bat": "batch", "cmd": "batch", "rb": "ruby", "go": "go", "rs": "rust",
|
||||
"java": "java", "c": "c", "cpp": "cpp", "h": "c", "hpp": "cpp",
|
||||
"php": "php", "swift": "swift", "kt": "kotlin", "scala": "scala",
|
||||
"sql": "sql", "r": "r", "vue": "html", "svelte": "html",
|
||||
"astro": "html", "properties": "ini", "md": "markdown",
|
||||
"markdown": "markdown", "txt": "plaintext", "log": "plaintext",
|
||||
"text": "plaintext",
|
||||
}
|
||||
_MARKDOWN_MIMES = {"text/markdown", "text/x-markdown", "application/octet-stream"}
|
||||
|
||||
|
||||
def _data_root() -> Path:
|
||||
"""Directory that contains ``uploads/`` (mirrors dashboard.py /data)."""
|
||||
return Path(os.environ.get("FLOWDECK_DATA_DIR", "/data"))
|
||||
|
||||
|
||||
def _file_meta(page: dict) -> dict:
|
||||
try:
|
||||
meta = json.loads(page.get("content") or "{}")
|
||||
return meta if isinstance(meta, dict) else {}
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
return {}
|
||||
|
||||
|
||||
def _file_text(page: dict) -> str | None:
|
||||
"""Return the textual content of an uploaded ``file`` page, or None.
|
||||
|
||||
Only reads plain-text / code / markdown files. Binary (PDF, images…)
|
||||
returns None and is skipped by exporters (nothing meaningful to include).
|
||||
"""
|
||||
if (page.get("content_format") or "") != "file":
|
||||
return None
|
||||
meta = _file_meta(page)
|
||||
rel = (meta.get("file_path") or "").replace("\\", "/").strip()
|
||||
if not rel or ".." in rel.replace("\\", "/").split("/") or not rel.startswith("uploads/"):
|
||||
return None
|
||||
name = (rel.rsplit("/", 1)[-1] or "").lower()
|
||||
ext = name.rsplit(".", 1)[-1] if "." in name else ""
|
||||
mime = (meta.get("mime_type") or "").lower()
|
||||
if not (ext in _TEXTUAL_EXTS or mime.startswith("text/")):
|
||||
return None
|
||||
try:
|
||||
full = (_data_root() / rel).resolve()
|
||||
root = _data_root().resolve()
|
||||
if root not in full.parents:
|
||||
return None
|
||||
return full.read_text(encoding="utf-8", errors="replace")
|
||||
except (OSError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _page_source(page: dict):
|
||||
"""Return (kind, payload) describing where the page's real content lives.
|
||||
|
||||
kind ∈ {"blocks", "md", "code"}:
|
||||
- "blocks": payload is the block list (block editor pages)
|
||||
- "md" : payload is raw Markdown text
|
||||
- "code" : payload is (text, language)
|
||||
An empty/unsupported page yields ("blocks", []).
|
||||
"""
|
||||
fmt = (page.get("content_format") or "blocks")
|
||||
content = page.get("content") or ""
|
||||
|
||||
if fmt == "blocks":
|
||||
return "blocks", _blocks_of(page)
|
||||
|
||||
if fmt == "markdown":
|
||||
if content.strip():
|
||||
return "md", content
|
||||
return "blocks", []
|
||||
|
||||
if fmt == "file":
|
||||
text = _file_text(page)
|
||||
if text is None:
|
||||
return "blocks", []
|
||||
meta = _file_meta(page)
|
||||
name = (meta.get("file_path") or "").replace("\\", "/").rsplit("/", 1)[-1].lower()
|
||||
ext = name.rsplit(".", 1)[-1] if "." in name else ""
|
||||
mime = (meta.get("mime_type") or "").lower()
|
||||
if ext in ("md", "markdown") or mime in _MARKDOWN_MIMES or mime.startswith("text/markdown"):
|
||||
return "md", text
|
||||
lang = _CODE_LANG.get(ext, "plaintext")
|
||||
return "code", (text, lang)
|
||||
|
||||
# Unknown format (e.g. legacy) -> try to dump as raw text
|
||||
if content.strip():
|
||||
return "md", content
|
||||
return "blocks", []
|
||||
|
||||
|
||||
# ── Markdown renderer (raw markdown → exportable fragments) ──
|
||||
|
||||
def _md_to_blocks(md: str) -> list:
|
||||
"""Convert raw Markdown text into the same lightweight block list the
|
||||
editor produces (headings, lists, to-do, quote, code, divider, paragraph).
|
||||
|
||||
Kept intentionally simple: inline formatting (bold/links) is preserved as
|
||||
literal text, matching how the block editor treats imported .md files.
|
||||
"""
|
||||
blocks: list = []
|
||||
buf = md.replace("\r\n", "\n").replace("\r", "\n")
|
||||
lines = buf.split("\n")
|
||||
i = 0
|
||||
n = len(lines)
|
||||
para: list[str] = []
|
||||
|
||||
def flush_para():
|
||||
nonlocal para
|
||||
if para:
|
||||
blocks.append({"type": "paragraph", "content": "\n".join(para).strip()})
|
||||
para = []
|
||||
|
||||
while i < n:
|
||||
line = lines[i].rstrip()
|
||||
stripped = line.strip()
|
||||
if not stripped:
|
||||
flush_para()
|
||||
i += 1
|
||||
continue
|
||||
if stripped.startswith("```") or stripped.startswith("~~~"):
|
||||
flush_para()
|
||||
fence = stripped[0:3]
|
||||
lang = stripped[3:].strip()
|
||||
i += 1
|
||||
code: list[str] = []
|
||||
while i < n and not lines[i].strip().startswith(fence):
|
||||
code.append(lines[i])
|
||||
i += 1
|
||||
if i < n:
|
||||
i += 1 # closing fence
|
||||
blocks.append({"type": "code", "content": "\n".join(code), "language": lang})
|
||||
continue
|
||||
m = re.match(r"^(#{1,6})\s+(.*)$", stripped)
|
||||
if m and line == stripped: # ATX heading must be whole line
|
||||
level = len(m.group(1))
|
||||
flush_para()
|
||||
blocks.append({"type": f"heading_{min(level, 4)}", "content": m.group(2).strip()})
|
||||
i += 1
|
||||
continue
|
||||
if stripped == "---" or stripped == "***" or stripped == "___":
|
||||
flush_para()
|
||||
blocks.append({"type": "divider", "content": ""})
|
||||
i += 1
|
||||
continue
|
||||
if re.match(r"^\s*[-*+]\s+", line):
|
||||
flush_para()
|
||||
while i < n:
|
||||
s = lines[i].strip()
|
||||
m2 = re.match(r"^[-*+]\s+(.*)$", s)
|
||||
if not m2:
|
||||
break
|
||||
blocks.append({"type": "bulleted_list", "content": m2.group(1).strip()})
|
||||
i += 1
|
||||
continue
|
||||
if re.match(r"^\s*\d+[.)]\s+", line):
|
||||
flush_para()
|
||||
while i < n:
|
||||
s = lines[i].strip()
|
||||
m2 = re.match(r"^\d+[.)]\s+(.*)$", s)
|
||||
if not m2:
|
||||
break
|
||||
blocks.append({"type": "numbered_list", "content": m2.group(1).strip()})
|
||||
i += 1
|
||||
continue
|
||||
mtodo = re.match(r"^\s*[-*+]\s+\[([ xX])\]\s+(.*)$", stripped)
|
||||
if mtodo:
|
||||
flush_para()
|
||||
while i < n:
|
||||
s = lines[i].strip()
|
||||
m2 = re.match(r"^[-*+]\s+\[([ xX])\]\s+(.*)$", s)
|
||||
if not m2:
|
||||
break
|
||||
blocks.append({
|
||||
"type": "to_do",
|
||||
"content": m2.group(2).strip(),
|
||||
"checked": m2.group(1).lower() == "x",
|
||||
})
|
||||
i += 1
|
||||
continue
|
||||
mq = re.match(r"^>\s?(.*)$", stripped)
|
||||
if mq and line == stripped:
|
||||
flush_para()
|
||||
while i < n:
|
||||
s = lines[i].strip()
|
||||
m2 = re.match(r"^>\s?(.*)$", s)
|
||||
if not m2:
|
||||
break
|
||||
para.append(m2.group(1))
|
||||
i += 1
|
||||
blocks.append({"type": "quote", "content": "\n".join(para)})
|
||||
para = []
|
||||
continue
|
||||
para.append(stripped)
|
||||
i += 1
|
||||
flush_para()
|
||||
return blocks
|
||||
|
||||
|
||||
def _page_blocks(page: dict) -> list:
|
||||
"""Blocks used for HTML/PDF rendering regardless of storage format."""
|
||||
kind, payload = _page_source(page)
|
||||
if kind == "blocks":
|
||||
return payload
|
||||
if kind == "code":
|
||||
text, lang = payload
|
||||
return [{"type": "code", "content": text, "language": lang}] if text else []
|
||||
if kind == "md":
|
||||
return _md_to_blocks(payload)
|
||||
return []
|
||||
|
||||
|
||||
def _page_markdown_source(page: dict) -> str:
|
||||
"""Raw markdown when the page IS markdown-sourced, else empty string."""
|
||||
kind, payload = _page_source(page)
|
||||
if kind == "md":
|
||||
return payload
|
||||
return ""
|
||||
|
||||
|
||||
# ═══════════════ Markdown ═══════════════
|
||||
|
||||
def blocks_to_markdown(blocks: list) -> str:
|
||||
@@ -117,11 +363,11 @@ def _child_pages(page: dict) -> list:
|
||||
def page_to_markdown(page: dict, *, include_children: bool = True) -> str:
|
||||
"""Markdown for a single page, with optional sub-pages appended."""
|
||||
parts = [f"# {_page_title(page)}", ""]
|
||||
content_inner = _blocks_of(page)
|
||||
if page.get("content_format") == "markdown" and (page.get("content") or "").strip():
|
||||
parts.append((page["content"] or "").strip())
|
||||
md_source = _page_markdown_source(page)
|
||||
if md_source:
|
||||
parts.append(md_source.strip())
|
||||
else:
|
||||
md = blocks_to_markdown(content_inner)
|
||||
md = blocks_to_markdown(_page_blocks(page))
|
||||
if md:
|
||||
parts.append(md)
|
||||
md = "\n\n".join(filter(None, parts)).rstrip()
|
||||
@@ -253,11 +499,7 @@ def page_to_standalone_html(
|
||||
) -> str:
|
||||
"""Return a standalone, self-contained HTML document for a page."""
|
||||
title = _page_title(page)
|
||||
content_inner = _blocks_of(page)
|
||||
if page.get("content_format") == "markdown" and (page.get("content") or "").strip():
|
||||
body = "\n".join(f"<p>{_text(line)}</p>" for line in (page.get("content") or "").splitlines())
|
||||
else:
|
||||
body = blocks_to_html(content_inner)
|
||||
body = blocks_to_html(_page_blocks(page))
|
||||
|
||||
meta_updated = page.get("updated_at") or ""
|
||||
footer = f"<div class='footer'><span>FlowDeck · {_page_title(page)}</span><span>{meta_updated}</span></div>"
|
||||
@@ -293,11 +535,7 @@ def page_to_standalone_html(
|
||||
def _pdf_html(page: dict) -> str:
|
||||
"""A print-friendly, minimal-CSS HTML for PDF conversion."""
|
||||
title = _page_title(page)
|
||||
content_inner = _blocks_of(page)
|
||||
if page.get("content_format") == "markdown" and (page.get("content") or "").strip():
|
||||
body = "\n".join(f"<p>{_text(line)}</p>" for line in (page.get("content") or "").splitlines())
|
||||
else:
|
||||
body = blocks_to_html(content_inner)
|
||||
body = blocks_to_html(_page_blocks(page))
|
||||
return f"""<html><head><meta charset="utf-8"><title>{_text(title)}</title>
|
||||
<style>
|
||||
body{{font-family:Helvetica,Arial,sans-serif;color:#1f2328;font-size:12px;line-height:1.5;}}
|
||||
@@ -339,7 +577,6 @@ def _site_index_html(pages: list[dict]) -> str:
|
||||
"""Build the index.html of the static site (list of all pages)."""
|
||||
def link(p: dict) -> str:
|
||||
title = _page_title(p)
|
||||
slug = quote(f"{title}", safe="")
|
||||
return f'<li><a href="{quote(title, safe="")}.html">{_text(title)}</a></li>'
|
||||
|
||||
items = "".join(link(p) for p in pages)
|
||||
|
||||
Reference in New Issue
Block a user