fix(export): v4.7.2 contenu des documents absent des exports MD/HTML/PDF
FlowDeck CI / test (push) Failing after 19s
FlowDeck CI / docker (push) Skipped

Le service d'export ne lisait que les pages content_format='blocks'. Les docs
stockees autrement sortaient avec le seul titre:
- content_format='file' (.md/code uploades): content=JSON meta, le vrai texte
  est sur disque (/data/uploads/workspace_*) -> n'etait jamais lu.
- content_format='markdown': HTML/PDF enveloppait chaque ligne en <p> (headings
  et listes aplatis).

Resolution de la vraie source pour les 3 formats (app/services/export.py):
- lit le fichier upload sur disque pour les pages file (repertoire via
  FLOWDECK_DATA_DIR, defaut /data),
- rend les pages markdown / fichiers .md en blocs (headings, listes, code,
  quote, todo) pour un HTML/PDF riche,
- fichiers texte non-markdown -> bloc de code,
- binaires (PDF/images) ignores.

195/195 tests (5 nouveaux v4.7.2). Verifie en reel via HTTP sur README.md et
l'arborescence Base de Connaissances (contenu complet dans les 3 formats).
This commit is contained in:
2026-09-03 01:29:35 -04:00
parent 5390a6ceab
commit fa97f07ec8
6 changed files with 417 additions and 21 deletions
+2 -2
View File
@@ -39,13 +39,13 @@ async def lifespan(_app: FastAPI):
(admin_hash,)
)
conn.commit()
logger.info("FlowDeck v4.7.1 started on port %d", settings.app_port)
logger.info("FlowDeck v4.7.2 started on port %d", settings.app_port)
yield
app = FastAPI(
title="FlowDeck",
version="4.7.1",
version="4.7.2",
docs_url="/docs" if settings.log_level == "DEBUG" else None,
redoc_url=None,
lifespan=lifespan,
+253 -16
View File
@@ -1,16 +1,26 @@
"""FlowDeck — Export service (v4.7.0).
"""FlowDeck — Export service (v4.7.2).
Four types of export, all generated server-side:
- Markdown (``page_to_markdown``): title + blocks + récursif sous-pages
- HTML (``page_to_standalone_html``): document autonome (styles inline)
- PDF (``page_to_pdf_bytes``): convertit un HTML print-friendly
- Site (``build_static_site``): site statique multi-pages (zip)
Supports the three ways a page's content can be stored:
- content_format == "blocks" -> JSON list of blocks in ``content``
- content_format == "markdown" -> raw Markdown in ``content``
- content_format == "file" -> ``content`` is JSON metadata; the real text
lives in an uploaded file on disk (uploads/workspace_*). We read it back so
an exported document carries its actual content, not just its title.
"""
from __future__ import annotations
import io
import json
import os
import re
import zipfile
from pathlib import Path
from urllib.parse import quote
from app.db import get_conn
@@ -54,6 +64,242 @@ def _block_text(b: dict) -> str:
return _text(b.get("content"), escape=False)
# ── Source resolution: read real textual content for ANY page type ──
# Extensions whose content is plain text / code / markdown (textual exportable).
_TEXTUAL_EXTS = {
"md", "markdown", "txt", "log", "text",
"py", "js", "ts", "jsx", "tsx", "html", "htm", "css", "json", "xml",
"yaml", "yml", "toml", "ini", "cfg", "conf", "env", "sh", "bash", "zsh",
"ps1", "bat", "cmd", "rb", "go", "rs", "java", "c", "cpp", "h", "hpp",
"php", "swift", "kt", "scala", "sql", "r", "vue", "svelte", "astro",
"properties", "gitignore", "dockerfile", "makefile",
}
_CODE_LANG = {
"py": "python", "js": "javascript", "ts": "typescript", "jsx": "javascript",
"tsx": "typescript", "html": "html", "htm": "html", "css": "css",
"json": "json", "xml": "xml", "yaml": "yaml", "yml": "yaml",
"toml": "toml", "ini": "ini", "cfg": "ini", "conf": "ini", "env": "ini",
"sh": "bash", "bash": "bash", "zsh": "bash", "ps1": "powershell",
"bat": "batch", "cmd": "batch", "rb": "ruby", "go": "go", "rs": "rust",
"java": "java", "c": "c", "cpp": "cpp", "h": "c", "hpp": "cpp",
"php": "php", "swift": "swift", "kt": "kotlin", "scala": "scala",
"sql": "sql", "r": "r", "vue": "html", "svelte": "html",
"astro": "html", "properties": "ini", "md": "markdown",
"markdown": "markdown", "txt": "plaintext", "log": "plaintext",
"text": "plaintext",
}
_MARKDOWN_MIMES = {"text/markdown", "text/x-markdown", "application/octet-stream"}
def _data_root() -> Path:
"""Directory that contains ``uploads/`` (mirrors dashboard.py /data)."""
return Path(os.environ.get("FLOWDECK_DATA_DIR", "/data"))
def _file_meta(page: dict) -> dict:
try:
meta = json.loads(page.get("content") or "{}")
return meta if isinstance(meta, dict) else {}
except (json.JSONDecodeError, TypeError):
return {}
def _file_text(page: dict) -> str | None:
"""Return the textual content of an uploaded ``file`` page, or None.
Only reads plain-text / code / markdown files. Binary (PDF, images…)
returns None and is skipped by exporters (nothing meaningful to include).
"""
if (page.get("content_format") or "") != "file":
return None
meta = _file_meta(page)
rel = (meta.get("file_path") or "").replace("\\", "/").strip()
if not rel or ".." in rel.replace("\\", "/").split("/") or not rel.startswith("uploads/"):
return None
name = (rel.rsplit("/", 1)[-1] or "").lower()
ext = name.rsplit(".", 1)[-1] if "." in name else ""
mime = (meta.get("mime_type") or "").lower()
if not (ext in _TEXTUAL_EXTS or mime.startswith("text/")):
return None
try:
full = (_data_root() / rel).resolve()
root = _data_root().resolve()
if root not in full.parents:
return None
return full.read_text(encoding="utf-8", errors="replace")
except (OSError, ValueError):
return None
def _page_source(page: dict):
"""Return (kind, payload) describing where the page's real content lives.
kind ∈ {"blocks", "md", "code"}:
- "blocks": payload is the block list (block editor pages)
- "md" : payload is raw Markdown text
- "code" : payload is (text, language)
An empty/unsupported page yields ("blocks", []).
"""
fmt = (page.get("content_format") or "blocks")
content = page.get("content") or ""
if fmt == "blocks":
return "blocks", _blocks_of(page)
if fmt == "markdown":
if content.strip():
return "md", content
return "blocks", []
if fmt == "file":
text = _file_text(page)
if text is None:
return "blocks", []
meta = _file_meta(page)
name = (meta.get("file_path") or "").replace("\\", "/").rsplit("/", 1)[-1].lower()
ext = name.rsplit(".", 1)[-1] if "." in name else ""
mime = (meta.get("mime_type") or "").lower()
if ext in ("md", "markdown") or mime in _MARKDOWN_MIMES or mime.startswith("text/markdown"):
return "md", text
lang = _CODE_LANG.get(ext, "plaintext")
return "code", (text, lang)
# Unknown format (e.g. legacy) -> try to dump as raw text
if content.strip():
return "md", content
return "blocks", []
# ── Markdown renderer (raw markdown → exportable fragments) ──
def _md_to_blocks(md: str) -> list:
"""Convert raw Markdown text into the same lightweight block list the
editor produces (headings, lists, to-do, quote, code, divider, paragraph).
Kept intentionally simple: inline formatting (bold/links) is preserved as
literal text, matching how the block editor treats imported .md files.
"""
blocks: list = []
buf = md.replace("\r\n", "\n").replace("\r", "\n")
lines = buf.split("\n")
i = 0
n = len(lines)
para: list[str] = []
def flush_para():
nonlocal para
if para:
blocks.append({"type": "paragraph", "content": "\n".join(para).strip()})
para = []
while i < n:
line = lines[i].rstrip()
stripped = line.strip()
if not stripped:
flush_para()
i += 1
continue
if stripped.startswith("```") or stripped.startswith("~~~"):
flush_para()
fence = stripped[0:3]
lang = stripped[3:].strip()
i += 1
code: list[str] = []
while i < n and not lines[i].strip().startswith(fence):
code.append(lines[i])
i += 1
if i < n:
i += 1 # closing fence
blocks.append({"type": "code", "content": "\n".join(code), "language": lang})
continue
m = re.match(r"^(#{1,6})\s+(.*)$", stripped)
if m and line == stripped: # ATX heading must be whole line
level = len(m.group(1))
flush_para()
blocks.append({"type": f"heading_{min(level, 4)}", "content": m.group(2).strip()})
i += 1
continue
if stripped == "---" or stripped == "***" or stripped == "___":
flush_para()
blocks.append({"type": "divider", "content": ""})
i += 1
continue
if re.match(r"^\s*[-*+]\s+", line):
flush_para()
while i < n:
s = lines[i].strip()
m2 = re.match(r"^[-*+]\s+(.*)$", s)
if not m2:
break
blocks.append({"type": "bulleted_list", "content": m2.group(1).strip()})
i += 1
continue
if re.match(r"^\s*\d+[.)]\s+", line):
flush_para()
while i < n:
s = lines[i].strip()
m2 = re.match(r"^\d+[.)]\s+(.*)$", s)
if not m2:
break
blocks.append({"type": "numbered_list", "content": m2.group(1).strip()})
i += 1
continue
mtodo = re.match(r"^\s*[-*+]\s+\[([ xX])\]\s+(.*)$", stripped)
if mtodo:
flush_para()
while i < n:
s = lines[i].strip()
m2 = re.match(r"^[-*+]\s+\[([ xX])\]\s+(.*)$", s)
if not m2:
break
blocks.append({
"type": "to_do",
"content": m2.group(2).strip(),
"checked": m2.group(1).lower() == "x",
})
i += 1
continue
mq = re.match(r"^>\s?(.*)$", stripped)
if mq and line == stripped:
flush_para()
while i < n:
s = lines[i].strip()
m2 = re.match(r"^>\s?(.*)$", s)
if not m2:
break
para.append(m2.group(1))
i += 1
blocks.append({"type": "quote", "content": "\n".join(para)})
para = []
continue
para.append(stripped)
i += 1
flush_para()
return blocks
def _page_blocks(page: dict) -> list:
"""Blocks used for HTML/PDF rendering regardless of storage format."""
kind, payload = _page_source(page)
if kind == "blocks":
return payload
if kind == "code":
text, lang = payload
return [{"type": "code", "content": text, "language": lang}] if text else []
if kind == "md":
return _md_to_blocks(payload)
return []
def _page_markdown_source(page: dict) -> str:
"""Raw markdown when the page IS markdown-sourced, else empty string."""
kind, payload = _page_source(page)
if kind == "md":
return payload
return ""
# ═══════════════ Markdown ═══════════════
def blocks_to_markdown(blocks: list) -> str:
@@ -117,11 +363,11 @@ def _child_pages(page: dict) -> list:
def page_to_markdown(page: dict, *, include_children: bool = True) -> str:
"""Markdown for a single page, with optional sub-pages appended."""
parts = [f"# {_page_title(page)}", ""]
content_inner = _blocks_of(page)
if page.get("content_format") == "markdown" and (page.get("content") or "").strip():
parts.append((page["content"] or "").strip())
md_source = _page_markdown_source(page)
if md_source:
parts.append(md_source.strip())
else:
md = blocks_to_markdown(content_inner)
md = blocks_to_markdown(_page_blocks(page))
if md:
parts.append(md)
md = "\n\n".join(filter(None, parts)).rstrip()
@@ -253,11 +499,7 @@ def page_to_standalone_html(
) -> str:
"""Return a standalone, self-contained HTML document for a page."""
title = _page_title(page)
content_inner = _blocks_of(page)
if page.get("content_format") == "markdown" and (page.get("content") or "").strip():
body = "\n".join(f"<p>{_text(line)}</p>" for line in (page.get("content") or "").splitlines())
else:
body = blocks_to_html(content_inner)
body = blocks_to_html(_page_blocks(page))
meta_updated = page.get("updated_at") or ""
footer = f"<div class='footer'><span>FlowDeck · {_page_title(page)}</span><span>{meta_updated}</span></div>"
@@ -293,11 +535,7 @@ def page_to_standalone_html(
def _pdf_html(page: dict) -> str:
"""A print-friendly, minimal-CSS HTML for PDF conversion."""
title = _page_title(page)
content_inner = _blocks_of(page)
if page.get("content_format") == "markdown" and (page.get("content") or "").strip():
body = "\n".join(f"<p>{_text(line)}</p>" for line in (page.get("content") or "").splitlines())
else:
body = blocks_to_html(content_inner)
body = blocks_to_html(_page_blocks(page))
return f"""<html><head><meta charset="utf-8"><title>{_text(title)}</title>
<style>
body{{font-family:Helvetica,Arial,sans-serif;color:#1f2328;font-size:12px;line-height:1.5;}}
@@ -339,7 +577,6 @@ def _site_index_html(pages: list[dict]) -> str:
"""Build the index.html of the static site (list of all pages)."""
def link(p: dict) -> str:
title = _page_title(p)
slug = quote(f"{title}", safe="")
return f'<li><a href="{quote(title, safe="")}.html">{_text(title)}</a></li>'
items = "".join(link(p) for p in pages)