fix(export): v4.7.2 contenu des documents absent des exports MD/HTML/PDF
FlowDeck CI / test (push) Failing after 19s
FlowDeck CI / docker (push) Skipped

Le service d'export ne lisait que les pages content_format='blocks'. Les docs
stockees autrement sortaient avec le seul titre:
- content_format='file' (.md/code uploades): content=JSON meta, le vrai texte
  est sur disque (/data/uploads/workspace_*) -> n'etait jamais lu.
- content_format='markdown': HTML/PDF enveloppait chaque ligne en <p> (headings
  et listes aplatis).

Resolution de la vraie source pour les 3 formats (app/services/export.py):
- lit le fichier upload sur disque pour les pages file (repertoire via
  FLOWDECK_DATA_DIR, defaut /data),
- rend les pages markdown / fichiers .md en blocs (headings, listes, code,
  quote, todo) pour un HTML/PDF riche,
- fichiers texte non-markdown -> bloc de code,
- binaires (PDF/images) ignores.

195/195 tests (5 nouveaux v4.7.2). Verifie en reel via HTTP sur README.md et
l'arborescence Base de Connaissances (contenu complet dans les 3 formats).
This commit is contained in:
2026-09-03 01:29:35 -04:00
parent 5390a6ceab
commit fa97f07ec8
6 changed files with 417 additions and 21 deletions
+143 -1
View File
@@ -3024,4 +3024,146 @@ def test_v470_export_site_zip(client):
def test_v470_export_404(client):
"""Export endpoints return 404 for missing pages."""
resp = client.get("/api/export/markdown/999999")
assert resp.status_code == 404
assert resp.status_code == 404
# ── v4.7.2: Export resolves file & markdown page content (not just title) ──
# Tested at the service level: the export HTTP endpoints run behind an in-memory
# rate limiter whose store persists across the whole test process, so dozens of
# additional HTTP hits at the end of the suite trip 429. These assert on the
# service functions directly, which is where the content-resolution lives.
def _make_src_page(raw_md=None, file_info=None, title="Src Page", parent_id=None):
"""Insert a user + page storing raw markdown OR file metadata.
file_info = (rel_path_under_data_dir, mime). The real file is written to a
temp data dir whose path is exposed through ``FLOWDECK_DATA_DIR``.
"""
import uuid
import json as _json
from app.db import get_conn
login = "v472_" + uuid.uuid4().hex[:10]
with get_conn() as conn:
conn.execute("INSERT INTO users (login, full_name, email) VALUES (?, 'V472', ?)",
(login, login + "@t.com"))
uid = conn.execute("SELECT id FROM users WHERE login=?", (login,)).fetchone()["id"]
if file_info:
rel, mime = file_info
payload = _json.dumps({"file_path": rel, "mime_type": mime, "size": 100})
fmt = "file"
else:
payload = raw_md
fmt = "markdown"
cur = conn.execute(
"INSERT INTO pages (workspace, title, content, content_format, parent_id) "
"VALUES (?, ?, ?, ?, ?)",
(login, title, payload, fmt, parent_id),
)
pid = cur.lastrowid
conn.commit()
return pid, uid
def _cleanup_src(pid, uid):
from app.db import get_conn
with get_conn() as conn:
conn.execute("DELETE FROM pages WHERE parent_id=? OR id=?", (pid, pid))
conn.execute("DELETE FROM users WHERE id=?", (uid,))
conn.commit()
def _load(pid):
from app.db import get_conn
with get_conn() as conn:
return dict(conn.execute("SELECT * FROM pages WHERE id=?", (pid,)).fetchone())
def test_v472_markdown_sourced_page_exports_body(client):
"""content_format='markdown' page exports its body, not just the title."""
from app.services.export import page_to_markdown
raw = "# Intro\n\nCeci est le contenu réel de la page.\n\n- point un\n- point deux\n"
pid, uid = _make_src_page(raw_md=raw, title="Page MD")
try:
md = page_to_markdown(_load(pid))
assert "# Page MD" in md
assert "Ceci est le contenu réel de la page." in md
assert "point un" in md
finally:
_cleanup_src(pid, uid)
def test_v472_markdown_sourced_page_renders_headings_to_html(client):
"""Raw-markdown page renders headings/lists in HTML (not line-wrapped)."""
from app.services.export import page_to_standalone_html
raw = "# Titre Principal\n\nParagraphe de contenu.\n\n## Sous section\n\n- a\n- b\n"
pid, uid = _make_src_page(raw_md=raw, title="Page HTML")
try:
body = page_to_standalone_html(_load(pid), include_children=False)
assert "<h1" in body and "Titre Principal" in body
assert "<h2" in body and "Sous section" in body
assert "<li>" in body and "Paragraphe de contenu" in body
finally:
_cleanup_src(pid, uid)
def test_v472_file_page_exports_uploaded_content(client, monkeypatch, tmp_path):
"""Uploaded markdown file page exports its real disk content (v4.7.2 fix)."""
from app.services.export import page_to_markdown, page_to_standalone_html, page_to_pdf_bytes
monkeypatch.setenv("FLOWDECK_DATA_DIR", str(tmp_path))
rel = "uploads/workspace_1/note.md"
disk = tmp_path / rel
disk.parent.mkdir(parents=True, exist_ok=True)
disk.write_text("# Note technique\n\ncontenu du fichier sur disque\n", encoding="utf-8")
pid, uid = _make_src_page(file_info=(rel, "text/markdown"), title="note.md")
try:
md = page_to_markdown(_load(pid), include_children=False)
assert "# note.md" in md
assert "Note technique" in md
assert "contenu du fichier sur disque" in md
html = page_to_standalone_html(_load(pid), include_children=False)
assert "Note technique" in html
assert "contenu du fichier sur disque" in html
pdf = page_to_pdf_bytes(_load(pid))
assert pdf[:5] == b"%PDF-"
finally:
_cleanup_src(pid, uid)
def test_v472_code_file_page_exported_as_code(client, monkeypatch, tmp_path):
"""A non-markdown text file page exports its content, not blank."""
from app.services.export import page_to_markdown, page_to_standalone_html
monkeypatch.setenv("FLOWDECK_DATA_DIR", str(tmp_path))
rel = "uploads/workspace_1/app.py"
disk = tmp_path / rel
disk.parent.mkdir(parents=True, exist_ok=True)
disk.write_text("def hello():\n return 'world'\n", encoding="utf-8")
pid, uid = _make_src_page(file_info=(rel, "text/x-python"), title="app.py")
try:
md = page_to_markdown(_load(pid), include_children=False)
assert "def hello():" in md and "world" in md
html = page_to_standalone_html(_load(pid), include_children=False)
assert "def hello():" in html
finally:
_cleanup_src(pid, uid)
def test_v472_binary_file_page_not_exported(client, monkeypatch, tmp_path):
"""Non-textual files (e.g. PDF uploads) export only the title, no garbage."""
from app.services.export import page_to_markdown
monkeypatch.setenv("FLOWDECK_DATA_DIR", str(tmp_path))
rel = "uploads/workspace_1/manual.pdf"
disk = tmp_path / rel
disk.parent.mkdir(parents=True, exist_ok=True)
disk.write_bytes(b"%PDF-1.4\nfake binary content")
pid, uid = _make_src_page(file_info=(rel, "application/pdf"), title="manual.pdf")
try:
md = page_to_markdown(_load(pid), include_children=False)
assert "# manual.pdf" in md
assert "fake binary" not in md # never dump binary into markdown
finally:
_cleanup_src(pid, uid)