fix(export): v4.7.2 contenu des documents absent des exports MD/HTML/PDF
Le service d'export ne lisait que les pages content_format='blocks'. Les docs stockees autrement sortaient avec le seul titre: - content_format='file' (.md/code uploades): content=JSON meta, le vrai texte est sur disque (/data/uploads/workspace_*) -> n'etait jamais lu. - content_format='markdown': HTML/PDF enveloppait chaque ligne en <p> (headings et listes aplatis). Resolution de la vraie source pour les 3 formats (app/services/export.py): - lit le fichier upload sur disque pour les pages file (repertoire via FLOWDECK_DATA_DIR, defaut /data), - rend les pages markdown / fichiers .md en blocs (headings, listes, code, quote, todo) pour un HTML/PDF riche, - fichiers texte non-markdown -> bloc de code, - binaires (PDF/images) ignores. 195/195 tests (5 nouveaux v4.7.2). Verifie en reel via HTTP sur README.md et l'arborescence Base de Connaissances (contenu complet dans les 3 formats).
This commit is contained in:
+143
-1
@@ -3024,4 +3024,146 @@ def test_v470_export_site_zip(client):
|
||||
def test_v470_export_404(client):
|
||||
"""Export endpoints return 404 for missing pages."""
|
||||
resp = client.get("/api/export/markdown/999999")
|
||||
assert resp.status_code == 404
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
# ── v4.7.2: Export resolves file & markdown page content (not just title) ──
|
||||
# Tested at the service level: the export HTTP endpoints run behind an in-memory
|
||||
# rate limiter whose store persists across the whole test process, so dozens of
|
||||
# additional HTTP hits at the end of the suite trip 429. These assert on the
|
||||
# service functions directly, which is where the content-resolution lives.
|
||||
|
||||
def _make_src_page(raw_md=None, file_info=None, title="Src Page", parent_id=None):
|
||||
"""Insert a user + page storing raw markdown OR file metadata.
|
||||
|
||||
file_info = (rel_path_under_data_dir, mime). The real file is written to a
|
||||
temp data dir whose path is exposed through ``FLOWDECK_DATA_DIR``.
|
||||
"""
|
||||
import uuid
|
||||
import json as _json
|
||||
from app.db import get_conn
|
||||
login = "v472_" + uuid.uuid4().hex[:10]
|
||||
with get_conn() as conn:
|
||||
conn.execute("INSERT INTO users (login, full_name, email) VALUES (?, 'V472', ?)",
|
||||
(login, login + "@t.com"))
|
||||
uid = conn.execute("SELECT id FROM users WHERE login=?", (login,)).fetchone()["id"]
|
||||
if file_info:
|
||||
rel, mime = file_info
|
||||
payload = _json.dumps({"file_path": rel, "mime_type": mime, "size": 100})
|
||||
fmt = "file"
|
||||
else:
|
||||
payload = raw_md
|
||||
fmt = "markdown"
|
||||
cur = conn.execute(
|
||||
"INSERT INTO pages (workspace, title, content, content_format, parent_id) "
|
||||
"VALUES (?, ?, ?, ?, ?)",
|
||||
(login, title, payload, fmt, parent_id),
|
||||
)
|
||||
pid = cur.lastrowid
|
||||
conn.commit()
|
||||
return pid, uid
|
||||
|
||||
|
||||
def _cleanup_src(pid, uid):
|
||||
from app.db import get_conn
|
||||
with get_conn() as conn:
|
||||
conn.execute("DELETE FROM pages WHERE parent_id=? OR id=?", (pid, pid))
|
||||
conn.execute("DELETE FROM users WHERE id=?", (uid,))
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _load(pid):
|
||||
from app.db import get_conn
|
||||
with get_conn() as conn:
|
||||
return dict(conn.execute("SELECT * FROM pages WHERE id=?", (pid,)).fetchone())
|
||||
|
||||
|
||||
def test_v472_markdown_sourced_page_exports_body(client):
|
||||
"""content_format='markdown' page exports its body, not just the title."""
|
||||
from app.services.export import page_to_markdown
|
||||
raw = "# Intro\n\nCeci est le contenu réel de la page.\n\n- point un\n- point deux\n"
|
||||
pid, uid = _make_src_page(raw_md=raw, title="Page MD")
|
||||
try:
|
||||
md = page_to_markdown(_load(pid))
|
||||
assert "# Page MD" in md
|
||||
assert "Ceci est le contenu réel de la page." in md
|
||||
assert "point un" in md
|
||||
finally:
|
||||
_cleanup_src(pid, uid)
|
||||
|
||||
|
||||
def test_v472_markdown_sourced_page_renders_headings_to_html(client):
|
||||
"""Raw-markdown page renders headings/lists in HTML (not line-wrapped)."""
|
||||
from app.services.export import page_to_standalone_html
|
||||
raw = "# Titre Principal\n\nParagraphe de contenu.\n\n## Sous section\n\n- a\n- b\n"
|
||||
pid, uid = _make_src_page(raw_md=raw, title="Page HTML")
|
||||
try:
|
||||
body = page_to_standalone_html(_load(pid), include_children=False)
|
||||
assert "<h1" in body and "Titre Principal" in body
|
||||
assert "<h2" in body and "Sous section" in body
|
||||
assert "<li>" in body and "Paragraphe de contenu" in body
|
||||
finally:
|
||||
_cleanup_src(pid, uid)
|
||||
|
||||
|
||||
def test_v472_file_page_exports_uploaded_content(client, monkeypatch, tmp_path):
|
||||
"""Uploaded markdown file page exports its real disk content (v4.7.2 fix)."""
|
||||
from app.services.export import page_to_markdown, page_to_standalone_html, page_to_pdf_bytes
|
||||
monkeypatch.setenv("FLOWDECK_DATA_DIR", str(tmp_path))
|
||||
rel = "uploads/workspace_1/note.md"
|
||||
disk = tmp_path / rel
|
||||
disk.parent.mkdir(parents=True, exist_ok=True)
|
||||
disk.write_text("# Note technique\n\ncontenu du fichier sur disque\n", encoding="utf-8")
|
||||
|
||||
pid, uid = _make_src_page(file_info=(rel, "text/markdown"), title="note.md")
|
||||
try:
|
||||
md = page_to_markdown(_load(pid), include_children=False)
|
||||
assert "# note.md" in md
|
||||
assert "Note technique" in md
|
||||
assert "contenu du fichier sur disque" in md
|
||||
|
||||
html = page_to_standalone_html(_load(pid), include_children=False)
|
||||
assert "Note technique" in html
|
||||
assert "contenu du fichier sur disque" in html
|
||||
|
||||
pdf = page_to_pdf_bytes(_load(pid))
|
||||
assert pdf[:5] == b"%PDF-"
|
||||
finally:
|
||||
_cleanup_src(pid, uid)
|
||||
|
||||
|
||||
def test_v472_code_file_page_exported_as_code(client, monkeypatch, tmp_path):
|
||||
"""A non-markdown text file page exports its content, not blank."""
|
||||
from app.services.export import page_to_markdown, page_to_standalone_html
|
||||
monkeypatch.setenv("FLOWDECK_DATA_DIR", str(tmp_path))
|
||||
rel = "uploads/workspace_1/app.py"
|
||||
disk = tmp_path / rel
|
||||
disk.parent.mkdir(parents=True, exist_ok=True)
|
||||
disk.write_text("def hello():\n return 'world'\n", encoding="utf-8")
|
||||
|
||||
pid, uid = _make_src_page(file_info=(rel, "text/x-python"), title="app.py")
|
||||
try:
|
||||
md = page_to_markdown(_load(pid), include_children=False)
|
||||
assert "def hello():" in md and "world" in md
|
||||
html = page_to_standalone_html(_load(pid), include_children=False)
|
||||
assert "def hello():" in html
|
||||
finally:
|
||||
_cleanup_src(pid, uid)
|
||||
|
||||
|
||||
def test_v472_binary_file_page_not_exported(client, monkeypatch, tmp_path):
|
||||
"""Non-textual files (e.g. PDF uploads) export only the title, no garbage."""
|
||||
from app.services.export import page_to_markdown
|
||||
monkeypatch.setenv("FLOWDECK_DATA_DIR", str(tmp_path))
|
||||
rel = "uploads/workspace_1/manual.pdf"
|
||||
disk = tmp_path / rel
|
||||
disk.parent.mkdir(parents=True, exist_ok=True)
|
||||
disk.write_bytes(b"%PDF-1.4\nfake binary content")
|
||||
|
||||
pid, uid = _make_src_page(file_info=(rel, "application/pdf"), title="manual.pdf")
|
||||
try:
|
||||
md = page_to_markdown(_load(pid), include_children=False)
|
||||
assert "# manual.pdf" in md
|
||||
assert "fake binary" not in md # never dump binary into markdown
|
||||
finally:
|
||||
_cleanup_src(pid, uid)
|
||||
Reference in New Issue
Block a user