- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
878 lines
34 KiB
Python
878 lines
34 KiB
Python
"""FlowDeck — v5.6.0 import framework tests (Phases 0–4).
|
||
|
||
Covers the unified pipeline, hierarchy, attachments, dedup, background jobs and
|
||
the per-source importers (Obsidian, Notion, Logseq/Roam, Google Keep, HTML
|
||
notes, typed CSV/TSV, Excel, JSON, bookmarks, .ics, OPML, Standard Notes, Word,
|
||
PDF, forge issues).
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import io
|
||
import json
|
||
import os
|
||
import secrets
|
||
import time
|
||
import zipfile
|
||
|
||
from app.db import get_conn
|
||
|
||
|
||
def _login(client):
|
||
from app.auth.session import SessionManager
|
||
|
||
with get_conn() as conn:
|
||
login = f"importer_{secrets.token_hex(4)}"
|
||
conn.execute(
|
||
"INSERT INTO users (login, full_name, email, password_hash, is_admin) "
|
||
"VALUES (?, 'Importer', ?, ?, 1)",
|
||
(login, f"{login}@test.com", ""),
|
||
)
|
||
uid = conn.execute("SELECT id FROM users WHERE login=?", (login,)).fetchone()["id"]
|
||
conn.commit()
|
||
session = SessionManager.create_session({"id": uid, "login": login, "is_admin": 1})
|
||
r = client.get("/api/csrf-token", cookies={"flowdeck_session": session})
|
||
return session, r.json()["csrf_token"], login
|
||
|
||
|
||
def _create_workspace(client, cookie, name="Import WS"):
|
||
r = client.post("/workspace", json={"name": name}, cookies={"flowdeck_session": cookie})
|
||
assert r.status_code == 200
|
||
ws = r.json()
|
||
client.cookies.set("flowdeck_workspace", str(ws["id"]))
|
||
return ws["id"]
|
||
|
||
|
||
def _run(client, cookie, csrf, filename, data, *, source=None, extra=None):
|
||
form = dict(extra or {})
|
||
if source:
|
||
form["source"] = source
|
||
return client.post(
|
||
"/api/import/run",
|
||
files={"file": (filename, data, "application/octet-stream")},
|
||
data=form,
|
||
headers={"X-CSRF-Token": csrf},
|
||
cookies={"flowdeck_session": cookie},
|
||
)
|
||
|
||
|
||
def _page_blocks(page_id):
|
||
with get_conn() as conn:
|
||
row = conn.execute("SELECT content, content_format FROM pages WHERE id=?", (page_id,)).fetchone()
|
||
return json.loads(row["content"]) if row and row["content_format"] == "blocks" else []
|
||
|
||
|
||
def _collection_by_name(name):
|
||
with get_conn() as conn:
|
||
return conn.execute("SELECT * FROM collections WHERE name=?", (name,)).fetchone()
|
||
|
||
|
||
def _props_of(collection_id):
|
||
with get_conn() as conn:
|
||
return {
|
||
r["name"]: r["prop_type"]
|
||
for r in conn.execute(
|
||
"SELECT name, prop_type FROM collection_properties WHERE collection_id=?",
|
||
(collection_id,),
|
||
).fetchall()
|
||
}
|
||
|
||
|
||
def _pages_of(collection_id):
|
||
with get_conn() as conn:
|
||
return conn.execute(
|
||
"SELECT * FROM collection_pages WHERE collection_id=? ORDER BY position",
|
||
(collection_id,),
|
||
).fetchall()
|
||
|
||
|
||
def _props_named(collection_id, page_row):
|
||
"""Map a row's id-keyed property values back to property names."""
|
||
with get_conn() as conn:
|
||
names = {
|
||
r["id"]: r["name"]
|
||
for r in conn.execute(
|
||
"SELECT id, name FROM collection_properties WHERE collection_id=?",
|
||
(collection_id,),
|
||
).fetchall()
|
||
}
|
||
values = json.loads(page_row["property_values_json"])
|
||
return {names.get(int(k), k): v for k, v in values.items()}
|
||
|
||
|
||
class TestImportRegistry:
|
||
def test_sources_listed(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = client.get("/api/import/sources", cookies={"flowdeck_session": cookie})
|
||
assert r.status_code == 200
|
||
ids = {s["source_id"] for s in r.json()["sources"]}
|
||
assert {"obsidian", "notion", "logseq", "roam", "google_keep", "html_notes",
|
||
"csv", "excel", "json", "markdown"} <= ids
|
||
|
||
def test_preview_is_dry_run(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = client.post(
|
||
"/api/import/preview",
|
||
files={"file": ("notes.csv", b"Name,Age\nAlice,30", "text/csv")},
|
||
headers={"X-CSRF-Token": csrf},
|
||
cookies={"flowdeck_session": cookie},
|
||
)
|
||
assert r.status_code == 200
|
||
body = r.json()
|
||
assert body["dry_run"] is True
|
||
assert body["detected_source"] == "csv"
|
||
assert body["pages"][0]["type"] == "collection"
|
||
assert body["pages"][0]["rows"] == 1
|
||
assert _collection_by_name("notes") is None
|
||
|
||
|
||
class TestCsvTypedImport:
|
||
def test_types_inferred_and_rows_created(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
csv_data = (
|
||
b"Name,Age,Due,Active,Email,Website,Tags\n"
|
||
b"Alice,30,2026-01-02,true,[email protected],https://x.com,red;blue\n"
|
||
b"Bob,42,2026-02-03,false,[email protected],https://y.com,green\n"
|
||
)
|
||
r = _run(client, cookie, csrf, "people.csv", csv_data)
|
||
assert r.status_code == 200, r.text
|
||
report = r.json()
|
||
assert report["collections_created"] == 1
|
||
assert report["rows_created"] == 2
|
||
|
||
coll = _collection_by_name("people")
|
||
assert coll is not None
|
||
types = _props_of(coll["id"])
|
||
assert types["Age"] == "number"
|
||
assert types["Due"] == "date"
|
||
assert types["Active"] == "checkbox"
|
||
assert types["Email"] == "email"
|
||
assert types["Website"] == "url"
|
||
assert types["Tags"] == "multi_select"
|
||
|
||
pages = _pages_of(coll["id"])
|
||
assert pages[0]["title"] == "Alice"
|
||
props = _props_named(coll["id"], pages[0])
|
||
assert props["Age"] == 30
|
||
assert props["Active"] is True
|
||
assert props["Tags"] == ["red", "blue"]
|
||
|
||
def test_dedup_skips_second_import(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
data = b"Name,Age\nAlice,30"
|
||
assert _run(client, cookie, csrf, "d.csv", data).json()["rows_created"] == 1
|
||
second = _run(client, cookie, csrf, "d.csv", data).json()
|
||
assert second["rows_created"] == 0
|
||
assert second["skipped"] == 1
|
||
|
||
def test_tsv_delimiter(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
data = b"Name\tScore\nAlice\t9\nBob\t8"
|
||
r = _run(client, cookie, csrf, "scores.tsv", data)
|
||
assert r.status_code == 200
|
||
coll = _collection_by_name("scores")
|
||
assert _props_of(coll["id"])["Score"] == "number"
|
||
assert len(_pages_of(coll["id"])) == 2
|
||
|
||
|
||
class TestExcelJsonImport:
|
||
def test_excel_multi_sheet(self, client):
|
||
from openpyxl import Workbook
|
||
|
||
wb = Workbook()
|
||
ws1 = wb.active
|
||
ws1.title = "Tasks"
|
||
ws1.append(["Task", "Priority"])
|
||
ws1.append(["Write", "High"])
|
||
ws2 = wb.create_sheet("Contacts")
|
||
ws2.append(["Name", "Email"])
|
||
ws2.append(["Alice", "[email protected]"])
|
||
buf = io.BytesIO()
|
||
wb.save(buf)
|
||
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = _run(client, cookie, csrf, "book.xlsx", buf.getvalue())
|
||
assert r.status_code == 200, r.text
|
||
assert r.json()["collections_created"] == 2
|
||
assert _collection_by_name("book — Tasks") is not None
|
||
assert _collection_by_name("book — Contacts") is not None
|
||
|
||
def test_generic_json(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
data = json.dumps([
|
||
{"title": "One", "count": 3},
|
||
{"title": "Two", "count": 5},
|
||
]).encode()
|
||
r = _run(client, cookie, csrf, "items.json", data)
|
||
assert r.status_code == 200
|
||
coll = _collection_by_name("items")
|
||
assert _props_of(coll["id"])["count"] == "number"
|
||
assert len(_pages_of(coll["id"])) == 2
|
||
|
||
|
||
class TestObsidianImport:
|
||
def _vault(self) -> bytes:
|
||
buf = io.BytesIO()
|
||
with zipfile.ZipFile(buf, "w") as zf:
|
||
zf.writestr(".obsidian/app.json", "{}")
|
||
zf.writestr(
|
||
"Projects/Roadmap.md",
|
||
"---\ntitle: Feuille de route\ntags: [projet, 2026]\n---\n"
|
||
"# Roadmap\n\nVoir [[Ideas]] et l'image ![[diagram.png]]\n",
|
||
)
|
||
zf.writestr("Ideas.md", "Quelques idées\n")
|
||
zf.writestr("Projects/diagram.png", b"\x89PNG\r\n\x1a\nfake")
|
||
return buf.getvalue()
|
||
|
||
def test_hierarchy_frontmatter_and_attachment(self, client, tmp_path):
|
||
cookie, csrf, _ = _login(client)
|
||
ws_id = _create_workspace(client, cookie)
|
||
old = os.environ.get("FLOWDECK_DATA_DIR")
|
||
os.environ["FLOWDECK_DATA_DIR"] = str(tmp_path)
|
||
try:
|
||
r = _run(client, cookie, csrf, "vault.zip", self._vault())
|
||
assert r.status_code == 200, r.text
|
||
report = r.json()
|
||
assert report["detected_source"] == "obsidian"
|
||
assert report["attachments"] == 1
|
||
|
||
with get_conn() as conn:
|
||
folder = conn.execute("SELECT * FROM pages WHERE title='Projects'").fetchone()
|
||
note = conn.execute("SELECT * FROM pages WHERE title='Feuille de route'").fetchone()
|
||
assert folder is not None and note is not None
|
||
assert note["parent_id"] == folder["id"]
|
||
|
||
blocks = _page_blocks(note["id"])
|
||
assert any(b["type"] == "heading_1" and b["content"] == "Roadmap" for b in blocks)
|
||
images = [b for b in blocks if b["type"] == "image"]
|
||
assert images and f"/api/files/{ws_id}/import/" in images[0]["src"]
|
||
assert any(b["type"] == "callout" for b in blocks)
|
||
finally:
|
||
if old is None:
|
||
os.environ.pop("FLOWDECK_DATA_DIR", None)
|
||
else:
|
||
os.environ["FLOWDECK_DATA_DIR"] = old
|
||
|
||
def test_existing_md_import_still_works(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = client.post(
|
||
"/board/api/pages/import",
|
||
json={"markdown": "# Hello\n\n- a\n- b", "title": "Legacy"},
|
||
headers={"X-CSRF-Token": csrf, "Content-Type": "application/json"},
|
||
cookies={"flowdeck_session": cookie},
|
||
)
|
||
assert r.status_code == 200
|
||
|
||
|
||
class TestNotionImport:
|
||
def test_pages_and_database(self, client):
|
||
buf = io.BytesIO()
|
||
with zipfile.ZipFile(buf, "w") as zf:
|
||
zf.writestr("Home abcdef0123456789abcdef0123456789.md",
|
||
"# Home\n\nBienvenue [Child](Child%20abcdef0123456789abcdef0123456789.md)\n")
|
||
zf.writestr("Home/Child abcdef0123456789abcdef0123456789.md", "# Child\n\ntexte\n")
|
||
zf.writestr("Tasks 11111111111111111111111111111111.csv",
|
||
"Name,Status,Done\nT1,Done,true\nT2,Todo,false\n")
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = _run(client, cookie, csrf, "notion.zip", buf.getvalue())
|
||
assert r.status_code == 200, r.text
|
||
report = r.json()
|
||
assert report["detected_source"] == "notion"
|
||
assert report["collections_created"] == 1
|
||
coll = _collection_by_name("Tasks")
|
||
assert coll is not None
|
||
assert _props_of(coll["id"])["Done"] == "checkbox"
|
||
assert len(_pages_of(coll["id"])) == 2
|
||
with get_conn() as conn:
|
||
home = conn.execute("SELECT * FROM pages WHERE title='Home'").fetchone()
|
||
text = json.dumps(_page_blocks(home["id"]))
|
||
assert "abcdef0123456789" not in text
|
||
assert "Child" in text
|
||
|
||
|
||
class TestOutlineImport:
|
||
def test_logseq_properties_and_journal(self, client):
|
||
data = (
|
||
"title:: Ma page\n"
|
||
"tags:: travail, urgent\n"
|
||
"- Première puce\n"
|
||
"- Deuxième [[Lien]]\n"
|
||
).encode()
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = _run(client, cookie, csrf, "logseq-page.md", data)
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "logseq"
|
||
with get_conn() as conn:
|
||
page = conn.execute("SELECT * FROM pages WHERE title='Ma page'").fetchone()
|
||
assert page is not None
|
||
blocks = _page_blocks(page["id"])
|
||
assert any(b["type"] == "bulleted_list" for b in blocks)
|
||
|
||
def test_roam_todo_marker(self, client):
|
||
data = b"- {{[[TODO]]}} finir\n- {{[[DONE]]}} fait\n"
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = _run(client, cookie, csrf, "roam.md", data)
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "roam"
|
||
with get_conn() as conn:
|
||
page = conn.execute("SELECT * FROM pages WHERE title='roam'").fetchone()
|
||
todos = [b for b in _page_blocks(page["id"]) if b["type"] == "to_do"]
|
||
assert todos and todos[0]["checked"] is False
|
||
|
||
|
||
class TestHtmlAndKeepImport:
|
||
def test_html_note(self, client):
|
||
html = b"<html><head><title>Ma note</title></head><body><h1>Ma note</h1><p>Texte</p><ul><li>un</li><li>deux</li></ul></body></html>"
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = _run(client, cookie, csrf, "note.html", html)
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "html_notes"
|
||
with get_conn() as conn:
|
||
page = conn.execute("SELECT * FROM pages WHERE title='Ma note'").fetchone()
|
||
blocks = _page_blocks(page["id"])
|
||
assert any(b["type"] == "heading_1" for b in blocks)
|
||
assert sum(1 for b in blocks if b["type"] == "bulleted_list") == 2
|
||
|
||
def test_google_keep_json(self, client):
|
||
note = {
|
||
"title": "Liste courses",
|
||
"listContent": [
|
||
{"text": "Pain", "isChecked": False},
|
||
{"text": "Lait", "isChecked": True},
|
||
],
|
||
"labels": ["perso"],
|
||
"isTrashed": False,
|
||
"color": "white",
|
||
}
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = _run(client, cookie, csrf, "keep.json", json.dumps(note).encode())
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "google_keep"
|
||
with get_conn() as conn:
|
||
page = conn.execute("SELECT * FROM pages WHERE title='Liste courses'").fetchone()
|
||
todos = [b for b in _page_blocks(page["id"]) if b["type"] == "to_do"]
|
||
assert len(todos) == 2
|
||
|
||
|
||
class TestMappingAndWizard:
|
||
def test_column_type_mapping_override(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
data = b"Name,Code\nA,007\nB,042\n"
|
||
r = _run(
|
||
client, cookie, csrf, "codes.csv", data,
|
||
extra={"mapping": json.dumps({"Code": "text"})},
|
||
)
|
||
assert r.status_code == 200
|
||
coll = _collection_by_name("codes")
|
||
assert _props_of(coll["id"])["Code"] == "text"
|
||
props = _props_named(coll["id"], _pages_of(coll["id"])[0])
|
||
assert props["Code"] == "007"
|
||
|
||
def test_wizard_page_requires_auth(self, client):
|
||
r = client.get("/import", follow_redirects=False)
|
||
assert r.status_code in (302, 307)
|
||
|
||
def test_wizard_page_renders(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = client.get("/import", cookies={"flowdeck_session": cookie})
|
||
assert r.status_code == 200
|
||
assert "Importer des données" in r.text
|
||
|
||
|
||
class TestAsyncJob:
|
||
def test_background_import(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = client.post(
|
||
"/api/import/run",
|
||
files={"file": ("async.csv", b"Name,Age\nAlice,30", "text/csv")},
|
||
data={"async": "true"},
|
||
headers={"X-CSRF-Token": csrf},
|
||
cookies={"flowdeck_session": cookie},
|
||
)
|
||
assert r.status_code == 200
|
||
job_id = r.json()["job_id"]
|
||
status = None
|
||
for _ in range(50):
|
||
jr = client.get(f"/api/import/jobs/{job_id}", cookies={"flowdeck_session": cookie})
|
||
assert jr.status_code == 200
|
||
status = jr.json()["status"]
|
||
if status in ("done", "error"):
|
||
break
|
||
time.sleep(0.1)
|
||
assert status == "done", jr.json()
|
||
assert jr.json()["report"]["rows_created"] == 1
|
||
|
||
|
||
# ── Phase 4: bookmarks, calendar, OPML, Standard Notes, forge ────────────
|
||
|
||
|
||
class TestBookmarkImport:
|
||
def test_raindrop_csv(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
csv_data = (
|
||
b"id,title,note,excerpt,url,folder,tags,created\n"
|
||
b"1,Notion,super outil,,https://notion.so,Tools,\"prod,web\",2026-01-02T10:00:00Z\n"
|
||
)
|
||
r = _run(client, cookie, csrf, "raindrop.csv", csv_data)
|
||
assert r.status_code == 200, r.text
|
||
assert r.json()["detected_source"] == "raindrop"
|
||
coll = _collection_by_name("Raindrop")
|
||
assert coll is not None
|
||
row = _pages_of(coll["id"])[0]
|
||
props = _props_named(coll["id"], row)
|
||
assert props["URL"] == "https://notion.so"
|
||
assert props["Tags"] == ["prod", "web"]
|
||
|
||
def test_pocket_csv(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
csv_data = b"title,url,time_added,tags,status\nArticle,https://ex.com/1,1700000000,news,unread\n"
|
||
r = _run(client, cookie, csrf, "pocket.csv", csv_data)
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "pocket"
|
||
coll = _collection_by_name("Pocket")
|
||
assert _pages_of(coll["id"])[0]["title"] == "Article"
|
||
|
||
def test_readwise_csv(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
csv_data = (
|
||
"Highlight,Book Title,Book Author,Note,Document Tags,Highlighted at\n"
|
||
"Une idée clé,Livre X,Auteur Y,à retenir,philo,2026-03-01\n"
|
||
).encode()
|
||
r = _run(client, cookie, csrf, "readwise.csv", csv_data)
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "readwise"
|
||
coll = _collection_by_name("Readwise")
|
||
props = _props_named(coll["id"], _pages_of(coll["id"])[0])
|
||
assert props["Book"] == "Livre X"
|
||
assert props["Tags"] == ["philo"]
|
||
|
||
def test_shaarli_json(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
data = json.dumps([
|
||
{"url": "https://a.io", "title": "A", "tags": "dev", "created": "2026-01-01"},
|
||
]).encode()
|
||
r = _run(client, cookie, csrf, "shaarli.json", data)
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "shaarli"
|
||
coll = _collection_by_name("Shaarli")
|
||
assert _pages_of(coll["id"])[0]["title"] == "A"
|
||
|
||
def test_netscape_bookmarks(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
html = (
|
||
b'<!DOCTYPE NETSCAPE-Bookmark-file-1><DL><p>'
|
||
b'<DT><H3>Dev</H3><DL><p>'
|
||
b'<DT><A HREF="https://python.org" ADD_DATE="1700000000">Python</A>'
|
||
b'</DL><p></DL><p>'
|
||
)
|
||
r = _run(client, cookie, csrf, "bookmarks.html", html)
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "bookmarks"
|
||
coll = _collection_by_name("Bookmarks")
|
||
props = _props_named(coll["id"], _pages_of(coll["id"])[0])
|
||
assert props["URL"] == "https://python.org"
|
||
assert props["Tags"] == ["Dev"]
|
||
|
||
|
||
class TestCalendarOpmlNotes:
|
||
def test_ics_events(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
ics = (
|
||
"BEGIN:VCALENDAR\r\nX-WR-CALNAME:Work\r\nBEGIN:VEVENT\r\n"
|
||
"SUMMARY:Réunion\r\nDTSTART;VALUE=DATE:20260115\r\n"
|
||
"LOCATION:Bureau\r\nEND:VEVENT\r\n"
|
||
"BEGIN:VEVENT\r\nSUMMARY:Call\r\nDTSTART:20260116T140000Z\r\n"
|
||
"DTEND:20260116T150000Z\r\nEND:VEVENT\r\nEND:VCALENDAR\r\n"
|
||
).encode()
|
||
r = _run(client, cookie, csrf, "agenda.ics", ics)
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "ics"
|
||
coll = _collection_by_name("Calendar")
|
||
pages = _pages_of(coll["id"])
|
||
assert len(pages) == 2
|
||
props = _props_named(coll["id"], pages[0])
|
||
assert props["Start"] == "2026-01-15"
|
||
assert props["All day"] is True
|
||
assert props["Calendar"] == "Work"
|
||
|
||
def test_opml_feeds(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
opml = (
|
||
b'<?xml version="1.0"?><opml version="2.0"><body>'
|
||
b'<outline text="Tech"><outline text="HN" xmlUrl="https://hn/rss" htmlUrl="https://hn"/></outline>'
|
||
b'</body></opml>'
|
||
)
|
||
r = _run(client, cookie, csrf, "feeds.opml", opml)
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "opml"
|
||
coll = _collection_by_name("OPML feeds")
|
||
rows = [_props_named(coll["id"], p) for p in _pages_of(coll["id"])]
|
||
feed = next(p for p in rows if p.get("Feed URL"))
|
||
assert feed["Feed URL"] == "https://hn/rss"
|
||
assert feed["Folder"] == "Tech"
|
||
|
||
def test_standard_notes(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
backup = {
|
||
"items": [
|
||
{"uuid": "u1", "content_type": "Note",
|
||
"content": json.dumps({"title": "Ma note", "text": "Contenu clair"})},
|
||
{"uuid": "u2", "content_type": "Note",
|
||
"content": json.dumps({"encrypted": True, "content": "xxx"})},
|
||
{"uuid": "u3", "content_type": "Tag", "content": "ignore"},
|
||
]
|
||
}
|
||
r = _run(client, cookie, csrf, "standard-notes.json", json.dumps(backup).encode())
|
||
assert r.status_code == 200
|
||
assert r.json()["detected_source"] == "standard_notes"
|
||
with get_conn() as conn:
|
||
page = conn.execute("SELECT * FROM pages WHERE title='Ma note'").fetchone()
|
||
assert page is not None
|
||
assert r.json()["warnings"]
|
||
|
||
|
||
class TestDocumentImport:
|
||
def test_docx(self, client):
|
||
from docx import Document
|
||
from docx.shared import Inches
|
||
from PIL import Image
|
||
|
||
doc = Document()
|
||
doc.add_heading("Titre Word", level=1)
|
||
doc.add_paragraph("Un paragraphe.")
|
||
doc.add_paragraph("Puce un", style="List Bullet")
|
||
table = doc.add_table(rows=2, cols=2)
|
||
table.cell(0, 0).text = "A"
|
||
table.cell(0, 1).text = "B"
|
||
table.cell(1, 0).text = "1"
|
||
table.cell(1, 1).text = "2"
|
||
img = io.BytesIO()
|
||
Image.new("RGB", (10, 10), "red").save(img, format="PNG")
|
||
img.seek(0)
|
||
doc.add_picture(img, width=Inches(1))
|
||
buf = io.BytesIO()
|
||
doc.save(buf)
|
||
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = _run(client, cookie, csrf, "doc.docx", buf.getvalue())
|
||
assert r.status_code == 200, r.text
|
||
assert r.json()["detected_source"] == "docx"
|
||
assert r.json()["attachments"] == 1
|
||
with get_conn() as conn:
|
||
page = conn.execute("SELECT * FROM pages WHERE title='doc'").fetchone()
|
||
blocks = _page_blocks(page["id"])
|
||
assert any(b["type"] == "heading_1" and b["content"] == "Titre Word" for b in blocks)
|
||
assert any(b["type"] == "table" for b in blocks)
|
||
assert any(b["type"] == "image" for b in blocks)
|
||
|
||
def test_pdf_text(self, client):
|
||
from xhtml2pdf import pisa
|
||
|
||
buf = io.BytesIO()
|
||
pisa.CreatePDF("<h1>Hello PDF</h1><p>Bonjour le monde</p>", dest=buf)
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = _run(client, cookie, csrf, "report.pdf", buf.getvalue())
|
||
assert r.status_code == 200, r.text
|
||
assert r.json()["detected_source"] == "pdf"
|
||
with get_conn() as conn:
|
||
page = conn.execute("SELECT * FROM pages WHERE title='report'").fetchone()
|
||
text = json.dumps(_page_blocks(page["id"]))
|
||
assert "Bonjour le monde" in text
|
||
|
||
|
||
class TestForgeImport:
|
||
def test_build_issues_result(self):
|
||
from app.services.importers.forge import build_issues_result
|
||
|
||
issues = [
|
||
{"number": 1, "title": "Bug", "state": "open", "labels": [{"name": "bug"}],
|
||
"milestone": {"title": "v1"}, "assignees": [{"login": "bob"}],
|
||
"created_at": "2026-01-01T00:00:00Z", "html_url": "https://x/1", "body": "desc"},
|
||
{"number": 2, "title": "PR", "pull_request": {"url": "x"}},
|
||
]
|
||
labels = [{"name": "bug", "color": "red"}]
|
||
milestones = [{"title": "v1", "state": "open", "due_on": "2026-02-01"}]
|
||
result = build_issues_result(issues, labels=labels, milestones=milestones,
|
||
owner="me", repo="proj", provider="gitea")
|
||
assert result.source == "forge:gitea"
|
||
assert len(result.pages) == 3
|
||
issue_page = result.pages[0]
|
||
assert issue_page.collection["name"] == "me/proj issues"
|
||
assert len(issue_page.collection["rows"]) == 1
|
||
props = issue_page.collection["rows"][0]["properties"]
|
||
assert props["Labels"] == ["bug"]
|
||
assert props["Milestone"] == "v1"
|
||
assert props["Assignee"] == "bob"
|
||
|
||
def test_fetch_forge_issues(self):
|
||
import asyncio
|
||
|
||
from app.services.importers.forge import fetch_forge_issues
|
||
|
||
class FakeAdapter:
|
||
async def list_issues(self, owner, repo, state="all"):
|
||
return [{"number": 1, "title": "T", "state": "closed", "labels": []}]
|
||
|
||
async def list_labels(self, owner, repo):
|
||
return [{"name": "x"}]
|
||
|
||
async def list_milestones(self, owner, repo, state="all"):
|
||
return [{"title": "M", "state": "closed"}]
|
||
|
||
result = asyncio.run(fetch_forge_issues(
|
||
FakeAdapter(), "o", "r", provider="github",
|
||
))
|
||
assert len(result.pages) == 3
|
||
|
||
def test_gitea_adapter_wraps_client(self):
|
||
import asyncio
|
||
|
||
from app.services.importers.forge import GiteaForgeAdapter
|
||
|
||
class FakeGiteaClient:
|
||
async def get_issues(self, owner, repo, state="all", page=1, limit=50):
|
||
return [{"number": 1, "title": "G", "state": "open"}] if page == 1 else []
|
||
|
||
async def get_labels(self, owner, repo):
|
||
return [{"name": "bug"}]
|
||
|
||
async def get_milestones(self, owner, repo, state="open"):
|
||
return [{"title": "M1", "state": "open"}]
|
||
|
||
async def run():
|
||
adapter = GiteaForgeAdapter(FakeGiteaClient())
|
||
issues = await adapter.list_issues("o", "r")
|
||
labels = await adapter.list_labels("o", "r")
|
||
milestones = await adapter.list_milestones("o", "r")
|
||
return issues, labels, milestones
|
||
|
||
issues, labels, milestones = asyncio.run(run())
|
||
assert len(issues) == 1 and labels and milestones
|
||
|
||
def test_forge_endpoint_requires_connection(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = client.post(
|
||
"/api/import/forge",
|
||
json={"provider": "gitea", "owner": "o", "repo": "r"},
|
||
headers={"X-CSRF-Token": csrf},
|
||
cookies={"flowdeck_session": cookie},
|
||
)
|
||
assert r.status_code == 400
|
||
|
||
|
||
# ── Phase 5: incremental, batch, URL, forge-repo, relations, reports ──────
|
||
|
||
|
||
class TestIncrementalImport:
|
||
def test_mode_update_resyncs_rows(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
assert _run(client, cookie, csrf, "m.csv", b"Name,Age\nAlice,30").json()["rows_created"] == 1
|
||
second = _run(client, cookie, csrf, "m.csv", b"Name,Age\nAlice,31\nBob,20",
|
||
extra={"mode": "update"}).json()
|
||
assert second["rows_updated"] == 1
|
||
assert second["rows_created"] == 1
|
||
coll = _collection_by_name("m")
|
||
pages = {p["title"]: _props_named(coll["id"], p) for p in _pages_of(coll["id"])}
|
||
assert pages["Alice"]["Age"] == 31
|
||
|
||
def test_mode_duplicate_creates_again(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
_run(client, cookie, csrf, "d.csv", b"Name\nA")
|
||
second = _run(client, cookie, csrf, "d.csv", b"Name\nA",
|
||
extra={"mode": "duplicate"}).json()
|
||
assert second["collections_created"] == 1
|
||
assert second["skipped"] == 0
|
||
|
||
def test_mode_skip_is_default(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
_run(client, cookie, csrf, "s.csv", b"Name\nA")
|
||
second = _run(client, cookie, csrf, "s.csv", b"Name\nA").json()
|
||
assert second["skipped"] == 1
|
||
|
||
def test_report_shape(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
report = _run(client, cookie, csrf, "r.csv", b"Name\nA").json()
|
||
assert report["status"] == "ok"
|
||
assert report["errors"] == []
|
||
assert report["mode"] == "skip"
|
||
|
||
|
||
class TestBatchImport:
|
||
def test_run_batch_multiple_files(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = client.post(
|
||
"/api/import/run-batch",
|
||
files=[
|
||
("file", ("a.csv", b"Name,Age\nA,1", "text/csv")),
|
||
("file", ("b.csv", b"Name,Age\nB,2", "text/csv")),
|
||
],
|
||
headers={"X-CSRF-Token": csrf},
|
||
cookies={"flowdeck_session": cookie},
|
||
)
|
||
assert r.status_code == 200, r.text
|
||
body = r.json()
|
||
assert body["summary"]["files"] == 2
|
||
assert len(body["results"]) == 2
|
||
assert body["summary"]["rows_created"] == 2
|
||
|
||
|
||
class TestUrlImport:
|
||
def test_fetch_url_result(self, monkeypatch):
|
||
import asyncio
|
||
|
||
import httpx
|
||
|
||
from app.services.importers import url_fetch
|
||
|
||
monkeypatch.setattr(url_fetch, "_is_public_host", lambda host: True)
|
||
html = (
|
||
"<html><head><title>Titre</title>"
|
||
"<meta property='og:title' content='Titre OG'>"
|
||
"<meta property='og:description' content='Desc'></head>"
|
||
"<body><h1>Salut</h1><p>Contenu</p></body></html>"
|
||
)
|
||
|
||
def handler(request):
|
||
return httpx.Response(200, headers={"content-type": "text/html"},
|
||
text=html, request=request)
|
||
|
||
result = asyncio.run(url_fetch.fetch_url_result(
|
||
"https://example.com/post", transport=httpx.MockTransport(handler),
|
||
))
|
||
assert result.source == "url"
|
||
assert len(result.pages) == 1
|
||
blocks = result.pages[0].blocks
|
||
assert blocks[0]["type"] == "bookmark"
|
||
assert blocks[0]["url"] == "https://example.com/post"
|
||
|
||
def test_url_endpoint_rejects_private_host(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = client.post(
|
||
"/api/import/url",
|
||
json={"url": "http://localhost/secret"},
|
||
headers={"X-CSRF-Token": csrf},
|
||
cookies={"flowdeck_session": cookie},
|
||
)
|
||
assert r.status_code == 400
|
||
|
||
|
||
class TestForgeRepoImport:
|
||
def test_build_repo_result(self):
|
||
from app.services.importers.forge_repo import build_repo_result
|
||
|
||
result = build_repo_result(
|
||
[("docs/README.md", "# Hi\n\ntext"), ("src/app.py", "print('x')")],
|
||
owner="o", repo="r", provider="github",
|
||
)
|
||
assert len(result.pages) == 2
|
||
md = next(p for p in result.pages if p.title == "README.md")
|
||
assert md.parent_path == "docs"
|
||
assert md.markdown == "# Hi\n\ntext"
|
||
py = next(p for p in result.pages if p.title == "app.py")
|
||
assert "```python" in py.markdown
|
||
|
||
def test_fetch_forge_repo(self):
|
||
import asyncio
|
||
|
||
from app.services.importers.forge_repo import fetch_forge_repo
|
||
|
||
class FakeAdapter:
|
||
async def list_repo_files(self, owner, repo, path=""):
|
||
return [{"path": "a.md", "size": 10},
|
||
{"path": "big.bin", "size": 9999999},
|
||
{"path": "img.png", "size": 10}]
|
||
|
||
async def get_file_content(self, owner, repo, path):
|
||
return "# A"
|
||
|
||
result = asyncio.run(fetch_forge_repo(
|
||
FakeAdapter(), "o", "r", provider="github",
|
||
))
|
||
assert len(result.pages) == 1
|
||
assert result.pages[0].title == "a.md"
|
||
|
||
|
||
class TestRelationsResolve:
|
||
def test_text_column_becomes_relation(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
_run(client, cookie, csrf, "members.csv", b"Name,Team\nAlice,Red\nBob,Blue")
|
||
_run(client, cookie, csrf, "teams.csv", b"Name\nRed\nBlue")
|
||
r = client.post(
|
||
"/api/import/relations/resolve",
|
||
headers={"X-CSRF-Token": csrf},
|
||
cookies={"flowdeck_session": cookie},
|
||
)
|
||
assert r.status_code == 200
|
||
assert r.json()["relations_resolved"] == 1
|
||
|
||
members = _collection_by_name("members")
|
||
with get_conn() as conn:
|
||
prop = conn.execute(
|
||
"SELECT id, prop_type, related_collection_id FROM collection_properties "
|
||
"WHERE collection_id=? AND name='Team'", (members["id"],),
|
||
).fetchone()
|
||
assert prop["prop_type"] == "relation"
|
||
assert prop["related_collection_id"] is not None
|
||
|
||
teams = _collection_by_name("teams")
|
||
team_ids = {p["title"]: p["id"] for p in _pages_of(teams["id"])}
|
||
row = _pages_of(members["id"])[0]
|
||
values = json.loads(row["property_values_json"])
|
||
assert values[str(prop["id"])] == [team_ids["Red"]]
|
||
|
||
|
||
class TestJobReport:
|
||
def test_download_job_report(self, client):
|
||
cookie, csrf, _ = _login(client)
|
||
_create_workspace(client, cookie)
|
||
r = client.post(
|
||
"/api/import/run",
|
||
files={"file": ("j.csv", b"Name,Age\nA,1", "text/csv")},
|
||
data={"async": "true"},
|
||
headers={"X-CSRF-Token": csrf},
|
||
cookies={"flowdeck_session": cookie},
|
||
)
|
||
job_id = r.json()["job_id"]
|
||
for _ in range(50):
|
||
jr = client.get(f"/api/import/jobs/{job_id}", cookies={"flowdeck_session": cookie})
|
||
if jr.json()["status"] in ("done", "error"):
|
||
break
|
||
time.sleep(0.1)
|
||
rep = client.get(f"/api/import/jobs/{job_id}/report", cookies={"flowdeck_session": cookie})
|
||
assert rep.status_code == 200
|
||
assert "attachment" in rep.headers.get("content-disposition", "")
|
||
assert rep.json()["rows_created"] == 1
|