Files
flowdeck/app/services/importers/pipeline.py
T
bruno 7be96f0618
FlowDeck CI / lint (push) Canceled after 0s
FlowDeck CI / test (push) Canceled after 0s
FlowDeck CI / docker (push) Canceled after 0s
fix: A29 + A42(partiel) — publish partagé, fuite password_hash, data_dir (v7.5.0)
- A29 — `app/services/publish.py` : slugify titré unique (fallback aléatoire),
  404 si la page n'existe pas, événements centralisés. Les 3 paires
  publish/unpublish déléguent (sharing = front, board, v2) :
  · board : mise à jour aveugle → 404 + contrôle de session ajouté
  · board : perd `share_mode='anyone'` en bonus, v2 : perd `is_shared=1` —
    le share dialog reste l'unique propriétaire de ces drapeaux
  · v2 : slug fourni conservé, slug vidé aussi à la dépublication (avant : laissé)
  · `/users/me` ×2 et listings collections ×3 = contrats versionnés distincts,
    décision documentée (on garde)
- Byproduct sécurité — `GET /api/users/me` (v1) et le contexte de `/accounts`
  faisaient `SELECT *` sur users → password_hash / login_attempts / locked_until
  exposés → colonnes whitelistées (liste v2)
- A42 (partiel) — 9 copies de `Path(os.environ.get("FLOWDECK_DATA_DIR", "/data"))`
  → `settings.data_dir` (property : lecture à chaque accès, les tests
  monkeypatchent l'env) ; cache Gitea : évacuation des entrées expirées à chaque
  écriture. Reste : client httpx partagé (52 créations, cache par event loop)

tests : test_publish_service_shared_and_safe, test_users_me_no_secret_columns,
test_gitea_cache_evicts_expired

suite **1034/1034** · `ruff check app tests` OK · OpenAPI 511 chemins / 7.5.0
docs (ROADMAP/CHANGELOG/WORKLOAD/VERSION) à jour
2026-10-01 10:01:39 -04:00

567 lines
22 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""FlowDeck — common import pipeline (v5.6.0, Phase 0).
Persists an :class:`~app.services.importers.base.ImportResult` into FlowDeck:
resolves the workspace, rebuilds the folder hierarchy (``parent_id``), stores
attachments, rewrites links, creates collections + rows, and records imported
items for idempotent re-imports.
"""
from __future__ import annotations
import hashlib
import json
import logging
import re
from pathlib import Path
from typing import Any
from app.config import settings
from app.db import get_conn
from app.services.db_templates import materialize_properties
from app.services.export import markdown_to_blocks
from app.services.importers.base import ImportPage, ImportResult
from app.services.importers.tabular import apply_type_mapping
logger = logging.getLogger(__name__)
_IMG_RE = re.compile(r"!\[([^\]]*)\]\(([^)\s]+)(?:\s+\"[^\"]*\")?\)")
def _data_dir() -> Path:
return Path(settings.data_dir)
def _safe_filename(name: str) -> str:
base = Path(name.replace("\\", "/")).name
base = re.sub(r"[^\w.\- ()]+", "_", base).strip() or "file"
return base[:150]
def _sha1(*parts: str) -> str:
return hashlib.sha1("||".join(parts).encode("utf-8")).hexdigest()
def _resolve_workspace(conn, workspace_id: int | None, workspace_name: str | None,
user_login: str) -> tuple[int | None, str]:
if workspace_id:
row = conn.execute("SELECT id, name FROM workspaces WHERE id=?", (workspace_id,)).fetchone()
if row:
return row["id"], row["name"]
name = workspace_name or user_login
if name:
row = conn.execute("SELECT id, name FROM workspaces WHERE name=?", (name,)).fetchone()
if row:
return row["id"], row["name"]
return workspace_id, name or ""
def _rewrite_links(text: str, attachment_map: dict[str, str]) -> str:
"""Point Markdown links/images at uploaded attachment URLs."""
def repl(m: re.Match) -> str:
alt, target = m.group(1), m.group(2)
url = attachment_map.get(target) or attachment_map.get(Path(target).name.lower())
return f"![{alt}]({url})" if url else m.group(0)
text = re.sub(r"!\[([^\]]*)\]\(([^)\s]+)(?:\s+\"[^\"]*\")?\)", repl, text)
return text
def _extract_media(blocks: list[dict]) -> list[dict]:
"""Split inline image markdown out of paragraphs into real image blocks."""
out: list[dict] = []
for block in blocks:
if block.get("type") != "paragraph":
out.append(block)
continue
content = str(block.get("content", ""))
pos = 0
found = False
for match in _IMG_RE.finditer(content):
found = True
before = content[pos:match.start()].strip()
if before:
out.append({"type": "paragraph", "content": before})
out.append({"type": "image", "src": match.group(2), "alt": match.group(1)})
pos = match.end()
if not found:
out.append(block)
continue
tail = content[pos:].strip()
if tail:
out.append({"type": "paragraph", "content": tail})
return out
def _properties_callout(props: dict) -> dict | None:
if not props:
return None
lines = [f"**{k}** : {', '.join(map(str, v)) if isinstance(v, list) else v}" for k, v in props.items()]
return {"type": "callout", "icon": "ℹ️", "content": "\n".join(lines)}
def _blocks_for(page: ImportPage, attachment_map: dict[str, str], *, include_properties: bool) -> list[dict]:
if page.blocks:
return _extract_media(page.blocks)
md = _rewrite_links(page.markdown or "", attachment_map)
blocks = _extract_media(markdown_to_blocks(md))
if include_properties and page.properties:
callout = _properties_callout(page.properties)
if callout:
blocks.insert(0, callout)
return blocks
def preview_result(result: ImportResult) -> dict[str, Any]:
"""Dry-run preview: what would be created, without touching the database."""
pages = []
for page in result.pages:
if page.collection:
kind = "collection"
rows = len(page.collection.get("rows", []))
blocks = len(_blocks_for(page, {}, include_properties=False))
schema = page.collection.get("schema", [])
else:
kind = "page"
rows = 0
blocks = len(_blocks_for(page, {}, include_properties=False))
schema = []
pages.append({
"title": page.title,
"type": kind,
"source_path": page.source_path,
"parent_path": page.parent_path,
"properties": list(page.properties.keys()),
"rows": rows,
"blocks": blocks,
"schema": schema,
})
return {
"source": result.source,
"dry_run": True,
"pages": pages,
"stats": {
**result.stats,
"pages": len(result.pages),
"collections": sum(1 for p in result.pages if p.collection),
"attachments": len(result.attachments),
"warnings": len(result.warnings),
},
"warnings": result.warnings,
}
def _insert_page(conn, *, workspace: str, workspace_id: int | None, title: str,
blocks: list[dict], parent_id: int | None, sort_order: int,
content_format: str = "blocks") -> int:
cur = conn.execute(
"INSERT INTO pages (workspace, title, content, content_format, parent_section, "
"sort_order, workspace_id, parent_id) VALUES (?,?,?,?,'Private',?,?,?)",
(workspace, title or "Untitled", json.dumps(blocks), content_format,
sort_order, workspace_id, parent_id),
)
return cur.lastrowid
def _prop_id_map(conn, collection_id: int) -> dict[str, int]:
return {
r["name"]: r["id"]
for r in conn.execute(
"SELECT id, name FROM collection_properties WHERE collection_id=?",
(collection_id,),
).fetchall()
}
def _rows_to_values(rows: list[dict], id_map: dict[str, int]) -> list[tuple[str, dict]]:
"""Key row properties by property id (the shape the editor reads)."""
out: list[tuple[str, dict]] = []
for row in rows:
values: dict[str, Any] = {}
for name, value in (row.get("properties") or {}).items():
prop_id = id_map.get(name)
if prop_id is not None:
values[str(prop_id)] = value
out.append((row.get("title") or "Untitled", values))
return out
def _ensure_default_view(conn, collection_id: int) -> None:
conn.execute(
"INSERT INTO collection_views (collection_id, name, view_type, config_json) "
"VALUES (?,?,?,?)",
(collection_id, "Default View", "table",
json.dumps({"visible_properties": ["Title"], "sorts": [], "filters": []})),
)
def _insert_collection(conn, page: ImportPage, *, workspace_id: int | None,
parent_page_id: int | None) -> tuple[int, int]:
spec = page.collection or {}
schema = spec.get("schema", [])
cur = conn.execute(
"INSERT INTO collections (name, description, icon, schema_json, is_inline, "
"parent_page_id, workspace_id) VALUES (?,?,?,?,1,?,?)",
(page.title or spec.get("name") or "Imported database", "", "📥",
json.dumps(schema), parent_page_id, workspace_id),
)
collection_id = cur.lastrowid
materialize_properties(conn, collection_id, schema)
_ensure_default_view(conn, collection_id)
id_map = _prop_id_map(conn, collection_id)
position = 0
for title, values in _rows_to_values(spec.get("rows", []), id_map):
conn.execute(
"INSERT INTO collection_pages (collection_id, title, position, property_values_json) "
"VALUES (?,?,?,?)",
(collection_id, title, position, json.dumps(values)),
)
position += 1
return collection_id, position
def _upsert_collection_rows(conn, collection_id: int, rows: list[dict]) -> tuple[int, int]:
"""Update existing rows by title, insert the new ones. Returns (created, updated)."""
id_map = _prop_id_map(conn, collection_id)
existing = {
(r["title"] or "").strip(): r["id"]
for r in conn.execute(
"SELECT id, title FROM collection_pages WHERE collection_id=?", (collection_id,)
).fetchall()
}
max_pos = conn.execute(
"SELECT COALESCE(MAX(position), -1) FROM collection_pages WHERE collection_id=?",
(collection_id,),
).fetchone()[0]
created = updated = 0
for title, values in _rows_to_values(rows, id_map):
pid = existing.get((title or "").strip())
if pid:
conn.execute(
"UPDATE collection_pages SET property_values_json=?, updated_at=CURRENT_TIMESTAMP WHERE id=?",
(json.dumps(values), pid),
)
updated += 1
else:
max_pos += 1
conn.execute(
"INSERT INTO collection_pages (collection_id, title, position, property_values_json) "
"VALUES (?,?,?,?)",
(collection_id, title, max_pos, json.dumps(values)),
)
created += 1
return created, updated
def _update_page(conn, page_id: int, title: str, blocks: list[dict]) -> None:
conn.execute(
"UPDATE pages SET title=?, content=?, content_format='blocks', "
"updated_at=CURRENT_TIMESTAMP WHERE id=?",
(title or "Untitled", json.dumps(blocks), page_id),
)
def run_import(
result: ImportResult,
*,
workspace_id: int | None = None,
workspace_name: str | None = None,
user_login: str = "",
parent_page_id: int | None = None,
target_collection_id: int | None = None,
dry_run: bool = False,
dedup: bool = True,
include_properties: bool = True,
mapping: dict[str, str] | None = None,
mode: str | None = None,
) -> dict[str, Any]:
"""Persist an import result. Returns a report dict.
``mode`` controls re-import behaviour for items already imported (matched by
``import_items``): ``skip`` (default), ``update`` (re-sync in place) or
``duplicate`` (always create a new page).
"""
if dry_run:
return preview_result(result)
if mapping:
for page in result.pages:
if page.collection:
apply_type_mapping(page.collection, mapping)
if not mode:
mode = "skip" if dedup else "duplicate"
report: dict[str, Any] = {
"source": result.source,
"status": "ok",
"mode": mode,
"pages_created": 0,
"pages_updated": 0,
"collections_created": 0,
"rows_created": 0,
"rows_updated": 0,
"attachments": 0,
"skipped": 0,
"page_ids": [],
"errors": [],
"warnings": list(result.warnings),
}
with get_conn() as conn:
ws_id, ws_name = _resolve_workspace(conn, workspace_id, workspace_name, user_login)
if not ws_name:
report["status"] = "error"
report["warnings"].append("Workspace introuvable")
return report
attachment_map: dict[str, str] = {}
if result.attachments and ws_id:
dest_dir = _data_dir() / "uploads" / f"workspace_{ws_id}" / "import"
dest_dir.mkdir(parents=True, exist_ok=True)
for att in result.attachments:
safe = _safe_filename(att.filename)
target = dest_dir / safe
if target.exists():
target = dest_dir / f"{_sha1(att.source_path)[:8]}_{safe}"
try:
target.write_bytes(att.data)
except OSError:
continue
url = f"/api/files/{ws_id}/import/{target.name}"
attachment_map[att.source_path] = url
attachment_map[att.source_path.lower()] = url
attachment_map[Path(att.source_path).name.lower()] = url
report["attachments"] += 1
if target_collection_id:
return _import_into_collection(
conn, result, target_collection_id, report, attachment_map,
dedup=dedup, workspace_id=ws_id, include_properties=include_properties,
)
next_order = conn.execute(
"SELECT COALESCE(MAX(sort_order), -1) + 1 FROM pages WHERE workspace=? AND parent_id IS NULL",
(ws_name,),
).fetchone()[0]
order_counter = [next_order]
path_to_page: dict[str, int] = {}
folder_cache: dict[str, int | None] = {}
def ensure_folder(path: str, _depth: int = 0) -> int | None:
path = (path or "").strip("/")
if not path:
return parent_page_id
if path in folder_cache:
return folder_cache[path]
parent_path = "/".join(path.split("/")[:-1])
parent_id = ensure_folder(parent_path, _depth + 1)
title = path.split("/")[-1] or "Folder"
pid = _insert_page(
conn, workspace=ws_name, workspace_id=ws_id, title=title,
blocks=[], parent_id=parent_id, sort_order=order_counter[0],
)
order_counter[0] += 1
folder_cache[path] = pid
path_to_page[path] = pid
report["pages_created"] += 1
report["page_ids"].append(pid)
return pid
ordered = sorted(
result.pages,
key=lambda p: (p.parent_path.count("/") if p.parent_path else -1, p.source_path.lower()),
)
for page in ordered:
external = page.external_id or page.source_path or page.title
existing_pid = None
if external:
row = conn.execute(
"SELECT page_id FROM import_items WHERE workspace_id IS ? AND source=? AND external_id=?",
(ws_id, result.source, external),
).fetchone()
if row:
existing_pid = row["page_id"]
if existing_pid and mode == "skip":
report["skipped"] += 1
continue
try:
page_parent = ensure_folder(page.parent_path) if page.parent_path else parent_page_id
blocks = _blocks_for(page, attachment_map, include_properties=include_properties)
if existing_pid and mode == "update":
if page.collection:
cid = _collection_for_page(conn, existing_pid)
if cid:
materialize_properties(conn, cid, (page.collection or {}).get("schema", []))
created, updated = _upsert_collection_rows(
conn, cid, (page.collection or {}).get("rows", []))
report["rows_created"] += created
report["rows_updated"] += updated
blocks = blocks + [{"type": "embed", "embed_type": "collection",
"collection_id": cid, "content": ""}]
_update_page(conn, existing_pid, page.title, blocks)
report["pages_updated"] += 1
report["page_ids"].append(existing_pid)
path_to_page[page.source_path] = existing_pid
continue
if page.collection:
pid = _insert_page(
conn, workspace=ws_name, workspace_id=ws_id, title=page.title,
blocks=blocks, parent_id=page_parent, sort_order=order_counter[0],
)
order_counter[0] += 1
cid, rows = _insert_collection(conn, page, workspace_id=ws_id, parent_page_id=pid)
conn.execute(
"UPDATE pages SET content=? WHERE id=?",
(json.dumps(blocks + [{"type": "embed", "embed_type": "collection",
"collection_id": cid, "content": ""}]), pid),
)
report["collections_created"] += 1
report["rows_created"] += rows
report["pages_created"] += 1
report["page_ids"].append(pid)
else:
pid = _insert_page(
conn, workspace=ws_name, workspace_id=ws_id, title=page.title,
blocks=blocks, parent_id=page_parent, sort_order=order_counter[0],
)
order_counter[0] += 1
report["pages_created"] += 1
report["page_ids"].append(pid)
path_to_page[page.source_path] = pid
if external:
conn.execute(
"INSERT OR IGNORE INTO import_items (workspace_id, source, external_id, page_id) VALUES (?,?,?,?)",
(ws_id, result.source, external, pid),
)
except Exception as exc: # noqa: BLE001 - partial import must keep going
logger.warning("import failed for %r: %s", page.title, exc)
report["errors"].append({"title": page.title, "error": str(exc)})
conn.commit()
if report["errors"]:
report["status"] = "partial"
return report
def _collection_for_page(conn, page_id: int) -> int | None:
row = conn.execute(
"SELECT id FROM collections WHERE parent_page_id=? ORDER BY id LIMIT 1", (page_id,)
).fetchone()
return row["id"] if row else None
def _import_into_collection(conn, result: ImportResult, collection_id: int, report: dict,
attachment_map: dict[str, str], *, dedup: bool,
workspace_id: int | None, include_properties: bool) -> dict:
exists = conn.execute("SELECT id FROM collections WHERE id=?", (collection_id,)).fetchone()
if not exists:
report["status"] = "error"
report["warnings"].append("Collection cible introuvable")
return report
max_pos = conn.execute(
"SELECT COALESCE(MAX(position), -1) + 1 FROM collection_pages WHERE collection_id=?",
(collection_id,),
).fetchone()[0]
id_map = _prop_id_map(conn, collection_id)
for page in result.pages:
external = page.external_id or page.source_path or page.title
if dedup and external:
row = conn.execute(
"SELECT id FROM collection_pages WHERE collection_id=? AND title=?",
(collection_id, page.title),
).fetchone()
if row:
report["skipped"] += 1
continue
props = {
str(id_map[name]): value
for name, value in page.properties.items()
if name in id_map
}
conn.execute(
"INSERT INTO collection_pages (collection_id, title, position, property_values_json) "
"VALUES (?,?,?,?)",
(collection_id, page.title, max_pos, json.dumps(props)),
)
max_pos += 1
report["rows_created"] += 1
conn.commit()
return report
def resolve_relations(conn, workspace_id: int | None) -> dict[str, Any]:
"""Convert text columns that reference another imported collection's titles
into real ``relation`` properties (array of ``collection_pages`` ids)."""
collections = conn.execute(
"SELECT id, name FROM collections WHERE workspace_id IS ?", (workspace_id,)
).fetchall()
if not collections:
return {"relations_resolved": 0, "details": []}
titles: dict[int, dict[str, int]] = {}
for coll in collections:
rows = conn.execute(
"SELECT id, title FROM collection_pages WHERE collection_id=?", (coll["id"],)
).fetchall()
titles[coll["id"]] = {
(r["title"] or "").strip(): r["id"]
for r in rows if (r["title"] or "").strip()
}
resolved = 0
details: list[dict] = []
for coll in collections:
props = conn.execute(
"SELECT id, name, prop_type FROM collection_properties WHERE collection_id=?",
(coll["id"],),
).fetchall()
pages = conn.execute(
"SELECT id, property_values_json FROM collection_pages WHERE collection_id=?",
(coll["id"],),
).fetchall()
for prop in props:
if prop["prop_type"] not in ("text", "select", "multi_select"):
continue
key = str(prop["id"])
values: list[str] = []
for page in pages:
pv = json.loads(page["property_values_json"] or "{}")
value = pv.get(key)
if value in (None, "", []):
continue
items = value if isinstance(value, list) else [value]
values.extend(str(v) for v in items)
if not values:
continue
for target in collections:
if target["id"] == coll["id"]:
continue
target_titles = titles.get(target["id"]) or {}
if target_titles and all(v in target_titles for v in values):
conn.execute(
"UPDATE collection_properties SET prop_type='relation', "
"related_collection_id=? WHERE id=?",
(target["id"], prop["id"]),
)
for page in pages:
pv = json.loads(page["property_values_json"] or "{}")
value = pv.get(key)
if value in (None, "", []):
continue
items = value if isinstance(value, list) else [value]
pv[key] = [target_titles[str(v)] for v in items if str(v) in target_titles]
conn.execute(
"UPDATE collection_pages SET property_values_json=? WHERE id=?",
(json.dumps(pv), page["id"]),
)
resolved += 1
details.append({
"collection": coll["name"], "property": prop["name"],
"related": target["name"],
})
break
conn.commit()
return {"relations_resolved": resolved, "details": details}