Files
bruno 3b00cbc371 feat(import): v5.6.0 unified data import (phases 0-5)
- unified importer framework (app/services/importers/): normalized model,
  registry, common pipeline (hierarchy, attachments, collections, dedup),
  async jobs, dry-run preview, column->type mapping
- Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/
  OneNote), Google Keep, generic Markdown
- Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON
- Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders
- Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics,
  OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones)
- Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume,
  forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue,
  Notion relation resolution, exportable JSON reports
- /import wizard, API /api/import/*, migration 9 (import_items, import_jobs)
- fix: property values stored by property id (correct DB view rendering)
- deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf
- 43 import tests; full suite 491 green; ruff clean
- bump version 5.11.5
2026-09-13 10:43:20 -04:00

123 lines
4.3 KiB
Python

"""FlowDeck — Word (.docx) importer (v5.6.0, Phase 3).
Converts a Word document (including Google Docs Takeout ``.docx`` exports) into
a FlowDeck page: headings, lists, tables and inline images.
"""
from __future__ import annotations
import io
import re
from pathlib import Path
from app.services.importers.base import (
ImportAttachment,
Importer,
ImportResult,
register_importer,
)
_HEADING_STYLES = {
"title": 1, "heading 1": 1, "heading 2": 2, "heading 3": 3,
"heading 4": 4, "heading 5": 4, "heading 6": 4,
}
_MIME = {
".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg",
".gif": "image/gif", ".webp": "image/webp", ".bmp": "image/bmp",
".emf": "image/emf", ".wmf": "image/wmf", ".tiff": "image/tiff",
}
def _escape_cell(text: str) -> str:
return text.strip().replace("|", "\\|").replace("\n", " ")
def _table_markdown(table) -> str:
rows: list[list[str]] = []
for row in table.rows:
rows.append([_escape_cell(cell.text) for cell in row.cells])
if not rows:
return ""
width = max(len(r) for r in rows)
rows = [r + [""] * (width - len(r)) for r in rows]
header = "| " + " | ".join(rows[0]) + " |"
sep = "| " + " | ".join(["---"] * width) + " |"
body = "\n".join("| " + " | ".join(r) + " |" for r in rows[1:])
return "\n".join(x for x in (header, sep, body) if x)
@register_importer
class DocxImporter(Importer):
source_id = "docx"
label = "Word / Google Docs (.docx)"
description = "Document Word : titres, listes, tableaux et images."
extensions = (".docx", ".docm")
order = 36
def detect(self, filename: str, data: bytes) -> bool:
return filename.lower().endswith((".docx", ".docm"))
def parse(self, filename: str, data: bytes) -> ImportResult:
result = ImportResult(source=self.source_id)
try:
from docx import Document
from docx.oxml.ns import qn
except ImportError:
result.warn("python-docx n'est pas installé : import Word indisponible")
return result.finalize()
try:
doc = Document(io.BytesIO(data))
except Exception as exc: # noqa: BLE001
result.warn(f"Document illisible : {exc}")
return result.finalize()
lines: list[str] = []
for para in doc.paragraphs:
text = para.text.strip()
style = (para.style.name or "").lower() if para.style else ""
images = self._paragraph_images(doc, para, qn, result)
if text:
level = _HEADING_STYLES.get(style)
if level:
lines.append("#" * level + " " + text)
elif "list bullet" in style or "list paragraph" in style:
lines.append("- " + text)
elif "list number" in style:
lines.append("1. " + text)
elif style == "quote":
lines.append("> " + text)
else:
lines.append(text)
lines.extend(images)
for table in doc.tables:
md = _table_markdown(table)
if md:
lines.append(md)
markdown = re.sub(r"\n{3,}", "\n\n", "\n\n".join(lines)).strip()
title = Path(filename).stem or "Document"
result.pages.append(_page(title, markdown, filename))
return result.finalize()
def _paragraph_images(self, doc, para, qn, result: ImportResult) -> list[str]:
images: list[str] = []
for blip in para._p.iter(qn("a:blip")):
rid = blip.get(qn("r:embed")) or blip.get(qn("r:link"))
if not rid:
continue
part = doc.part.related_parts.get(rid)
if part is None or not hasattr(part, "blob"):
continue
name = Path(str(part.partname)).name or f"image_{len(result.attachments)}.png"
result.attachments.append(ImportAttachment(
source_path=name, filename=name, data=part.blob,
mime=_MIME.get(Path(name).suffix.lower(), "application/octet-stream"),
))
images.append(f"![{name}]({name})")
return images
def _page(title: str, markdown: str, filename: str):
from app.services.importers.base import ImportPage
return ImportPage(title=title, markdown=markdown, source_path=filename, external_id=filename)