- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
123 lines
4.3 KiB
Python
123 lines
4.3 KiB
Python
"""FlowDeck — Word (.docx) importer (v5.6.0, Phase 3).
|
|
|
|
Converts a Word document (including Google Docs Takeout ``.docx`` exports) into
|
|
a FlowDeck page: headings, lists, tables and inline images.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
import re
|
|
from pathlib import Path
|
|
|
|
from app.services.importers.base import (
|
|
ImportAttachment,
|
|
Importer,
|
|
ImportResult,
|
|
register_importer,
|
|
)
|
|
|
|
_HEADING_STYLES = {
|
|
"title": 1, "heading 1": 1, "heading 2": 2, "heading 3": 3,
|
|
"heading 4": 4, "heading 5": 4, "heading 6": 4,
|
|
}
|
|
_MIME = {
|
|
".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg",
|
|
".gif": "image/gif", ".webp": "image/webp", ".bmp": "image/bmp",
|
|
".emf": "image/emf", ".wmf": "image/wmf", ".tiff": "image/tiff",
|
|
}
|
|
|
|
|
|
def _escape_cell(text: str) -> str:
|
|
return text.strip().replace("|", "\\|").replace("\n", " ")
|
|
|
|
|
|
def _table_markdown(table) -> str:
|
|
rows: list[list[str]] = []
|
|
for row in table.rows:
|
|
rows.append([_escape_cell(cell.text) for cell in row.cells])
|
|
if not rows:
|
|
return ""
|
|
width = max(len(r) for r in rows)
|
|
rows = [r + [""] * (width - len(r)) for r in rows]
|
|
header = "| " + " | ".join(rows[0]) + " |"
|
|
sep = "| " + " | ".join(["---"] * width) + " |"
|
|
body = "\n".join("| " + " | ".join(r) + " |" for r in rows[1:])
|
|
return "\n".join(x for x in (header, sep, body) if x)
|
|
|
|
|
|
@register_importer
|
|
class DocxImporter(Importer):
|
|
source_id = "docx"
|
|
label = "Word / Google Docs (.docx)"
|
|
description = "Document Word : titres, listes, tableaux et images."
|
|
extensions = (".docx", ".docm")
|
|
order = 36
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
return filename.lower().endswith((".docx", ".docm"))
|
|
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
result = ImportResult(source=self.source_id)
|
|
try:
|
|
from docx import Document
|
|
from docx.oxml.ns import qn
|
|
except ImportError:
|
|
result.warn("python-docx n'est pas installé : import Word indisponible")
|
|
return result.finalize()
|
|
try:
|
|
doc = Document(io.BytesIO(data))
|
|
except Exception as exc: # noqa: BLE001
|
|
result.warn(f"Document illisible : {exc}")
|
|
return result.finalize()
|
|
|
|
lines: list[str] = []
|
|
for para in doc.paragraphs:
|
|
text = para.text.strip()
|
|
style = (para.style.name or "").lower() if para.style else ""
|
|
images = self._paragraph_images(doc, para, qn, result)
|
|
if text:
|
|
level = _HEADING_STYLES.get(style)
|
|
if level:
|
|
lines.append("#" * level + " " + text)
|
|
elif "list bullet" in style or "list paragraph" in style:
|
|
lines.append("- " + text)
|
|
elif "list number" in style:
|
|
lines.append("1. " + text)
|
|
elif style == "quote":
|
|
lines.append("> " + text)
|
|
else:
|
|
lines.append(text)
|
|
lines.extend(images)
|
|
for table in doc.tables:
|
|
md = _table_markdown(table)
|
|
if md:
|
|
lines.append(md)
|
|
|
|
markdown = re.sub(r"\n{3,}", "\n\n", "\n\n".join(lines)).strip()
|
|
title = Path(filename).stem or "Document"
|
|
result.pages.append(_page(title, markdown, filename))
|
|
return result.finalize()
|
|
|
|
def _paragraph_images(self, doc, para, qn, result: ImportResult) -> list[str]:
|
|
images: list[str] = []
|
|
for blip in para._p.iter(qn("a:blip")):
|
|
rid = blip.get(qn("r:embed")) or blip.get(qn("r:link"))
|
|
if not rid:
|
|
continue
|
|
part = doc.part.related_parts.get(rid)
|
|
if part is None or not hasattr(part, "blob"):
|
|
continue
|
|
name = Path(str(part.partname)).name or f"image_{len(result.attachments)}.png"
|
|
result.attachments.append(ImportAttachment(
|
|
source_path=name, filename=name, data=part.blob,
|
|
mime=_MIME.get(Path(name).suffix.lower(), "application/octet-stream"),
|
|
))
|
|
images.append(f"")
|
|
return images
|
|
|
|
|
|
def _page(title: str, markdown: str, filename: str):
|
|
from app.services.importers.base import ImportPage
|
|
|
|
return ImportPage(title=title, markdown=markdown, source_path=filename, external_id=filename)
|