- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
95 lines
3.2 KiB
Python
95 lines
3.2 KiB
Python
"""FlowDeck — PDF importer (v5.6.0, Phase 3).
|
|
|
|
Best-effort text + image extraction from a PDF into a FlowDeck page (fidelity
|
|
depends on the source PDF; scanned documents have no text layer).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
import re
|
|
from pathlib import Path
|
|
|
|
from app.services.importers.base import (
|
|
ImportAttachment,
|
|
Importer,
|
|
ImportPage,
|
|
ImportResult,
|
|
register_importer,
|
|
)
|
|
|
|
_MIME = {".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg",
|
|
".gif": "image/gif", ".webp": "image/webp", ".bmp": "image/bmp",
|
|
".tiff": "image/tiff", ".tif": "image/tiff"}
|
|
|
|
|
|
@register_importer
|
|
class PdfImporter(Importer):
|
|
source_id = "pdf"
|
|
label = "PDF"
|
|
description = "Extraction texte + images d'un PDF (fidélité limitée)."
|
|
extensions = (".pdf",)
|
|
order = 37
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
return filename.lower().endswith(".pdf")
|
|
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
result = ImportResult(source=self.source_id)
|
|
try:
|
|
from pypdf import PdfReader
|
|
except ImportError:
|
|
result.warn("pypdf n'est pas installé : import PDF indisponible")
|
|
return result.finalize()
|
|
try:
|
|
reader = PdfReader(io.BytesIO(data))
|
|
except Exception as exc: # noqa: BLE001
|
|
result.warn(f"PDF illisible : {exc}")
|
|
return result.finalize()
|
|
|
|
chunks: list[str] = []
|
|
empty_pages = 0
|
|
for index, page in enumerate(reader.pages, start=1):
|
|
try:
|
|
text = (page.extract_text() or "").strip()
|
|
except Exception: # noqa: BLE001
|
|
text = ""
|
|
if text:
|
|
if len(reader.pages) > 1:
|
|
chunks.append(f"## Page {index}\n\n{text}")
|
|
else:
|
|
chunks.append(text)
|
|
else:
|
|
empty_pages += 1
|
|
chunks.extend(self._page_images(page, index, result))
|
|
|
|
if empty_pages:
|
|
result.warn(f"{empty_pages} page(s) sans couche texte (document scanné ?)")
|
|
markdown = re.sub(r"\n{3,}", "\n\n", "\n\n".join(chunks)).strip()
|
|
title = Path(filename).stem or "Document"
|
|
result.pages.append(ImportPage(
|
|
title=title, markdown=markdown, source_path=filename, external_id=filename,
|
|
))
|
|
return result.finalize()
|
|
|
|
def _page_images(self, page, index: int, result: ImportResult) -> list[str]:
|
|
images: list[str] = []
|
|
try:
|
|
page_images = list(page.images)
|
|
except Exception: # noqa: BLE001
|
|
return images
|
|
for i, image in enumerate(page_images, start=1):
|
|
name = getattr(image, "name", "") or f"page{index}_img{i}.png"
|
|
name = Path(name).name
|
|
try:
|
|
payload = image.data
|
|
except Exception: # noqa: BLE001
|
|
continue
|
|
if not payload:
|
|
continue
|
|
result.attachments.append(ImportAttachment(
|
|
source_path=f"page{index}/{name}", filename=name, data=payload,
|
|
mime=_MIME.get(Path(name).suffix.lower(), "application/octet-stream"),
|
|
))
|
|
images.append(f"")
|
|
return images
|