CI / lint (push) Successful in 1m1s
CI / security (push) Successful in 40s
CI / test (push) Successful in 1m17s
CI / build (push) Successful in 38s
CI / e2e (push) Successful in 10m55s
Desktop Build / build-windows (push) Canceled after 0s
Desktop Build / build-linux (push) Canceled after 0s
BUG-003: annotations de types, gardes None sur get_user(), PdfReader: Any et import PROVIDERS manquant (bug latent main.py:4523). Etape mypy du CI rendue bloquante (etait advisory). BUG-004: lien README.md -> docs/CONTRIBUTING.md corrige (+ DELIVERY_WORKFLOW.md), arbre projet mis a jour, parite README.fr.md. Verifie: mypy 0 erreur, ruff OK, pytest 728 passed / 5 skipped, frontend OK, liens md OK.
179 lines
6.5 KiB
Python
179 lines
6.5 KiB
Python
"""PDF text extraction for ObsiGate — pypdf backend, pymupdf optional fallback."""
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import os
|
|
from concurrent.futures import ThreadPoolExecutor
|
|
from concurrent.futures import TimeoutError as FuturesTimeout
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Configurable limits (see ROADMAP #74 G3):
|
|
# OBSIGATE_PDF_MAX_SIZE_MB — PDFs larger than this are not text-extracted (default 50)
|
|
# OBSIGATE_PDF_EXTRACT_TIMEOUT — seconds before extraction is abandoned (default 30)
|
|
PDF_MAX_SIZE_MB: int = int(os.environ.get("OBSIGATE_PDF_MAX_SIZE_MB", "50"))
|
|
PDF_EXTRACT_TIMEOUT: float = float(os.environ.get("OBSIGATE_PDF_EXTRACT_TIMEOUT", "30"))
|
|
|
|
PDF_READER: str = "pypdf"
|
|
PdfReader: Any = None
|
|
try:
|
|
import fitz # pymupdf
|
|
PDF_READER = "pymupdf"
|
|
logger.info("PDF reader: pymupdf (high performance)")
|
|
except ImportError:
|
|
try:
|
|
from pypdf import PdfReader # type: ignore
|
|
logger.info("PDF reader: pypdf (pure Python)")
|
|
except ImportError:
|
|
logger.warning("No PDF reader available — install pypdf or pymupdf")
|
|
|
|
|
|
def pdf_exceeds_size_limit(file_path: Path) -> bool:
|
|
"""True if the PDF is larger than OBSIGATE_PDF_MAX_SIZE_MB (skip text extraction)."""
|
|
try:
|
|
return file_path.stat().st_size > PDF_MAX_SIZE_MB * 1024 * 1024
|
|
except OSError:
|
|
return False
|
|
|
|
|
|
def _run_with_timeout(fn, *args, timeout: float):
|
|
"""Run a sync function in a worker thread with a hard timeout.
|
|
|
|
A timed-out extraction leaves its worker thread running (daemon-style pool
|
|
is abandoned), but the request itself is freed — acceptable trade-off for
|
|
pathological PDFs on a self-hosted single-user server.
|
|
"""
|
|
executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="pdf-extract")
|
|
try:
|
|
future = executor.submit(fn, *args)
|
|
return future.result(timeout=timeout)
|
|
finally:
|
|
executor.shutdown(wait=False)
|
|
|
|
|
|
def extract_pdf_text(file_path: Path, max_chars: int = 100000) -> str:
|
|
"""Extract text from a PDF file. Returns empty string on failure.
|
|
|
|
Oversized PDFs (> OBSIGATE_PDF_MAX_SIZE_MB) and extractions exceeding
|
|
OBSIGATE_PDF_EXTRACT_TIMEOUT seconds return "" instead of blocking.
|
|
"""
|
|
if PdfReader is None and PDF_READER == "pypdf":
|
|
return ""
|
|
if pdf_exceeds_size_limit(file_path):
|
|
logger.info("PDF too large to index (> %d MB), skipping text extraction: %s",
|
|
PDF_MAX_SIZE_MB, file_path)
|
|
return ""
|
|
try:
|
|
if PDF_READER == "pymupdf":
|
|
return _run_with_timeout(_extract_pymupdf, file_path, max_chars,
|
|
timeout=PDF_EXTRACT_TIMEOUT)
|
|
else:
|
|
return _run_with_timeout(_extract_pypdf, file_path, max_chars,
|
|
timeout=PDF_EXTRACT_TIMEOUT)
|
|
except FuturesTimeout:
|
|
logger.warning("PDF text extraction timed out (%ss): %s", PDF_EXTRACT_TIMEOUT, file_path)
|
|
return ""
|
|
except Exception as e:
|
|
logger.warning("Failed to extract PDF text from %s: %s", file_path, e)
|
|
return ""
|
|
|
|
|
|
def extract_pdf_metadata(file_path: Path) -> dict:
|
|
"""Extract metadata from a PDF file."""
|
|
info = {"pages": 0, "title": "", "author": ""}
|
|
try:
|
|
if PDF_READER == "pymupdf":
|
|
doc = fitz.open(str(file_path))
|
|
info["pages"] = doc.page_count
|
|
meta = doc.metadata or {}
|
|
info["title"] = meta.get("title", "")
|
|
info["author"] = meta.get("author", "")
|
|
doc.close()
|
|
elif PdfReader is not None:
|
|
reader = PdfReader(str(file_path))
|
|
info["pages"] = len(reader.pages)
|
|
meta = reader.metadata or {}
|
|
if meta:
|
|
info["title"] = str(meta.get("/Title", ""))
|
|
info["author"] = str(meta.get("/Author", ""))
|
|
except Exception as e:
|
|
logger.warning("Failed to extract PDF metadata from %s: %s", file_path, e)
|
|
return info
|
|
|
|
|
|
def extract_pdf_toc(file_path: Path) -> list[dict]:
|
|
"""Extract table of contents (bookmarks/outline) from a PDF file.
|
|
|
|
Returns a list of {title, page, level} dicts, or empty list on failure.
|
|
"""
|
|
toc: list[dict] = []
|
|
try:
|
|
if PDF_READER == "pymupdf":
|
|
doc = fitz.open(str(file_path))
|
|
raw = doc.get_toc(simple=False)
|
|
for item in raw:
|
|
toc.append({
|
|
"title": str(item[1]),
|
|
"page": item[2],
|
|
"level": item[0],
|
|
})
|
|
doc.close()
|
|
elif PdfReader is not None:
|
|
reader = PdfReader(str(file_path))
|
|
outline = reader.outline
|
|
if outline:
|
|
def _flatten(items, level=1):
|
|
for item in items:
|
|
if isinstance(item, list):
|
|
_flatten(item, level + 1)
|
|
elif hasattr(item, 'title') and hasattr(item, 'page'):
|
|
page_num = reader.get_page_number(item.page) + 1 if hasattr(item, 'page') and item.page else 1
|
|
toc.append({
|
|
"title": str(item.title),
|
|
"page": page_num,
|
|
"level": level,
|
|
})
|
|
_flatten(outline)
|
|
except Exception as e:
|
|
logger.warning("Failed to extract PDF TOC from %s: %s", file_path, e)
|
|
return toc
|
|
|
|
|
|
def _extract_pymupdf(file_path: Path, max_chars: int) -> str:
|
|
doc = fitz.open(str(file_path))
|
|
parts = []
|
|
total = 0
|
|
for page in doc:
|
|
text = page.get_text()
|
|
if text:
|
|
if total + len(text) > max_chars:
|
|
remaining = max_chars - total
|
|
if remaining > 0:
|
|
parts.append(text[:remaining])
|
|
break
|
|
parts.append(text)
|
|
total += len(text)
|
|
doc.close()
|
|
return "\f".join(parts)
|
|
|
|
|
|
def _extract_pypdf(file_path: Path, max_chars: int) -> str:
|
|
if PdfReader is None:
|
|
return ""
|
|
reader = PdfReader(str(file_path))
|
|
parts = []
|
|
total = 0
|
|
for page in reader.pages:
|
|
text = page.extract_text() or ""
|
|
if text:
|
|
if total + len(text) > max_chars:
|
|
remaining = max_chars - total
|
|
if remaining > 0:
|
|
parts.append(text[:remaining])
|
|
break
|
|
parts.append(text)
|
|
total += len(text)
|
|
return "\f".join(parts)
|