"""PDF text extraction for ObsiGate — pypdf backend, pymupdf optional fallback.""" from __future__ import annotations import logging import os from concurrent.futures import ThreadPoolExecutor from concurrent.futures import TimeoutError as FuturesTimeout from pathlib import Path logger = logging.getLogger(__name__) # Configurable limits (see ROADMAP #74 G3): # OBSIGATE_PDF_MAX_SIZE_MB — PDFs larger than this are not text-extracted (default 50) # OBSIGATE_PDF_EXTRACT_TIMEOUT — seconds before extraction is abandoned (default 30) PDF_MAX_SIZE_MB: int = int(os.environ.get("OBSIGATE_PDF_MAX_SIZE_MB", "50")) PDF_EXTRACT_TIMEOUT: float = float(os.environ.get("OBSIGATE_PDF_EXTRACT_TIMEOUT", "30")) PDF_READER: str = "pypdf" PdfReader = None # type: ignore try: import fitz # pymupdf PDF_READER = "pymupdf" logger.info("PDF reader: pymupdf (high performance)") except ImportError: try: from pypdf import PdfReader # type: ignore logger.info("PDF reader: pypdf (pure Python)") except ImportError: logger.warning("No PDF reader available — install pypdf or pymupdf") def pdf_exceeds_size_limit(file_path: Path) -> bool: """True if the PDF is larger than OBSIGATE_PDF_MAX_SIZE_MB (skip text extraction).""" try: return file_path.stat().st_size > PDF_MAX_SIZE_MB * 1024 * 1024 except OSError: return False def _run_with_timeout(fn, *args, timeout: float): """Run a sync function in a worker thread with a hard timeout. A timed-out extraction leaves its worker thread running (daemon-style pool is abandoned), but the request itself is freed — acceptable trade-off for pathological PDFs on a self-hosted single-user server. """ executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="pdf-extract") try: future = executor.submit(fn, *args) return future.result(timeout=timeout) finally: executor.shutdown(wait=False) def extract_pdf_text(file_path: Path, max_chars: int = 100000) -> str: """Extract text from a PDF file. Returns empty string on failure. Oversized PDFs (> OBSIGATE_PDF_MAX_SIZE_MB) and extractions exceeding OBSIGATE_PDF_EXTRACT_TIMEOUT seconds return "" instead of blocking. """ if PdfReader is None and PDF_READER == "pypdf": return "" if pdf_exceeds_size_limit(file_path): logger.info("PDF too large to index (> %d MB), skipping text extraction: %s", PDF_MAX_SIZE_MB, file_path) return "" try: if PDF_READER == "pymupdf": return _run_with_timeout(_extract_pymupdf, file_path, max_chars, timeout=PDF_EXTRACT_TIMEOUT) else: return _run_with_timeout(_extract_pypdf, file_path, max_chars, timeout=PDF_EXTRACT_TIMEOUT) except FuturesTimeout: logger.warning("PDF text extraction timed out (%ss): %s", PDF_EXTRACT_TIMEOUT, file_path) return "" except Exception as e: logger.warning("Failed to extract PDF text from %s: %s", file_path, e) return "" def extract_pdf_metadata(file_path: Path) -> dict: """Extract metadata from a PDF file.""" info = {"pages": 0, "title": "", "author": ""} try: if PDF_READER == "pymupdf": doc = fitz.open(str(file_path)) info["pages"] = doc.page_count meta = doc.metadata or {} info["title"] = meta.get("title", "") info["author"] = meta.get("author", "") doc.close() else: reader = PdfReader(str(file_path)) info["pages"] = len(reader.pages) meta = reader.metadata or {} if meta: info["title"] = str(meta.get("/Title", "")) info["author"] = str(meta.get("/Author", "")) except Exception as e: logger.warning("Failed to extract PDF metadata from %s: %s", file_path, e) return info def extract_pdf_toc(file_path: Path) -> list[dict]: """Extract table of contents (bookmarks/outline) from a PDF file. Returns a list of {title, page, level} dicts, or empty list on failure. """ toc: list[dict] = [] try: if PDF_READER == "pymupdf": doc = fitz.open(str(file_path)) raw = doc.get_toc(simple=False) for item in raw: toc.append({ "title": str(item[1]), "page": item[2], "level": item[0], }) doc.close() elif PdfReader is not None: reader = PdfReader(str(file_path)) outline = reader.outline if outline: def _flatten(items, level=1): for item in items: if isinstance(item, list): _flatten(item, level + 1) elif hasattr(item, 'title') and hasattr(item, 'page'): page_num = reader.get_page_number(item.page) + 1 if hasattr(item, 'page') and item.page else 1 toc.append({ "title": str(item.title), "page": page_num, "level": level, }) _flatten(outline) except Exception as e: logger.warning("Failed to extract PDF TOC from %s: %s", file_path, e) return toc def _extract_pymupdf(file_path: Path, max_chars: int) -> str: doc = fitz.open(str(file_path)) parts = [] total = 0 for page in doc: text = page.get_text() if text: if total + len(text) > max_chars: remaining = max_chars - total if remaining > 0: parts.append(text[:remaining]) break parts.append(text) total += len(text) doc.close() return "\f".join(parts) def _extract_pypdf(file_path: Path, max_chars: int) -> str: reader = PdfReader(str(file_path)) parts = [] total = 0 for page in reader.pages: text = page.extract_text() or "" if text: if total + len(text) > max_chars: remaining = max_chars - total if remaining > 0: parts.append(text[:remaining]) break parts.append(text) total += len(text) return "\f".join(parts)