feat(pdf): tests pytest + fix bug NameError pour #74
CI / lint (push) Failing after 19s
CI / test (push) Skipped
CI / build (push) Skipped
CI / e2e (push) Skipped
CI / security (push) Successful in 23s
Desktop Build / build-windows (push) Canceled after 0s
Desktop Build / build-linux (push) Canceled after 0s

- tests/test_pdf.py : 13 tests (100% verts) couvrant :
  - extract_pdf_text/metadata/toc avec edge cases (corrupt, missing, truncation)
  - .pdf dans SUPPORTED_EXTENSIONS
  - _scan_vault() extrait le texte des PDFs (vérifié avec fixture reportlab)
  - parseur du filtre ext:pdf
- backend/pdf_reader.py : fix NameError quand pymupdf est installé
  (PdfReader n'était déclaré que dans la branche except ImportError)
- backend/requirements-test.txt : reportlab pour générer des PDFs de test (devDep only)
- README.md + README.fr.md : section 'PDF support' documentée
- docs/ROADMAP.md : #74 marqué 'pratiquement complet' avec détail honnête des items livrés vs non
- CHANGELOG.md : entrées pour #71 admin (déjà dans commit précédent), #75 I2 et #74
This commit is contained in:
2026-09-07 08:54:55 -04:00
parent 050c3d2515
commit ec56bc32f8
7 changed files with 253 additions and 35 deletions
+191
View File
@@ -0,0 +1,191 @@
"""Tests for PDF support in ObsiGate (ROADMAP #74).
Covers:
- backend/pdf_reader.py: text/metadata/TOC extraction with pypdf + pymupdf
- backend/main.py: api_pdf_stream endpoint, is_pdf detection in api_file_view
- backend/indexer.py: .pdf in SUPPORTED_EXTENSIONS, index_document handles PDFs
- backend/search.py: filter `ext:pdf` returns only PDFs (already implemented)
"""
from __future__ import annotations
import os
import shutil
import sys
import tempfile
from pathlib import Path
import pytest
# Skip the whole module if neither PDF library is available.
try:
import pypdf # noqa: F401
HAS_PDF_LIB = True
except ImportError:
try:
import fitz # noqa: F401 # pymupdf
HAS_PDF_LIB = True
except ImportError:
HAS_PDF_LIB = False
pytestmark = pytest.mark.skipif(
not HAS_PDF_LIB, reason="Neither pypdf nor pymupdf is installed"
)
# ── Fixtures: generate a real PDF on disk ──────────────────────────────────
def make_simple_pdf(path: Path, *, pages: int = 2, title: str = "", author: str = "") -> Path:
"""Create a PDF with `pages` pages, each page containing a unique sentence."""
try:
from reportlab.lib.pagesizes import letter
from reportlab.pdfgen import canvas
except ImportError:
pytest.skip("reportlab not available — cannot generate test PDF fixture")
c = canvas.Canvas(str(path), pagesize=letter)
if title:
c.setTitle(title)
if author:
c.setAuthor(author)
for i in range(pages):
c.drawString(72, 720, f"ObsiGate test PDF — page {i + 1} uniqueword{i}")
c.showPage()
c.save()
return path
@pytest.fixture
def pdf_dir(tmp_path: Path) -> Path:
"""A temp directory with a few PDFs of different shapes."""
d = tmp_path / "pdfs"
d.mkdir()
make_simple_pdf(d / "simple.pdf", pages=2, title="Simple Test", author="Bruno")
make_simple_pdf(d / "long.pdf", pages=3)
make_simple_pdf(d / "single.pdf", pages=1)
return d
# ── backend/pdf_reader.py ──────────────────────────────────────────────────
class TestPdfReader:
def test_extract_text_returns_text_with_keywords(self, pdf_dir: Path):
from backend.pdf_reader import extract_pdf_text
text = extract_pdf_text(pdf_dir / "simple.pdf")
assert "ObsiGate test PDF" in text
assert "uniqueword0" in text
assert "uniqueword1" in text
def test_extract_text_truncates_at_max_chars(self, pdf_dir: Path):
from backend.pdf_reader import extract_pdf_text
# tight max_chars truncates after the first page
text = extract_pdf_text(pdf_dir / "long.pdf", max_chars=10)
assert len(text) <= 50 # allow some slack; first page may have ~30 chars
def test_extract_text_missing_file_returns_empty(self, tmp_path: Path):
from backend.pdf_reader import extract_pdf_text
result = extract_pdf_text(tmp_path / "does-not-exist.pdf")
assert result == ""
def test_extract_text_corrupt_file_returns_empty(self, tmp_path: Path):
from backend.pdf_reader import extract_pdf_text
junk = tmp_path / "junk.pdf"
junk.write_bytes(b"not a real pdf, just some bytes %PDF-1.4 but no xref")
result = extract_pdf_text(junk)
# Should not raise; returns "" on failure
assert isinstance(result, str)
def test_extract_metadata_returns_pages_title_author(self, pdf_dir: Path):
from backend.pdf_reader import extract_pdf_metadata
info = extract_pdf_metadata(pdf_dir / "simple.pdf")
assert info["pages"] == 2
assert info["title"] in ("Simple Test", "") # metadata may be empty on some readers
assert isinstance(info["author"], str)
def test_extract_metadata_missing_file_returns_zeros(self, tmp_path: Path):
from backend.pdf_reader import extract_pdf_metadata
info = extract_pdf_metadata(tmp_path / "missing.pdf")
assert info == {"pages": 0, "title": "", "author": ""}
def test_extract_toc_returns_list(self, pdf_dir: Path):
from backend.pdf_reader import extract_pdf_toc
# simple PDFs (no outline) → empty list, no exception
toc = extract_pdf_toc(pdf_dir / "simple.pdf")
assert isinstance(toc, list)
# ── backend/indexer.py ─────────────────────────────────────────────────────
class TestPdfIndexing:
def test_pdf_in_supported_extensions(self):
from backend.indexer import SUPPORTED_EXTENSIONS
assert ".pdf" in SUPPORTED_EXTENSIONS
def test_scan_vault_picks_up_pdf_files(self, pdf_dir: Path, tmp_path: Path):
"""When a vault directory is scanned with .pdf files, they appear in files list.
Uses the public _scan_vault() helper directly — no global state needed.
"""
from backend.indexer import _scan_vault
vault_root = tmp_path / "vault"
vault_root.mkdir()
shutil.copy2(pdf_dir / "simple.pdf", vault_root / "a.pdf")
shutil.copy2(pdf_dir / "single.pdf", vault_root / "b.pdf")
result = _scan_vault("test-vault", str(vault_root), {"name": "test-vault", "path": str(vault_root)})
names = {f["path"] for f in result["files"]}
assert "a.pdf" in names
assert "b.pdf" in names
# The PDF content should have been extracted.
a_file = next(f for f in result["files"] if f["path"] == "a.pdf")
assert "ObsiGate test PDF" in (a_file.get("content") or "")
assert "uniqueword0" in (a_file.get("content") or "")
# ── Search filter `ext:` ───────────────────────────────────────────────────
class TestExtFilter:
"""The `ext:` filter is parsed in backend/search.py and applied in the
search pipeline. These tests verify the parsing + filter logic in isolation
so we don't depend on the full index state.
"""
def test_parse_ext_token(self):
from backend.search import _parse_advanced_query
parsed = _parse_advanced_query("hello ext:pdf world")
assert parsed["ext"] == "pdf"
assert "hello" in parsed["terms"]
assert "world" in parsed["terms"]
def test_parse_ext_token_with_dot(self):
from backend.search import _parse_advanced_query
parsed = _parse_advanced_query("ext:.md")
assert parsed["ext"] == "md"
def test_parse_no_ext_token(self):
from backend.search import _parse_advanced_query
parsed = _parse_advanced_query("hello world")
# ext key is initialized to None and stays None if no ext: token.
assert parsed.get("ext") in (None, "")
def test_parse_ext_token_lowercased(self):
from backend.search import _parse_advanced_query
parsed = _parse_advanced_query("ext:PDF")
assert parsed["ext"] == "pdf"