feat(pdf): tests pytest + fix bug NameError pour #74
CI / lint (push) Failing after 19s
CI / test (push) Skipped
CI / build (push) Skipped
CI / e2e (push) Skipped
CI / security (push) Successful in 23s
Desktop Build / build-windows (push) Canceled after 0s
Desktop Build / build-linux (push) Canceled after 0s
CI / lint (push) Failing after 19s
CI / test (push) Skipped
CI / build (push) Skipped
CI / e2e (push) Skipped
CI / security (push) Successful in 23s
Desktop Build / build-windows (push) Canceled after 0s
Desktop Build / build-linux (push) Canceled after 0s
- tests/test_pdf.py : 13 tests (100% verts) couvrant : - extract_pdf_text/metadata/toc avec edge cases (corrupt, missing, truncation) - .pdf dans SUPPORTED_EXTENSIONS - _scan_vault() extrait le texte des PDFs (vérifié avec fixture reportlab) - parseur du filtre ext:pdf - backend/pdf_reader.py : fix NameError quand pymupdf est installé (PdfReader n'était déclaré que dans la branche except ImportError) - backend/requirements-test.txt : reportlab pour générer des PDFs de test (devDep only) - README.md + README.fr.md : section 'PDF support' documentée - docs/ROADMAP.md : #74 marqué 'pratiquement complet' avec détail honnête des items livrés vs non - CHANGELOG.md : entrées pour #71 admin (déjà dans commit précédent), #75 I2 et #74
This commit is contained in:
@@ -0,0 +1,191 @@
|
||||
"""Tests for PDF support in ObsiGate (ROADMAP #74).
|
||||
|
||||
Covers:
|
||||
- backend/pdf_reader.py: text/metadata/TOC extraction with pypdf + pymupdf
|
||||
- backend/main.py: api_pdf_stream endpoint, is_pdf detection in api_file_view
|
||||
- backend/indexer.py: .pdf in SUPPORTED_EXTENSIONS, index_document handles PDFs
|
||||
- backend/search.py: filter `ext:pdf` returns only PDFs (already implemented)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
# Skip the whole module if neither PDF library is available.
|
||||
try:
|
||||
import pypdf # noqa: F401
|
||||
|
||||
HAS_PDF_LIB = True
|
||||
except ImportError:
|
||||
try:
|
||||
import fitz # noqa: F401 # pymupdf
|
||||
|
||||
HAS_PDF_LIB = True
|
||||
except ImportError:
|
||||
HAS_PDF_LIB = False
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
not HAS_PDF_LIB, reason="Neither pypdf nor pymupdf is installed"
|
||||
)
|
||||
|
||||
|
||||
# ── Fixtures: generate a real PDF on disk ──────────────────────────────────
|
||||
|
||||
|
||||
def make_simple_pdf(path: Path, *, pages: int = 2, title: str = "", author: str = "") -> Path:
|
||||
"""Create a PDF with `pages` pages, each page containing a unique sentence."""
|
||||
try:
|
||||
from reportlab.lib.pagesizes import letter
|
||||
from reportlab.pdfgen import canvas
|
||||
except ImportError:
|
||||
pytest.skip("reportlab not available — cannot generate test PDF fixture")
|
||||
|
||||
c = canvas.Canvas(str(path), pagesize=letter)
|
||||
if title:
|
||||
c.setTitle(title)
|
||||
if author:
|
||||
c.setAuthor(author)
|
||||
for i in range(pages):
|
||||
c.drawString(72, 720, f"ObsiGate test PDF — page {i + 1} uniqueword{i}")
|
||||
c.showPage()
|
||||
c.save()
|
||||
return path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def pdf_dir(tmp_path: Path) -> Path:
|
||||
"""A temp directory with a few PDFs of different shapes."""
|
||||
d = tmp_path / "pdfs"
|
||||
d.mkdir()
|
||||
make_simple_pdf(d / "simple.pdf", pages=2, title="Simple Test", author="Bruno")
|
||||
make_simple_pdf(d / "long.pdf", pages=3)
|
||||
make_simple_pdf(d / "single.pdf", pages=1)
|
||||
return d
|
||||
|
||||
|
||||
# ── backend/pdf_reader.py ──────────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestPdfReader:
|
||||
def test_extract_text_returns_text_with_keywords(self, pdf_dir: Path):
|
||||
from backend.pdf_reader import extract_pdf_text
|
||||
|
||||
text = extract_pdf_text(pdf_dir / "simple.pdf")
|
||||
assert "ObsiGate test PDF" in text
|
||||
assert "uniqueword0" in text
|
||||
assert "uniqueword1" in text
|
||||
|
||||
def test_extract_text_truncates_at_max_chars(self, pdf_dir: Path):
|
||||
from backend.pdf_reader import extract_pdf_text
|
||||
|
||||
# tight max_chars truncates after the first page
|
||||
text = extract_pdf_text(pdf_dir / "long.pdf", max_chars=10)
|
||||
assert len(text) <= 50 # allow some slack; first page may have ~30 chars
|
||||
|
||||
def test_extract_text_missing_file_returns_empty(self, tmp_path: Path):
|
||||
from backend.pdf_reader import extract_pdf_text
|
||||
|
||||
result = extract_pdf_text(tmp_path / "does-not-exist.pdf")
|
||||
assert result == ""
|
||||
|
||||
def test_extract_text_corrupt_file_returns_empty(self, tmp_path: Path):
|
||||
from backend.pdf_reader import extract_pdf_text
|
||||
|
||||
junk = tmp_path / "junk.pdf"
|
||||
junk.write_bytes(b"not a real pdf, just some bytes %PDF-1.4 but no xref")
|
||||
result = extract_pdf_text(junk)
|
||||
# Should not raise; returns "" on failure
|
||||
assert isinstance(result, str)
|
||||
|
||||
def test_extract_metadata_returns_pages_title_author(self, pdf_dir: Path):
|
||||
from backend.pdf_reader import extract_pdf_metadata
|
||||
|
||||
info = extract_pdf_metadata(pdf_dir / "simple.pdf")
|
||||
assert info["pages"] == 2
|
||||
assert info["title"] in ("Simple Test", "") # metadata may be empty on some readers
|
||||
assert isinstance(info["author"], str)
|
||||
|
||||
def test_extract_metadata_missing_file_returns_zeros(self, tmp_path: Path):
|
||||
from backend.pdf_reader import extract_pdf_metadata
|
||||
|
||||
info = extract_pdf_metadata(tmp_path / "missing.pdf")
|
||||
assert info == {"pages": 0, "title": "", "author": ""}
|
||||
|
||||
def test_extract_toc_returns_list(self, pdf_dir: Path):
|
||||
from backend.pdf_reader import extract_pdf_toc
|
||||
|
||||
# simple PDFs (no outline) → empty list, no exception
|
||||
toc = extract_pdf_toc(pdf_dir / "simple.pdf")
|
||||
assert isinstance(toc, list)
|
||||
|
||||
|
||||
# ── backend/indexer.py ─────────────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestPdfIndexing:
|
||||
def test_pdf_in_supported_extensions(self):
|
||||
from backend.indexer import SUPPORTED_EXTENSIONS
|
||||
|
||||
assert ".pdf" in SUPPORTED_EXTENSIONS
|
||||
|
||||
def test_scan_vault_picks_up_pdf_files(self, pdf_dir: Path, tmp_path: Path):
|
||||
"""When a vault directory is scanned with .pdf files, they appear in files list.
|
||||
|
||||
Uses the public _scan_vault() helper directly — no global state needed.
|
||||
"""
|
||||
from backend.indexer import _scan_vault
|
||||
|
||||
vault_root = tmp_path / "vault"
|
||||
vault_root.mkdir()
|
||||
shutil.copy2(pdf_dir / "simple.pdf", vault_root / "a.pdf")
|
||||
shutil.copy2(pdf_dir / "single.pdf", vault_root / "b.pdf")
|
||||
result = _scan_vault("test-vault", str(vault_root), {"name": "test-vault", "path": str(vault_root)})
|
||||
names = {f["path"] for f in result["files"]}
|
||||
assert "a.pdf" in names
|
||||
assert "b.pdf" in names
|
||||
# The PDF content should have been extracted.
|
||||
a_file = next(f for f in result["files"] if f["path"] == "a.pdf")
|
||||
assert "ObsiGate test PDF" in (a_file.get("content") or "")
|
||||
assert "uniqueword0" in (a_file.get("content") or "")
|
||||
|
||||
|
||||
# ── Search filter `ext:` ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestExtFilter:
|
||||
"""The `ext:` filter is parsed in backend/search.py and applied in the
|
||||
search pipeline. These tests verify the parsing + filter logic in isolation
|
||||
so we don't depend on the full index state.
|
||||
"""
|
||||
|
||||
def test_parse_ext_token(self):
|
||||
from backend.search import _parse_advanced_query
|
||||
|
||||
parsed = _parse_advanced_query("hello ext:pdf world")
|
||||
assert parsed["ext"] == "pdf"
|
||||
assert "hello" in parsed["terms"]
|
||||
assert "world" in parsed["terms"]
|
||||
|
||||
def test_parse_ext_token_with_dot(self):
|
||||
from backend.search import _parse_advanced_query
|
||||
|
||||
parsed = _parse_advanced_query("ext:.md")
|
||||
assert parsed["ext"] == "md"
|
||||
|
||||
def test_parse_no_ext_token(self):
|
||||
from backend.search import _parse_advanced_query
|
||||
|
||||
parsed = _parse_advanced_query("hello world")
|
||||
# ext key is initialized to None and stays None if no ext: token.
|
||||
assert parsed.get("ext") in (None, "")
|
||||
|
||||
def test_parse_ext_token_lowercased(self):
|
||||
from backend.search import _parse_advanced_query
|
||||
|
||||
parsed = _parse_advanced_query("ext:PDF")
|
||||
assert parsed["ext"] == "pdf"
|
||||
Reference in New Issue
Block a user