417 lines
15 KiB
Python
417 lines
15 KiB
Python
"""Tests for PDF support in ObsiGate (ROADMAP #74).
|
|
|
|
Covers:
|
|
- backend/pdf_reader.py: text/metadata/TOC extraction with pypdf + pymupdf
|
|
- backend/main.py: api_pdf_stream endpoint, is_pdf detection in api_file_view
|
|
- backend/indexer.py: .pdf in SUPPORTED_EXTENSIONS, index_document handles PDFs
|
|
- backend/search.py: filter `ext:pdf` returns only PDFs (already implemented)
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import shutil
|
|
import sys
|
|
import tempfile
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
# Skip the whole module if neither PDF library is available.
|
|
try:
|
|
import pypdf # noqa: F401
|
|
|
|
HAS_PDF_LIB = True
|
|
except ImportError:
|
|
try:
|
|
import fitz # noqa: F401 # pymupdf
|
|
|
|
HAS_PDF_LIB = True
|
|
except ImportError:
|
|
HAS_PDF_LIB = False
|
|
|
|
pytestmark = pytest.mark.skipif(
|
|
not HAS_PDF_LIB, reason="Neither pypdf nor pymupdf is installed"
|
|
)
|
|
|
|
|
|
# ── Fixtures: generate a real PDF on disk ──────────────────────────────────
|
|
|
|
|
|
def make_simple_pdf(path: Path, *, pages: int = 2, title: str = "", author: str = "") -> Path:
|
|
"""Create a PDF with `pages` pages, each page containing a unique sentence."""
|
|
try:
|
|
from reportlab.lib.pagesizes import letter
|
|
from reportlab.pdfgen import canvas
|
|
except ImportError:
|
|
pytest.skip("reportlab not available — cannot generate test PDF fixture")
|
|
|
|
c = canvas.Canvas(str(path), pagesize=letter)
|
|
if title:
|
|
c.setTitle(title)
|
|
if author:
|
|
c.setAuthor(author)
|
|
for i in range(pages):
|
|
c.drawString(72, 720, f"ObsiGate test PDF — page {i + 1} uniqueword{i}")
|
|
c.showPage()
|
|
c.save()
|
|
return path
|
|
|
|
|
|
@pytest.fixture
|
|
def pdf_dir(tmp_path: Path) -> Path:
|
|
"""A temp directory with a few PDFs of different shapes."""
|
|
d = tmp_path / "pdfs"
|
|
d.mkdir()
|
|
make_simple_pdf(d / "simple.pdf", pages=2, title="Simple Test", author="Bruno")
|
|
make_simple_pdf(d / "long.pdf", pages=3)
|
|
make_simple_pdf(d / "single.pdf", pages=1)
|
|
return d
|
|
|
|
|
|
# ── backend/pdf_reader.py ──────────────────────────────────────────────────
|
|
|
|
|
|
class TestPdfReader:
|
|
def test_extract_text_returns_text_with_keywords(self, pdf_dir: Path):
|
|
from backend.pdf_reader import extract_pdf_text
|
|
|
|
text = extract_pdf_text(pdf_dir / "simple.pdf")
|
|
assert "ObsiGate test PDF" in text
|
|
assert "uniqueword0" in text
|
|
assert "uniqueword1" in text
|
|
|
|
def test_extract_text_truncates_at_max_chars(self, pdf_dir: Path):
|
|
from backend.pdf_reader import extract_pdf_text
|
|
|
|
# tight max_chars truncates after the first page
|
|
text = extract_pdf_text(pdf_dir / "long.pdf", max_chars=10)
|
|
assert len(text) <= 50 # allow some slack; first page may have ~30 chars
|
|
|
|
def test_extract_text_missing_file_returns_empty(self, tmp_path: Path):
|
|
from backend.pdf_reader import extract_pdf_text
|
|
|
|
result = extract_pdf_text(tmp_path / "does-not-exist.pdf")
|
|
assert result == ""
|
|
|
|
def test_extract_text_corrupt_file_returns_empty(self, tmp_path: Path):
|
|
from backend.pdf_reader import extract_pdf_text
|
|
|
|
junk = tmp_path / "junk.pdf"
|
|
junk.write_bytes(b"not a real pdf, just some bytes %PDF-1.4 but no xref")
|
|
result = extract_pdf_text(junk)
|
|
# Should not raise; returns "" on failure
|
|
assert isinstance(result, str)
|
|
|
|
def test_extract_metadata_returns_pages_title_author(self, pdf_dir: Path):
|
|
from backend.pdf_reader import extract_pdf_metadata
|
|
|
|
info = extract_pdf_metadata(pdf_dir / "simple.pdf")
|
|
assert info["pages"] == 2
|
|
assert info["title"] in ("Simple Test", "") # metadata may be empty on some readers
|
|
assert isinstance(info["author"], str)
|
|
|
|
def test_extract_metadata_missing_file_returns_zeros(self, tmp_path: Path):
|
|
from backend.pdf_reader import extract_pdf_metadata
|
|
|
|
info = extract_pdf_metadata(tmp_path / "missing.pdf")
|
|
assert info == {"pages": 0, "title": "", "author": ""}
|
|
|
|
def test_extract_toc_returns_list(self, pdf_dir: Path):
|
|
from backend.pdf_reader import extract_pdf_toc
|
|
|
|
# simple PDFs (no outline) → empty list, no exception
|
|
toc = extract_pdf_toc(pdf_dir / "simple.pdf")
|
|
assert isinstance(toc, list)
|
|
|
|
|
|
class TestPdfIncrementalIndexing:
|
|
"""_index_single_file_sync (watcher/update path) must handle binary PDFs.
|
|
|
|
Regression: it used to read_text() every file, producing garbage for PDFs.
|
|
"""
|
|
|
|
def test_index_single_file_sync_extracts_pdf_text(self, pdf_dir: Path):
|
|
from backend.indexer import _index_single_file_sync
|
|
|
|
info = _index_single_file_sync("v", str(pdf_dir), str(pdf_dir / "simple.pdf"))
|
|
assert info is not None
|
|
assert info["extension"] == ".pdf"
|
|
assert "ObsiGate test PDF" in info["content"]
|
|
assert "uniqueword0" in info["content"]
|
|
|
|
def test_index_single_file_sync_pdf_title_from_metadata(self, pdf_dir: Path):
|
|
from backend.indexer import _index_single_file_sync
|
|
|
|
info = _index_single_file_sync("v", str(pdf_dir), str(pdf_dir / "simple.pdf"))
|
|
# title metadata was set at generation time
|
|
assert info["title"] == "Simple Test"
|
|
|
|
|
|
class TestPdfSizeLimit:
|
|
def test_oversized_pdf_skips_text_extraction(self, pdf_dir: Path, monkeypatch):
|
|
import backend.pdf_reader as pr
|
|
|
|
monkeypatch.setattr(pr, "PDF_MAX_SIZE_MB", 0) # everything is "too large"
|
|
text = pr.extract_pdf_text(pdf_dir / "simple.pdf")
|
|
assert text == ""
|
|
|
|
def test_normal_pdf_within_limit_extracts(self, pdf_dir: Path, monkeypatch):
|
|
import backend.pdf_reader as pr
|
|
|
|
monkeypatch.setattr(pr, "PDF_MAX_SIZE_MB", 50)
|
|
text = pr.extract_pdf_text(pdf_dir / "simple.pdf")
|
|
assert "ObsiGate test PDF" in text
|
|
|
|
|
|
# ── API: /pdf/stream (206 Range) + /pdf/info ───────────────────────────────
|
|
|
|
|
|
@pytest.fixture
|
|
def pdf_client():
|
|
"""TestClient (auth disabled) over a vault containing one generated PDF."""
|
|
tmp = Path(tempfile.mkdtemp())
|
|
vault = tmp / "PdfVault"
|
|
vault.mkdir()
|
|
make_simple_pdf(vault / "doc.pdf", pages=2, title="Doc Test", author="Bruno")
|
|
|
|
os.environ["VAULT_1_NAME"] = "PdfVault"
|
|
os.environ["VAULT_1_PATH"] = str(vault)
|
|
os.environ["OBSIGATE_AUTH_ENABLED"] = "false"
|
|
os.environ["OBSIGATE_WATCHER_ENABLED"] = "false"
|
|
|
|
import backend.main
|
|
backend.main._load_config = lambda: {"watcher_enabled": False}
|
|
|
|
from backend.indexer import build_index, index
|
|
for key in list(index.keys()):
|
|
del index[key]
|
|
|
|
import asyncio
|
|
loop = asyncio.new_event_loop()
|
|
asyncio.set_event_loop(loop)
|
|
loop.run_until_complete(build_index())
|
|
|
|
from fastapi.testclient import TestClient
|
|
client = TestClient(backend.main.app)
|
|
yield client
|
|
client.close()
|
|
shutil.rmtree(str(tmp), ignore_errors=True)
|
|
for k in ["VAULT_1_NAME", "VAULT_1_PATH", "OBSIGATE_AUTH_ENABLED", "OBSIGATE_WATCHER_ENABLED"]:
|
|
os.environ.pop(k, None)
|
|
|
|
|
|
class TestPdfStreamApi:
|
|
def test_stream_returns_200_application_pdf(self, pdf_client):
|
|
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf")
|
|
assert r.status_code == 200
|
|
assert r.headers["content-type"] == "application/pdf"
|
|
assert r.headers.get("accept-ranges") == "bytes"
|
|
assert r.content[:4] == b"%PDF"
|
|
|
|
def test_stream_full_range_returns_whole_file(self, pdf_client):
|
|
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
|
|
headers={"Range": "bytes=0-99999999"})
|
|
assert r.status_code == 206
|
|
assert r.headers["content-range"].startswith("bytes 0-")
|
|
assert r.content[:4] == b"%PDF"
|
|
|
|
def test_stream_partial_range(self, pdf_client):
|
|
full = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf").content
|
|
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
|
|
headers={"Range": "bytes=10-19"})
|
|
assert r.status_code == 206
|
|
assert r.content == full[10:20]
|
|
assert len(r.content) == 10
|
|
|
|
def test_stream_open_ended_range(self, pdf_client):
|
|
full = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf").content
|
|
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
|
|
headers={"Range": "bytes=100-"})
|
|
assert r.status_code == 206
|
|
assert r.content == full[100:]
|
|
|
|
def test_stream_bad_range_returns_416(self, pdf_client):
|
|
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
|
|
headers={"Range": "bytes=999999999-"})
|
|
assert r.status_code == 416
|
|
assert "bytes */" in r.headers.get("content-range", "")
|
|
|
|
def test_stream_rejects_non_pdf(self, pdf_client):
|
|
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.md")
|
|
assert r.status_code == 404 or r.status_code == 400
|
|
|
|
|
|
class TestPdfInfoApi:
|
|
def test_info_returns_metadata_without_content(self, pdf_client):
|
|
r = pdf_client.get("/api/file/PdfVault/pdf/info?path=doc.pdf")
|
|
assert r.status_code == 200
|
|
body = r.json()
|
|
assert body["pages"] == 2
|
|
assert body["title"] in ("Doc Test", "doc.pdf")
|
|
assert body["size_bytes"] > 0
|
|
assert "path" in body and body["vault"] == "PdfVault"
|
|
# no heavy content in the payload
|
|
assert "html" not in body
|
|
|
|
def test_info_missing_file_404(self, pdf_client):
|
|
r = pdf_client.get("/api/file/PdfVault/pdf/info?path=nope.pdf")
|
|
assert r.status_code == 404
|
|
|
|
def test_info_missing_non_pdf_404(self, pdf_client):
|
|
r = pdf_client.get("/api/file/PdfVault/pdf/info?path=readme.txt")
|
|
assert r.status_code == 404 # missing file checked before extension
|
|
|
|
|
|
# ── backend/indexer.py ─────────────────────────────────────────────────────
|
|
|
|
|
|
class TestPdfIndexing:
|
|
def test_pdf_in_supported_extensions(self):
|
|
from backend.indexer import SUPPORTED_EXTENSIONS
|
|
|
|
assert ".pdf" in SUPPORTED_EXTENSIONS
|
|
|
|
def test_scan_vault_picks_up_pdf_files(self, pdf_dir: Path, tmp_path: Path):
|
|
"""When a vault directory is scanned with .pdf files, they appear in files list.
|
|
|
|
Uses the public _scan_vault() helper directly — no global state needed.
|
|
BUG-040: text extraction is deferred, so the scan only carries metadata.
|
|
"""
|
|
from backend.indexer import _scan_vault
|
|
|
|
vault_root = tmp_path / "vault"
|
|
vault_root.mkdir()
|
|
shutil.copy2(pdf_dir / "simple.pdf", vault_root / "a.pdf")
|
|
shutil.copy2(pdf_dir / "single.pdf", vault_root / "b.pdf")
|
|
result = _scan_vault("test-vault", str(vault_root), {"name": "test-vault", "path": str(vault_root)})
|
|
names = {f["path"] for f in result["files"]}
|
|
assert "a.pdf" in names
|
|
assert "b.pdf" in names
|
|
# The scan defers the expensive text extraction.
|
|
a_file = next(f for f in result["files"] if f["path"] == "a.pdf")
|
|
assert a_file["content"] == ""
|
|
assert a_file["pdf_text_pending"] is True
|
|
|
|
|
|
class TestPdfLazyEnrichment:
|
|
"""BUG-040: PDF text is extracted in a deferred background pass."""
|
|
|
|
def test_scan_defers_pdf_text_extraction(self, pdf_dir: Path, tmp_path: Path):
|
|
from backend.indexer import _scan_vault
|
|
|
|
vault_root = tmp_path / "vault"
|
|
vault_root.mkdir()
|
|
shutil.copy2(pdf_dir / "simple.pdf", vault_root / "a.pdf")
|
|
result = _scan_vault("v", str(vault_root), {})
|
|
a_file = next(f for f in result["files"] if f["path"] == "a.pdf")
|
|
assert a_file["content"] == ""
|
|
assert a_file["content_preview"] == ""
|
|
assert a_file["pdf_text_pending"] is True
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_enrich_pdf_texts_fills_content_and_clears_flag(
|
|
self, pdf_dir: Path, tmp_path: Path
|
|
):
|
|
import backend.indexer as idx
|
|
|
|
vault_root = tmp_path / "vault"
|
|
vault_root.mkdir()
|
|
shutil.copy2(pdf_dir / "simple.pdf", vault_root / "a.pdf")
|
|
file_info = {
|
|
"path": "a.pdf",
|
|
"title": "a",
|
|
"tags": [],
|
|
"content": "",
|
|
"content_preview": "",
|
|
"size": 0,
|
|
"modified": "",
|
|
"extension": ".pdf",
|
|
"pdf_text_pending": True,
|
|
}
|
|
with idx._index_lock:
|
|
idx.index["LazyV"] = {
|
|
"files": [file_info],
|
|
"tags": {},
|
|
"path": str(vault_root),
|
|
"paths": [],
|
|
}
|
|
try:
|
|
count = await idx.enrich_pdf_texts("LazyV")
|
|
assert count == 1
|
|
assert "ObsiGate test PDF" in file_info["content"]
|
|
assert "uniqueword0" in file_info["content"]
|
|
assert file_info["content_preview"]
|
|
assert "pdf_text_pending" not in file_info
|
|
finally:
|
|
with idx._index_lock:
|
|
idx.index.pop("LazyV", None)
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_enrich_pdf_texts_skips_other_vaults(self, pdf_dir: Path, tmp_path: Path):
|
|
import backend.indexer as idx
|
|
|
|
vault_root = tmp_path / "vault"
|
|
vault_root.mkdir()
|
|
shutil.copy2(pdf_dir / "simple.pdf", vault_root / "a.pdf")
|
|
file_info = {
|
|
"path": "a.pdf",
|
|
"title": "a",
|
|
"tags": [],
|
|
"content": "",
|
|
"content_preview": "",
|
|
"size": 0,
|
|
"modified": "",
|
|
"extension": ".pdf",
|
|
"pdf_text_pending": True,
|
|
}
|
|
with idx._index_lock:
|
|
idx.index["OtherV"] = {
|
|
"files": [file_info],
|
|
"tags": {},
|
|
"path": str(vault_root),
|
|
"paths": [],
|
|
}
|
|
try:
|
|
assert await idx.enrich_pdf_texts("LazyV") == 0
|
|
assert file_info["content"] == ""
|
|
finally:
|
|
with idx._index_lock:
|
|
idx.index.pop("OtherV", None)
|
|
|
|
|
|
# ── Search filter `ext:` ───────────────────────────────────────────────────
|
|
|
|
|
|
class TestExtFilter:
|
|
"""The `ext:` filter is parsed in backend/search.py and applied in the
|
|
search pipeline. These tests verify the parsing + filter logic in isolation
|
|
so we don't depend on the full index state.
|
|
"""
|
|
|
|
def test_parse_ext_token(self):
|
|
from backend.search import _parse_advanced_query
|
|
|
|
parsed = _parse_advanced_query("hello ext:pdf world")
|
|
assert parsed["ext"] == "pdf"
|
|
assert "hello" in parsed["terms"]
|
|
assert "world" in parsed["terms"]
|
|
|
|
def test_parse_ext_token_with_dot(self):
|
|
from backend.search import _parse_advanced_query
|
|
|
|
parsed = _parse_advanced_query("ext:.md")
|
|
assert parsed["ext"] == "md"
|
|
|
|
def test_parse_no_ext_token(self):
|
|
from backend.search import _parse_advanced_query
|
|
|
|
parsed = _parse_advanced_query("hello world")
|
|
# ext key is initialized to None and stays None if no ext: token.
|
|
assert parsed.get("ext") in (None, "")
|
|
|
|
def test_parse_ext_token_lowercased(self):
|
|
from backend.search import _parse_advanced_query
|
|
|
|
parsed = _parse_advanced_query("ext:PDF")
|
|
assert parsed["ext"] == "pdf"
|