diff --git a/.env.example b/.env.example index e50ed96..2e95f74 100644 --- a/.env.example +++ b/.env.example @@ -33,6 +33,10 @@ OBSIGATE_ADMIN_PASSWORD=chab30 # Backup # OBSIGATE_BACKUP_DIR=.obsigate-backup +# PDF (ROADMAP #74) +# OBSIGATE_PDF_MAX_SIZE_MB=50 # PDFs more volumineux = texte non indexé +# OBSIGATE_PDF_EXTRACT_TIMEOUT=30 # secondes avant abandon de l'extraction + # ── AI Provider Configuration ── # Définir au moins un provider pour activer les fonctionnalités AI dans l'éditeur diff --git a/README.fr.md b/README.fr.md index 7f06a5f..1a5b5cd 100644 --- a/README.fr.md +++ b/README.fr.md @@ -272,6 +272,8 @@ Un compte **admin** connecté voit une icône 🛡️ dans le header : liste, cr | `OBSIGATE_REFRESH_TOKEN_TTL` | Durée de vie refresh token (secondes) | `2592000` | | `OBSIGATE_LOGIN_MAX_ATTEMPTS` | Tentatives de login max par IP | `10` | | `OBSIGATE_LOGIN_WINDOW_SECONDS` | Fenêtre de rate limiting (secondes) | `900` | +| `OBSIGATE_PDF_MAX_SIZE_MB` | Taille max des PDF extraits (text indexation) | `50` | +| `OBSIGATE_PDF_EXTRACT_TIMEOUT` | Timeout extraction PDF (secondes) | `30` | ### Volume pour la persistance @@ -614,8 +616,10 @@ Les opérateurs sont combinables : `tag:linux vault:IT ext:md serveur web`. ### Support PDF Les fichiers PDF de vos vaults s'affichent en ligne dans le navigateur via le visualiseur PDF natif (iframe + ``). +Le visualiseur streame le fichier via HTTP Range (206 Partial Content) — les gros PDF se chargent progressivement. Le texte est extrait à l'indexation (pypdf / pymupdf) — le contenu PDF est donc recherchable via la recherche full-text. Filtrez avec `ext:pdf` pour restreindre les résultats aux PDF. +Les métadonnées (pages, titre, auteur) sont disponibles via `GET /api/file/{vault}/pdf/info` sans transférer le document. **Limitations :** pas d'OCR (les PDF scannés ne sont pas recherchables), pas d'annotation, pas d'édition du PDF lui-même. diff --git a/README.md b/README.md index 69d442a..19ea58b 100644 --- a/README.md +++ b/README.md @@ -310,6 +310,8 @@ When an **admin** account is logged in, a 🛡️ icon appears in the header. Cl | `OBSIGATE_REFRESH_TOKEN_TTL` | Refresh token lifetime (seconds) | `2592000` | | `OBSIGATE_LOGIN_MAX_ATTEMPTS` | Max login attempts per IP | `10` | | `OBSIGATE_LOGIN_WINDOW_SECONDS` | Rate limiting window (seconds) | `900` | +| `OBSIGATE_PDF_MAX_SIZE_MB` | Max PDF size for text extraction | `50` | +| `OBSIGATE_PDF_EXTRACT_TIMEOUT` | PDF extraction timeout (seconds) | `30` | >All these variables are documented in `.env.example`. @@ -743,8 +745,10 @@ Extension filter examples: `ext:sh` for bash scripts, `ext:py` for Python script ### PDF support PDF files in your vaults are rendered inline in the browser via the native PDF viewer (iframe + ``). +The viewer streams the file over HTTP Range requests (206 Partial Content), so large PDFs load progressively. Text is extracted on indexing (pypdf / pymupdf) so PDF content is searchable via the full-text search. Filter with `ext:pdf` to restrict results to PDFs. +PDF metadata (pages, title, author) is available via `GET /api/file/{vault}/pdf/info` without transferring the document. **Limitations:** no OCR (scanned PDFs aren't searchable), no annotation, no editing of the PDF itself. diff --git a/backend/indexer.py b/backend/indexer.py index edbc326..8700c62 100644 --- a/backend/indexer.py +++ b/backend/indexer.py @@ -546,19 +546,27 @@ def _index_single_file_sync(vault_name: str, vault_path: str, file_path: str, va stat = fpath.stat() modified = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc).isoformat() - raw = fpath.read_text(encoding="utf-8", errors="replace") + # PDF handling — binary, must go through pdf_reader (same as _scan_vault) tags: list[str] = [] title = fpath.stem.replace("-", " ").replace("_", " ") - content_preview = raw[:200].strip() + if ext == ".pdf": + from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text + raw = extract_pdf_text(fpath, max_chars=SEARCH_CONTENT_LIMIT) + pdf_meta = extract_pdf_metadata(fpath) + title = pdf_meta.get("title") or title + content_preview = raw[:200].strip() + else: + raw = fpath.read_text(encoding="utf-8", errors="replace") + content_preview = raw[:200].strip() - if ext == ".md": - post = parse_markdown_file(raw) - tags = _extract_tags(post) - inline_tags = _extract_inline_tags(post.content) - tags = list(set(tags) | set(inline_tags)) - title = _extract_title(post, fpath) - content_preview = post.content[:200].strip() + if ext == ".md": + post = parse_markdown_file(raw) + tags = _extract_tags(post) + inline_tags = _extract_inline_tags(post.content) + tags = list(set(tags) | set(inline_tags)) + title = _extract_title(post, fpath) + content_preview = post.content[:200].strip() return { "path": str(relative).replace("\\", "/"), diff --git a/backend/main.py b/backend/main.py index 4143784..fb3fcc4 100644 --- a/backend/main.py +++ b/backend/main.py @@ -19,7 +19,7 @@ from typing import Any import frontmatter import mistune -from fastapi import Body, Depends, FastAPI, HTTPException, Query +from fastapi import Body, Depends, FastAPI, HTTPException, Query, Request from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse from fastapi.staticfiles import StaticFiles from pydantic import BaseModel, Field @@ -2727,8 +2727,17 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative @app.get("/api/file/{vault_name}/pdf/stream") -async def api_pdf_stream(vault_name: str, path: str = Query(...)): - """Stream a PDF file with Content-Type: application/pdf for inline browser viewing.""" +async def api_pdf_stream( + request: Request, + vault_name: str, + path: str = Query(...), + current_user=Depends(require_auth), +): + """Stream a PDF file with Content-Type: application/pdf for inline browser viewing. + + Supports HTTP Range requests (206 Partial Content) so browsers can + progressively render large PDFs in the native viewer. + """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) @@ -2740,9 +2749,100 @@ async def api_pdf_stream(vault_name: str, path: str = Query(...)): raise HTTPException(status_code=404, detail=f"File not found: {path}") if file_path.suffix.lower() != ".pdf": raise HTTPException(status_code=400, detail="Not a PDF file") - from fastapi.responses import FileResponse + + file_size = file_path.stat().st_size + range_header = request.headers.get("range") + + if range_header: + # Parse "bytes=start-end" (single range only; multi-range is not used by viewers) + m = re.match(r"bytes=(\d*)-(\d*)", range_header) + if not m: + raise HTTPException(status_code=416, + headers={"Content-Range": f"bytes */{file_size}"}) + start_s, end_s = m.group(1), m.group(2) + if start_s == "" and end_s == "": + raise HTTPException(status_code=416, + headers={"Content-Range": f"bytes */{file_size}"}) + if start_s == "": + # suffix range: last N bytes + length = min(int(end_s), file_size) + start = file_size - length + end = file_size - 1 + else: + start = int(start_s) + end = int(end_s) if end_s else file_size - 1 + end = min(end, file_size - 1) + if start > end or start >= file_size: + raise HTTPException(status_code=416, + headers={"Content-Range": f"bytes */{file_size}"}) + + chunk_size = end - start + 1 + + async def _partial(): + # Open + reads offloaded to threads (avoid blocking the event loop — ASYNC230) + f = await asyncio.to_thread(open, str(file_path), "rb") + try: + await asyncio.to_thread(f.seek, start) + remaining = chunk_size + while remaining > 0: + data = await asyncio.to_thread(f.read, min(64 * 1024, remaining)) + if not data: + break + remaining -= len(data) + yield data + finally: + await asyncio.to_thread(f.close) + + return StreamingResponse( + _partial(), + status_code=206, + media_type="application/pdf", + headers={ + "Content-Range": f"bytes {start}-{end}/{file_size}", + "Accept-Ranges": "bytes", + "Content-Length": str(chunk_size), + "Content-Disposition": f'inline; filename="{file_path.name}"', + }, + ) + return FileResponse(str(file_path), media_type="application/pdf", headers={ - "Content-Disposition": f"inline; filename=\"{file_path.name}\""}) + "Accept-Ranges": "bytes", + "Content-Disposition": f'inline; filename="{file_path.name}"'}) + + +@app.get("/api/file/{vault_name}/pdf/info") +async def api_pdf_info( + vault_name: str, + path: str = Query(..., description="Relative path to PDF file"), + current_user=Depends(require_auth), +): + """Return PDF metadata (pages, title, author, size) without the document content. + + Lets the UI display file info before loading a heavy PDF into the viewer. + """ + if not check_vault_access(vault_name, current_user): + raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") + vault_data = get_vault_data(vault_name) + if not vault_data: + raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") + vault_root = Path(vault_data["path"]) + file_path = _resolve_safe_path(vault_root, path) + if not file_path.exists() or not file_path.is_file(): + raise HTTPException(status_code=404, detail=f"File not found: {path}") + if file_path.suffix.lower() != ".pdf": + raise HTTPException(status_code=400, detail="Not a PDF file") + + from backend.pdf_reader import extract_pdf_metadata + meta = extract_pdf_metadata(file_path) + stat = file_path.stat() + return { + "vault": vault_name, + "path": path, + "pages": meta.get("pages", 0), + "title": meta.get("title") or file_path.name, + "author": meta.get("author", ""), + "size_bytes": stat.st_size, + } @app.get("/api/search", response_model=SearchResponse) diff --git a/backend/pdf_reader.py b/backend/pdf_reader.py index db9cab9..f7f5d1b 100644 --- a/backend/pdf_reader.py +++ b/backend/pdf_reader.py @@ -2,10 +2,19 @@ from __future__ import annotations import logging +import os +from concurrent.futures import ThreadPoolExecutor +from concurrent.futures import TimeoutError as FuturesTimeout from pathlib import Path logger = logging.getLogger(__name__) +# Configurable limits (see ROADMAP #74 G3): +# OBSIGATE_PDF_MAX_SIZE_MB — PDFs larger than this are not text-extracted (default 50) +# OBSIGATE_PDF_EXTRACT_TIMEOUT — seconds before extraction is abandoned (default 30) +PDF_MAX_SIZE_MB: int = int(os.environ.get("OBSIGATE_PDF_MAX_SIZE_MB", "50")) +PDF_EXTRACT_TIMEOUT: float = float(os.environ.get("OBSIGATE_PDF_EXTRACT_TIMEOUT", "30")) + PDF_READER: str = "pypdf" PdfReader = None # type: ignore try: @@ -20,15 +29,51 @@ except ImportError: logger.warning("No PDF reader available — install pypdf or pymupdf") +def pdf_exceeds_size_limit(file_path: Path) -> bool: + """True if the PDF is larger than OBSIGATE_PDF_MAX_SIZE_MB (skip text extraction).""" + try: + return file_path.stat().st_size > PDF_MAX_SIZE_MB * 1024 * 1024 + except OSError: + return False + + +def _run_with_timeout(fn, *args, timeout: float): + """Run a sync function in a worker thread with a hard timeout. + + A timed-out extraction leaves its worker thread running (daemon-style pool + is abandoned), but the request itself is freed — acceptable trade-off for + pathological PDFs on a self-hosted single-user server. + """ + executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="pdf-extract") + try: + future = executor.submit(fn, *args) + return future.result(timeout=timeout) + finally: + executor.shutdown(wait=False) + + def extract_pdf_text(file_path: Path, max_chars: int = 100000) -> str: - """Extract text from a PDF file. Returns empty string on failure.""" + """Extract text from a PDF file. Returns empty string on failure. + + Oversized PDFs (> OBSIGATE_PDF_MAX_SIZE_MB) and extractions exceeding + OBSIGATE_PDF_EXTRACT_TIMEOUT seconds return "" instead of blocking. + """ if PdfReader is None and PDF_READER == "pypdf": return "" + if pdf_exceeds_size_limit(file_path): + logger.info("PDF too large to index (> %d MB), skipping text extraction: %s", + PDF_MAX_SIZE_MB, file_path) + return "" try: if PDF_READER == "pymupdf": - return _extract_pymupdf(file_path, max_chars) + return _run_with_timeout(_extract_pymupdf, file_path, max_chars, + timeout=PDF_EXTRACT_TIMEOUT) else: - return _extract_pypdf(file_path, max_chars) + return _run_with_timeout(_extract_pypdf, file_path, max_chars, + timeout=PDF_EXTRACT_TIMEOUT) + except FuturesTimeout: + logger.warning("PDF text extraction timed out (%ss): %s", PDF_EXTRACT_TIMEOUT, file_path) + return "" except Exception as e: logger.warning("Failed to extract PDF text from %s: %s", file_path, e) return "" diff --git a/tests/test_pdf.py b/tests/test_pdf.py index d82aa68..c076d19 100644 --- a/tests/test_pdf.py +++ b/tests/test_pdf.py @@ -124,6 +124,144 @@ class TestPdfReader: assert isinstance(toc, list) +class TestPdfIncrementalIndexing: + """_index_single_file_sync (watcher/update path) must handle binary PDFs. + + Regression: it used to read_text() every file, producing garbage for PDFs. + """ + + def test_index_single_file_sync_extracts_pdf_text(self, pdf_dir: Path): + from backend.indexer import _index_single_file_sync + + info = _index_single_file_sync("v", str(pdf_dir), str(pdf_dir / "simple.pdf")) + assert info is not None + assert info["extension"] == ".pdf" + assert "ObsiGate test PDF" in info["content"] + assert "uniqueword0" in info["content"] + + def test_index_single_file_sync_pdf_title_from_metadata(self, pdf_dir: Path): + from backend.indexer import _index_single_file_sync + + info = _index_single_file_sync("v", str(pdf_dir), str(pdf_dir / "simple.pdf")) + # title metadata was set at generation time + assert info["title"] == "Simple Test" + + +class TestPdfSizeLimit: + def test_oversized_pdf_skips_text_extraction(self, pdf_dir: Path, monkeypatch): + import backend.pdf_reader as pr + + monkeypatch.setattr(pr, "PDF_MAX_SIZE_MB", 0) # everything is "too large" + text = pr.extract_pdf_text(pdf_dir / "simple.pdf") + assert text == "" + + def test_normal_pdf_within_limit_extracts(self, pdf_dir: Path, monkeypatch): + import backend.pdf_reader as pr + + monkeypatch.setattr(pr, "PDF_MAX_SIZE_MB", 50) + text = pr.extract_pdf_text(pdf_dir / "simple.pdf") + assert "ObsiGate test PDF" in text + + +# ── API: /pdf/stream (206 Range) + /pdf/info ─────────────────────────────── + + +@pytest.fixture +def pdf_client(): + """TestClient (auth disabled) over a vault containing one generated PDF.""" + tmp = Path(tempfile.mkdtemp()) + vault = tmp / "PdfVault" + vault.mkdir() + make_simple_pdf(vault / "doc.pdf", pages=2, title="Doc Test", author="Bruno") + + os.environ["VAULT_1_NAME"] = "PdfVault" + os.environ["VAULT_1_PATH"] = str(vault) + os.environ["OBSIGATE_AUTH_ENABLED"] = "false" + os.environ["OBSIGATE_WATCHER_ENABLED"] = "false" + + import backend.main + backend.main._load_config = lambda: {"watcher_enabled": False} + + from backend.indexer import build_index, index + for key in list(index.keys()): + del index[key] + + import asyncio + loop = asyncio.new_event_loop() + asyncio.set_event_loop(loop) + loop.run_until_complete(build_index()) + + from fastapi.testclient import TestClient + client = TestClient(backend.main.app) + yield client + client.close() + shutil.rmtree(str(tmp), ignore_errors=True) + for k in ["VAULT_1_NAME", "VAULT_1_PATH", "OBSIGATE_AUTH_ENABLED", "OBSIGATE_WATCHER_ENABLED"]: + os.environ.pop(k, None) + + +class TestPdfStreamApi: + def test_stream_returns_200_application_pdf(self, pdf_client): + r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf") + assert r.status_code == 200 + assert r.headers["content-type"] == "application/pdf" + assert r.headers.get("accept-ranges") == "bytes" + assert r.content[:4] == b"%PDF" + + def test_stream_full_range_returns_whole_file(self, pdf_client): + r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf", + headers={"Range": "bytes=0-99999999"}) + assert r.status_code == 206 + assert r.headers["content-range"].startswith("bytes 0-") + assert r.content[:4] == b"%PDF" + + def test_stream_partial_range(self, pdf_client): + full = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf").content + r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf", + headers={"Range": "bytes=10-19"}) + assert r.status_code == 206 + assert r.content == full[10:20] + assert len(r.content) == 10 + + def test_stream_open_ended_range(self, pdf_client): + full = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf").content + r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf", + headers={"Range": "bytes=100-"}) + assert r.status_code == 206 + assert r.content == full[100:] + + def test_stream_bad_range_returns_416(self, pdf_client): + r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf", + headers={"Range": "bytes=999999999-"}) + assert r.status_code == 416 + assert "bytes */" in r.headers.get("content-range", "") + + def test_stream_rejects_non_pdf(self, pdf_client): + r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.md") + assert r.status_code == 404 or r.status_code == 400 + + +class TestPdfInfoApi: + def test_info_returns_metadata_without_content(self, pdf_client): + r = pdf_client.get("/api/file/PdfVault/pdf/info?path=doc.pdf") + assert r.status_code == 200 + body = r.json() + assert body["pages"] == 2 + assert body["title"] in ("Doc Test", "doc.pdf") + assert body["size_bytes"] > 0 + assert "path" in body and body["vault"] == "PdfVault" + # no heavy content in the payload + assert "html" not in body + + def test_info_missing_file_404(self, pdf_client): + r = pdf_client.get("/api/file/PdfVault/pdf/info?path=nope.pdf") + assert r.status_code == 404 + + def test_info_missing_non_pdf_404(self, pdf_client): + r = pdf_client.get("/api/file/PdfVault/pdf/info?path=readme.txt") + assert r.status_code == 404 # missing file checked before extension + + # ── backend/indexer.py ─────────────────────────────────────────────────────