feat(pdf): #74 au complet — fix auth 500 stream + endpoint pdf/info + HTTP Range 206 + indexation incrementale PDF + config taille/timeout
This commit is contained in:
@@ -33,6 +33,10 @@ OBSIGATE_ADMIN_PASSWORD=chab30
|
||||
# Backup
|
||||
# OBSIGATE_BACKUP_DIR=.obsigate-backup
|
||||
|
||||
# PDF (ROADMAP #74)
|
||||
# OBSIGATE_PDF_MAX_SIZE_MB=50 # PDFs more volumineux = texte non indexé
|
||||
# OBSIGATE_PDF_EXTRACT_TIMEOUT=30 # secondes avant abandon de l'extraction
|
||||
|
||||
# ── AI Provider Configuration ──
|
||||
# Définir au moins un provider pour activer les fonctionnalités AI dans l'éditeur
|
||||
|
||||
|
||||
@@ -272,6 +272,8 @@ Un compte **admin** connecté voit une icône 🛡️ dans le header : liste, cr
|
||||
| `OBSIGATE_REFRESH_TOKEN_TTL` | Durée de vie refresh token (secondes) | `2592000` |
|
||||
| `OBSIGATE_LOGIN_MAX_ATTEMPTS` | Tentatives de login max par IP | `10` |
|
||||
| `OBSIGATE_LOGIN_WINDOW_SECONDS` | Fenêtre de rate limiting (secondes) | `900` |
|
||||
| `OBSIGATE_PDF_MAX_SIZE_MB` | Taille max des PDF extraits (text indexation) | `50` |
|
||||
| `OBSIGATE_PDF_EXTRACT_TIMEOUT` | Timeout extraction PDF (secondes) | `30` |
|
||||
|
||||
### Volume pour la persistance
|
||||
|
||||
@@ -614,8 +616,10 @@ Les opérateurs sont combinables : `tag:linux vault:IT ext:md serveur web`.
|
||||
### Support PDF
|
||||
|
||||
Les fichiers PDF de vos vaults s'affichent en ligne dans le navigateur via le visualiseur PDF natif (iframe + `<embed>`).
|
||||
Le visualiseur streame le fichier via HTTP Range (206 Partial Content) — les gros PDF se chargent progressivement.
|
||||
Le texte est extrait à l'indexation (pypdf / pymupdf) — le contenu PDF est donc recherchable via la recherche full-text.
|
||||
Filtrez avec `ext:pdf` pour restreindre les résultats aux PDF.
|
||||
Les métadonnées (pages, titre, auteur) sont disponibles via `GET /api/file/{vault}/pdf/info` sans transférer le document.
|
||||
|
||||
**Limitations :** pas d'OCR (les PDF scannés ne sont pas recherchables), pas d'annotation, pas d'édition du PDF lui-même.
|
||||
|
||||
|
||||
@@ -310,6 +310,8 @@ When an **admin** account is logged in, a 🛡️ icon appears in the header. Cl
|
||||
| `OBSIGATE_REFRESH_TOKEN_TTL` | Refresh token lifetime (seconds) | `2592000` |
|
||||
| `OBSIGATE_LOGIN_MAX_ATTEMPTS` | Max login attempts per IP | `10` |
|
||||
| `OBSIGATE_LOGIN_WINDOW_SECONDS` | Rate limiting window (seconds) | `900` |
|
||||
| `OBSIGATE_PDF_MAX_SIZE_MB` | Max PDF size for text extraction | `50` |
|
||||
| `OBSIGATE_PDF_EXTRACT_TIMEOUT` | PDF extraction timeout (seconds) | `30` |
|
||||
|
||||
>All these variables are documented in `.env.example`.
|
||||
|
||||
@@ -743,8 +745,10 @@ Extension filter examples: `ext:sh` for bash scripts, `ext:py` for Python script
|
||||
### PDF support
|
||||
|
||||
PDF files in your vaults are rendered inline in the browser via the native PDF viewer (iframe + `<embed>`).
|
||||
The viewer streams the file over HTTP Range requests (206 Partial Content), so large PDFs load progressively.
|
||||
Text is extracted on indexing (pypdf / pymupdf) so PDF content is searchable via the full-text search.
|
||||
Filter with `ext:pdf` to restrict results to PDFs.
|
||||
PDF metadata (pages, title, author) is available via `GET /api/file/{vault}/pdf/info` without transferring the document.
|
||||
|
||||
**Limitations:** no OCR (scanned PDFs aren't searchable), no annotation, no editing of the PDF itself.
|
||||
|
||||
|
||||
+9
-1
@@ -546,10 +546,18 @@ def _index_single_file_sync(vault_name: str, vault_path: str, file_path: str, va
|
||||
|
||||
stat = fpath.stat()
|
||||
modified = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc).isoformat()
|
||||
raw = fpath.read_text(encoding="utf-8", errors="replace")
|
||||
|
||||
# PDF handling — binary, must go through pdf_reader (same as _scan_vault)
|
||||
tags: list[str] = []
|
||||
title = fpath.stem.replace("-", " ").replace("_", " ")
|
||||
if ext == ".pdf":
|
||||
from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text
|
||||
raw = extract_pdf_text(fpath, max_chars=SEARCH_CONTENT_LIMIT)
|
||||
pdf_meta = extract_pdf_metadata(fpath)
|
||||
title = pdf_meta.get("title") or title
|
||||
content_preview = raw[:200].strip()
|
||||
else:
|
||||
raw = fpath.read_text(encoding="utf-8", errors="replace")
|
||||
content_preview = raw[:200].strip()
|
||||
|
||||
if ext == ".md":
|
||||
|
||||
+105
-5
@@ -19,7 +19,7 @@ from typing import Any
|
||||
|
||||
import frontmatter
|
||||
import mistune
|
||||
from fastapi import Body, Depends, FastAPI, HTTPException, Query
|
||||
from fastapi import Body, Depends, FastAPI, HTTPException, Query, Request
|
||||
from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from pydantic import BaseModel, Field
|
||||
@@ -2727,8 +2727,17 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
|
||||
|
||||
|
||||
@app.get("/api/file/{vault_name}/pdf/stream")
|
||||
async def api_pdf_stream(vault_name: str, path: str = Query(...)):
|
||||
"""Stream a PDF file with Content-Type: application/pdf for inline browser viewing."""
|
||||
async def api_pdf_stream(
|
||||
request: Request,
|
||||
vault_name: str,
|
||||
path: str = Query(...),
|
||||
current_user=Depends(require_auth),
|
||||
):
|
||||
"""Stream a PDF file with Content-Type: application/pdf for inline browser viewing.
|
||||
|
||||
Supports HTTP Range requests (206 Partial Content) so browsers can
|
||||
progressively render large PDFs in the native viewer.
|
||||
"""
|
||||
if not check_vault_access(vault_name, current_user):
|
||||
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
||||
vault_data = get_vault_data(vault_name)
|
||||
@@ -2740,9 +2749,100 @@ async def api_pdf_stream(vault_name: str, path: str = Query(...)):
|
||||
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
||||
if file_path.suffix.lower() != ".pdf":
|
||||
raise HTTPException(status_code=400, detail="Not a PDF file")
|
||||
from fastapi.responses import FileResponse
|
||||
|
||||
file_size = file_path.stat().st_size
|
||||
range_header = request.headers.get("range")
|
||||
|
||||
if range_header:
|
||||
# Parse "bytes=start-end" (single range only; multi-range is not used by viewers)
|
||||
m = re.match(r"bytes=(\d*)-(\d*)", range_header)
|
||||
if not m:
|
||||
raise HTTPException(status_code=416,
|
||||
headers={"Content-Range": f"bytes */{file_size}"})
|
||||
start_s, end_s = m.group(1), m.group(2)
|
||||
if start_s == "" and end_s == "":
|
||||
raise HTTPException(status_code=416,
|
||||
headers={"Content-Range": f"bytes */{file_size}"})
|
||||
if start_s == "":
|
||||
# suffix range: last N bytes
|
||||
length = min(int(end_s), file_size)
|
||||
start = file_size - length
|
||||
end = file_size - 1
|
||||
else:
|
||||
start = int(start_s)
|
||||
end = int(end_s) if end_s else file_size - 1
|
||||
end = min(end, file_size - 1)
|
||||
if start > end or start >= file_size:
|
||||
raise HTTPException(status_code=416,
|
||||
headers={"Content-Range": f"bytes */{file_size}"})
|
||||
|
||||
chunk_size = end - start + 1
|
||||
|
||||
async def _partial():
|
||||
# Open + reads offloaded to threads (avoid blocking the event loop — ASYNC230)
|
||||
f = await asyncio.to_thread(open, str(file_path), "rb")
|
||||
try:
|
||||
await asyncio.to_thread(f.seek, start)
|
||||
remaining = chunk_size
|
||||
while remaining > 0:
|
||||
data = await asyncio.to_thread(f.read, min(64 * 1024, remaining))
|
||||
if not data:
|
||||
break
|
||||
remaining -= len(data)
|
||||
yield data
|
||||
finally:
|
||||
await asyncio.to_thread(f.close)
|
||||
|
||||
return StreamingResponse(
|
||||
_partial(),
|
||||
status_code=206,
|
||||
media_type="application/pdf",
|
||||
headers={
|
||||
"Content-Range": f"bytes {start}-{end}/{file_size}",
|
||||
"Accept-Ranges": "bytes",
|
||||
"Content-Length": str(chunk_size),
|
||||
"Content-Disposition": f'inline; filename="{file_path.name}"',
|
||||
},
|
||||
)
|
||||
|
||||
return FileResponse(str(file_path), media_type="application/pdf", headers={
|
||||
"Content-Disposition": f"inline; filename=\"{file_path.name}\""})
|
||||
"Accept-Ranges": "bytes",
|
||||
"Content-Disposition": f'inline; filename="{file_path.name}"'})
|
||||
|
||||
|
||||
@app.get("/api/file/{vault_name}/pdf/info")
|
||||
async def api_pdf_info(
|
||||
vault_name: str,
|
||||
path: str = Query(..., description="Relative path to PDF file"),
|
||||
current_user=Depends(require_auth),
|
||||
):
|
||||
"""Return PDF metadata (pages, title, author, size) without the document content.
|
||||
|
||||
Lets the UI display file info before loading a heavy PDF into the viewer.
|
||||
"""
|
||||
if not check_vault_access(vault_name, current_user):
|
||||
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
||||
vault_data = get_vault_data(vault_name)
|
||||
if not vault_data:
|
||||
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
||||
vault_root = Path(vault_data["path"])
|
||||
file_path = _resolve_safe_path(vault_root, path)
|
||||
if not file_path.exists() or not file_path.is_file():
|
||||
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
||||
if file_path.suffix.lower() != ".pdf":
|
||||
raise HTTPException(status_code=400, detail="Not a PDF file")
|
||||
|
||||
from backend.pdf_reader import extract_pdf_metadata
|
||||
meta = extract_pdf_metadata(file_path)
|
||||
stat = file_path.stat()
|
||||
return {
|
||||
"vault": vault_name,
|
||||
"path": path,
|
||||
"pages": meta.get("pages", 0),
|
||||
"title": meta.get("title") or file_path.name,
|
||||
"author": meta.get("author", ""),
|
||||
"size_bytes": stat.st_size,
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/search", response_model=SearchResponse)
|
||||
|
||||
+48
-3
@@ -2,10 +2,19 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from concurrent.futures import TimeoutError as FuturesTimeout
|
||||
from pathlib import Path
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Configurable limits (see ROADMAP #74 G3):
|
||||
# OBSIGATE_PDF_MAX_SIZE_MB — PDFs larger than this are not text-extracted (default 50)
|
||||
# OBSIGATE_PDF_EXTRACT_TIMEOUT — seconds before extraction is abandoned (default 30)
|
||||
PDF_MAX_SIZE_MB: int = int(os.environ.get("OBSIGATE_PDF_MAX_SIZE_MB", "50"))
|
||||
PDF_EXTRACT_TIMEOUT: float = float(os.environ.get("OBSIGATE_PDF_EXTRACT_TIMEOUT", "30"))
|
||||
|
||||
PDF_READER: str = "pypdf"
|
||||
PdfReader = None # type: ignore
|
||||
try:
|
||||
@@ -20,15 +29,51 @@ except ImportError:
|
||||
logger.warning("No PDF reader available — install pypdf or pymupdf")
|
||||
|
||||
|
||||
def pdf_exceeds_size_limit(file_path: Path) -> bool:
|
||||
"""True if the PDF is larger than OBSIGATE_PDF_MAX_SIZE_MB (skip text extraction)."""
|
||||
try:
|
||||
return file_path.stat().st_size > PDF_MAX_SIZE_MB * 1024 * 1024
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
def _run_with_timeout(fn, *args, timeout: float):
|
||||
"""Run a sync function in a worker thread with a hard timeout.
|
||||
|
||||
A timed-out extraction leaves its worker thread running (daemon-style pool
|
||||
is abandoned), but the request itself is freed — acceptable trade-off for
|
||||
pathological PDFs on a self-hosted single-user server.
|
||||
"""
|
||||
executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="pdf-extract")
|
||||
try:
|
||||
future = executor.submit(fn, *args)
|
||||
return future.result(timeout=timeout)
|
||||
finally:
|
||||
executor.shutdown(wait=False)
|
||||
|
||||
|
||||
def extract_pdf_text(file_path: Path, max_chars: int = 100000) -> str:
|
||||
"""Extract text from a PDF file. Returns empty string on failure."""
|
||||
"""Extract text from a PDF file. Returns empty string on failure.
|
||||
|
||||
Oversized PDFs (> OBSIGATE_PDF_MAX_SIZE_MB) and extractions exceeding
|
||||
OBSIGATE_PDF_EXTRACT_TIMEOUT seconds return "" instead of blocking.
|
||||
"""
|
||||
if PdfReader is None and PDF_READER == "pypdf":
|
||||
return ""
|
||||
if pdf_exceeds_size_limit(file_path):
|
||||
logger.info("PDF too large to index (> %d MB), skipping text extraction: %s",
|
||||
PDF_MAX_SIZE_MB, file_path)
|
||||
return ""
|
||||
try:
|
||||
if PDF_READER == "pymupdf":
|
||||
return _extract_pymupdf(file_path, max_chars)
|
||||
return _run_with_timeout(_extract_pymupdf, file_path, max_chars,
|
||||
timeout=PDF_EXTRACT_TIMEOUT)
|
||||
else:
|
||||
return _extract_pypdf(file_path, max_chars)
|
||||
return _run_with_timeout(_extract_pypdf, file_path, max_chars,
|
||||
timeout=PDF_EXTRACT_TIMEOUT)
|
||||
except FuturesTimeout:
|
||||
logger.warning("PDF text extraction timed out (%ss): %s", PDF_EXTRACT_TIMEOUT, file_path)
|
||||
return ""
|
||||
except Exception as e:
|
||||
logger.warning("Failed to extract PDF text from %s: %s", file_path, e)
|
||||
return ""
|
||||
|
||||
@@ -124,6 +124,144 @@ class TestPdfReader:
|
||||
assert isinstance(toc, list)
|
||||
|
||||
|
||||
class TestPdfIncrementalIndexing:
|
||||
"""_index_single_file_sync (watcher/update path) must handle binary PDFs.
|
||||
|
||||
Regression: it used to read_text() every file, producing garbage for PDFs.
|
||||
"""
|
||||
|
||||
def test_index_single_file_sync_extracts_pdf_text(self, pdf_dir: Path):
|
||||
from backend.indexer import _index_single_file_sync
|
||||
|
||||
info = _index_single_file_sync("v", str(pdf_dir), str(pdf_dir / "simple.pdf"))
|
||||
assert info is not None
|
||||
assert info["extension"] == ".pdf"
|
||||
assert "ObsiGate test PDF" in info["content"]
|
||||
assert "uniqueword0" in info["content"]
|
||||
|
||||
def test_index_single_file_sync_pdf_title_from_metadata(self, pdf_dir: Path):
|
||||
from backend.indexer import _index_single_file_sync
|
||||
|
||||
info = _index_single_file_sync("v", str(pdf_dir), str(pdf_dir / "simple.pdf"))
|
||||
# title metadata was set at generation time
|
||||
assert info["title"] == "Simple Test"
|
||||
|
||||
|
||||
class TestPdfSizeLimit:
|
||||
def test_oversized_pdf_skips_text_extraction(self, pdf_dir: Path, monkeypatch):
|
||||
import backend.pdf_reader as pr
|
||||
|
||||
monkeypatch.setattr(pr, "PDF_MAX_SIZE_MB", 0) # everything is "too large"
|
||||
text = pr.extract_pdf_text(pdf_dir / "simple.pdf")
|
||||
assert text == ""
|
||||
|
||||
def test_normal_pdf_within_limit_extracts(self, pdf_dir: Path, monkeypatch):
|
||||
import backend.pdf_reader as pr
|
||||
|
||||
monkeypatch.setattr(pr, "PDF_MAX_SIZE_MB", 50)
|
||||
text = pr.extract_pdf_text(pdf_dir / "simple.pdf")
|
||||
assert "ObsiGate test PDF" in text
|
||||
|
||||
|
||||
# ── API: /pdf/stream (206 Range) + /pdf/info ───────────────────────────────
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def pdf_client():
|
||||
"""TestClient (auth disabled) over a vault containing one generated PDF."""
|
||||
tmp = Path(tempfile.mkdtemp())
|
||||
vault = tmp / "PdfVault"
|
||||
vault.mkdir()
|
||||
make_simple_pdf(vault / "doc.pdf", pages=2, title="Doc Test", author="Bruno")
|
||||
|
||||
os.environ["VAULT_1_NAME"] = "PdfVault"
|
||||
os.environ["VAULT_1_PATH"] = str(vault)
|
||||
os.environ["OBSIGATE_AUTH_ENABLED"] = "false"
|
||||
os.environ["OBSIGATE_WATCHER_ENABLED"] = "false"
|
||||
|
||||
import backend.main
|
||||
backend.main._load_config = lambda: {"watcher_enabled": False}
|
||||
|
||||
from backend.indexer import build_index, index
|
||||
for key in list(index.keys()):
|
||||
del index[key]
|
||||
|
||||
import asyncio
|
||||
loop = asyncio.new_event_loop()
|
||||
asyncio.set_event_loop(loop)
|
||||
loop.run_until_complete(build_index())
|
||||
|
||||
from fastapi.testclient import TestClient
|
||||
client = TestClient(backend.main.app)
|
||||
yield client
|
||||
client.close()
|
||||
shutil.rmtree(str(tmp), ignore_errors=True)
|
||||
for k in ["VAULT_1_NAME", "VAULT_1_PATH", "OBSIGATE_AUTH_ENABLED", "OBSIGATE_WATCHER_ENABLED"]:
|
||||
os.environ.pop(k, None)
|
||||
|
||||
|
||||
class TestPdfStreamApi:
|
||||
def test_stream_returns_200_application_pdf(self, pdf_client):
|
||||
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf")
|
||||
assert r.status_code == 200
|
||||
assert r.headers["content-type"] == "application/pdf"
|
||||
assert r.headers.get("accept-ranges") == "bytes"
|
||||
assert r.content[:4] == b"%PDF"
|
||||
|
||||
def test_stream_full_range_returns_whole_file(self, pdf_client):
|
||||
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
|
||||
headers={"Range": "bytes=0-99999999"})
|
||||
assert r.status_code == 206
|
||||
assert r.headers["content-range"].startswith("bytes 0-")
|
||||
assert r.content[:4] == b"%PDF"
|
||||
|
||||
def test_stream_partial_range(self, pdf_client):
|
||||
full = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf").content
|
||||
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
|
||||
headers={"Range": "bytes=10-19"})
|
||||
assert r.status_code == 206
|
||||
assert r.content == full[10:20]
|
||||
assert len(r.content) == 10
|
||||
|
||||
def test_stream_open_ended_range(self, pdf_client):
|
||||
full = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf").content
|
||||
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
|
||||
headers={"Range": "bytes=100-"})
|
||||
assert r.status_code == 206
|
||||
assert r.content == full[100:]
|
||||
|
||||
def test_stream_bad_range_returns_416(self, pdf_client):
|
||||
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
|
||||
headers={"Range": "bytes=999999999-"})
|
||||
assert r.status_code == 416
|
||||
assert "bytes */" in r.headers.get("content-range", "")
|
||||
|
||||
def test_stream_rejects_non_pdf(self, pdf_client):
|
||||
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.md")
|
||||
assert r.status_code == 404 or r.status_code == 400
|
||||
|
||||
|
||||
class TestPdfInfoApi:
|
||||
def test_info_returns_metadata_without_content(self, pdf_client):
|
||||
r = pdf_client.get("/api/file/PdfVault/pdf/info?path=doc.pdf")
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert body["pages"] == 2
|
||||
assert body["title"] in ("Doc Test", "doc.pdf")
|
||||
assert body["size_bytes"] > 0
|
||||
assert "path" in body and body["vault"] == "PdfVault"
|
||||
# no heavy content in the payload
|
||||
assert "html" not in body
|
||||
|
||||
def test_info_missing_file_404(self, pdf_client):
|
||||
r = pdf_client.get("/api/file/PdfVault/pdf/info?path=nope.pdf")
|
||||
assert r.status_code == 404
|
||||
|
||||
def test_info_missing_non_pdf_404(self, pdf_client):
|
||||
r = pdf_client.get("/api/file/PdfVault/pdf/info?path=readme.txt")
|
||||
assert r.status_code == 404 # missing file checked before extension
|
||||
|
||||
|
||||
# ── backend/indexer.py ─────────────────────────────────────────────────────
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user