feat(pdf): #74 au complet — fix auth 500 stream + endpoint pdf/info + HTTP Range 206 + indexation incrementale PDF + config taille/timeout

This commit is contained in:
2026-09-07 23:24:13 -04:00
parent f37d5fda3c
commit 7042307955
7 changed files with 320 additions and 17 deletions
+4
View File
@@ -33,6 +33,10 @@ OBSIGATE_ADMIN_PASSWORD=chab30
# Backup
# OBSIGATE_BACKUP_DIR=.obsigate-backup
# PDF (ROADMAP #74)
# OBSIGATE_PDF_MAX_SIZE_MB=50 # PDFs more volumineux = texte non indexé
# OBSIGATE_PDF_EXTRACT_TIMEOUT=30 # secondes avant abandon de l'extraction
# ── AI Provider Configuration ──
# Définir au moins un provider pour activer les fonctionnalités AI dans l'éditeur
+4
View File
@@ -272,6 +272,8 @@ Un compte **admin** connecté voit une icône 🛡️ dans le header : liste, cr
| `OBSIGATE_REFRESH_TOKEN_TTL` | Durée de vie refresh token (secondes) | `2592000` |
| `OBSIGATE_LOGIN_MAX_ATTEMPTS` | Tentatives de login max par IP | `10` |
| `OBSIGATE_LOGIN_WINDOW_SECONDS` | Fenêtre de rate limiting (secondes) | `900` |
| `OBSIGATE_PDF_MAX_SIZE_MB` | Taille max des PDF extraits (text indexation) | `50` |
| `OBSIGATE_PDF_EXTRACT_TIMEOUT` | Timeout extraction PDF (secondes) | `30` |
### Volume pour la persistance
@@ -614,8 +616,10 @@ Les opérateurs sont combinables : `tag:linux vault:IT ext:md serveur web`.
### Support PDF
Les fichiers PDF de vos vaults s'affichent en ligne dans le navigateur via le visualiseur PDF natif (iframe + `<embed>`).
Le visualiseur streame le fichier via HTTP Range (206 Partial Content) — les gros PDF se chargent progressivement.
Le texte est extrait à l'indexation (pypdf / pymupdf) — le contenu PDF est donc recherchable via la recherche full-text.
Filtrez avec `ext:pdf` pour restreindre les résultats aux PDF.
Les métadonnées (pages, titre, auteur) sont disponibles via `GET /api/file/{vault}/pdf/info` sans transférer le document.
**Limitations :** pas d'OCR (les PDF scannés ne sont pas recherchables), pas d'annotation, pas d'édition du PDF lui-même.
+4
View File
@@ -310,6 +310,8 @@ When an **admin** account is logged in, a 🛡️ icon appears in the header. Cl
| `OBSIGATE_REFRESH_TOKEN_TTL` | Refresh token lifetime (seconds) | `2592000` |
| `OBSIGATE_LOGIN_MAX_ATTEMPTS` | Max login attempts per IP | `10` |
| `OBSIGATE_LOGIN_WINDOW_SECONDS` | Rate limiting window (seconds) | `900` |
| `OBSIGATE_PDF_MAX_SIZE_MB` | Max PDF size for text extraction | `50` |
| `OBSIGATE_PDF_EXTRACT_TIMEOUT` | PDF extraction timeout (seconds) | `30` |
>All these variables are documented in `.env.example`.
@@ -743,8 +745,10 @@ Extension filter examples: `ext:sh` for bash scripts, `ext:py` for Python script
### PDF support
PDF files in your vaults are rendered inline in the browser via the native PDF viewer (iframe + `<embed>`).
The viewer streams the file over HTTP Range requests (206 Partial Content), so large PDFs load progressively.
Text is extracted on indexing (pypdf / pymupdf) so PDF content is searchable via the full-text search.
Filter with `ext:pdf` to restrict results to PDFs.
PDF metadata (pages, title, author) is available via `GET /api/file/{vault}/pdf/info` without transferring the document.
**Limitations:** no OCR (scanned PDFs aren't searchable), no annotation, no editing of the PDF itself.
+17 -9
View File
@@ -546,19 +546,27 @@ def _index_single_file_sync(vault_name: str, vault_path: str, file_path: str, va
stat = fpath.stat()
modified = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc).isoformat()
raw = fpath.read_text(encoding="utf-8", errors="replace")
# PDF handling — binary, must go through pdf_reader (same as _scan_vault)
tags: list[str] = []
title = fpath.stem.replace("-", " ").replace("_", " ")
content_preview = raw[:200].strip()
if ext == ".pdf":
from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text
raw = extract_pdf_text(fpath, max_chars=SEARCH_CONTENT_LIMIT)
pdf_meta = extract_pdf_metadata(fpath)
title = pdf_meta.get("title") or title
content_preview = raw[:200].strip()
else:
raw = fpath.read_text(encoding="utf-8", errors="replace")
content_preview = raw[:200].strip()
if ext == ".md":
post = parse_markdown_file(raw)
tags = _extract_tags(post)
inline_tags = _extract_inline_tags(post.content)
tags = list(set(tags) | set(inline_tags))
title = _extract_title(post, fpath)
content_preview = post.content[:200].strip()
if ext == ".md":
post = parse_markdown_file(raw)
tags = _extract_tags(post)
inline_tags = _extract_inline_tags(post.content)
tags = list(set(tags) | set(inline_tags))
title = _extract_title(post, fpath)
content_preview = post.content[:200].strip()
return {
"path": str(relative).replace("\\", "/"),
+105 -5
View File
@@ -19,7 +19,7 @@ from typing import Any
import frontmatter
import mistune
from fastapi import Body, Depends, FastAPI, HTTPException, Query
from fastapi import Body, Depends, FastAPI, HTTPException, Query, Request
from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse
from fastapi.staticfiles import StaticFiles
from pydantic import BaseModel, Field
@@ -2727,8 +2727,17 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
@app.get("/api/file/{vault_name}/pdf/stream")
async def api_pdf_stream(vault_name: str, path: str = Query(...)):
"""Stream a PDF file with Content-Type: application/pdf for inline browser viewing."""
async def api_pdf_stream(
request: Request,
vault_name: str,
path: str = Query(...),
current_user=Depends(require_auth),
):
"""Stream a PDF file with Content-Type: application/pdf for inline browser viewing.
Supports HTTP Range requests (206 Partial Content) so browsers can
progressively render large PDFs in the native viewer.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
vault_data = get_vault_data(vault_name)
@@ -2740,9 +2749,100 @@ async def api_pdf_stream(vault_name: str, path: str = Query(...)):
raise HTTPException(status_code=404, detail=f"File not found: {path}")
if file_path.suffix.lower() != ".pdf":
raise HTTPException(status_code=400, detail="Not a PDF file")
from fastapi.responses import FileResponse
file_size = file_path.stat().st_size
range_header = request.headers.get("range")
if range_header:
# Parse "bytes=start-end" (single range only; multi-range is not used by viewers)
m = re.match(r"bytes=(\d*)-(\d*)", range_header)
if not m:
raise HTTPException(status_code=416,
headers={"Content-Range": f"bytes */{file_size}"})
start_s, end_s = m.group(1), m.group(2)
if start_s == "" and end_s == "":
raise HTTPException(status_code=416,
headers={"Content-Range": f"bytes */{file_size}"})
if start_s == "":
# suffix range: last N bytes
length = min(int(end_s), file_size)
start = file_size - length
end = file_size - 1
else:
start = int(start_s)
end = int(end_s) if end_s else file_size - 1
end = min(end, file_size - 1)
if start > end or start >= file_size:
raise HTTPException(status_code=416,
headers={"Content-Range": f"bytes */{file_size}"})
chunk_size = end - start + 1
async def _partial():
# Open + reads offloaded to threads (avoid blocking the event loop — ASYNC230)
f = await asyncio.to_thread(open, str(file_path), "rb")
try:
await asyncio.to_thread(f.seek, start)
remaining = chunk_size
while remaining > 0:
data = await asyncio.to_thread(f.read, min(64 * 1024, remaining))
if not data:
break
remaining -= len(data)
yield data
finally:
await asyncio.to_thread(f.close)
return StreamingResponse(
_partial(),
status_code=206,
media_type="application/pdf",
headers={
"Content-Range": f"bytes {start}-{end}/{file_size}",
"Accept-Ranges": "bytes",
"Content-Length": str(chunk_size),
"Content-Disposition": f'inline; filename="{file_path.name}"',
},
)
return FileResponse(str(file_path), media_type="application/pdf", headers={
"Content-Disposition": f"inline; filename=\"{file_path.name}\""})
"Accept-Ranges": "bytes",
"Content-Disposition": f'inline; filename="{file_path.name}"'})
@app.get("/api/file/{vault_name}/pdf/info")
async def api_pdf_info(
vault_name: str,
path: str = Query(..., description="Relative path to PDF file"),
current_user=Depends(require_auth),
):
"""Return PDF metadata (pages, title, author, size) without the document content.
Lets the UI display file info before loading a heavy PDF into the viewer.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
vault_data = get_vault_data(vault_name)
if not vault_data:
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
vault_root = Path(vault_data["path"])
file_path = _resolve_safe_path(vault_root, path)
if not file_path.exists() or not file_path.is_file():
raise HTTPException(status_code=404, detail=f"File not found: {path}")
if file_path.suffix.lower() != ".pdf":
raise HTTPException(status_code=400, detail="Not a PDF file")
from backend.pdf_reader import extract_pdf_metadata
meta = extract_pdf_metadata(file_path)
stat = file_path.stat()
return {
"vault": vault_name,
"path": path,
"pages": meta.get("pages", 0),
"title": meta.get("title") or file_path.name,
"author": meta.get("author", ""),
"size_bytes": stat.st_size,
}
@app.get("/api/search", response_model=SearchResponse)
+48 -3
View File
@@ -2,10 +2,19 @@
from __future__ import annotations
import logging
import os
from concurrent.futures import ThreadPoolExecutor
from concurrent.futures import TimeoutError as FuturesTimeout
from pathlib import Path
logger = logging.getLogger(__name__)
# Configurable limits (see ROADMAP #74 G3):
# OBSIGATE_PDF_MAX_SIZE_MB — PDFs larger than this are not text-extracted (default 50)
# OBSIGATE_PDF_EXTRACT_TIMEOUT — seconds before extraction is abandoned (default 30)
PDF_MAX_SIZE_MB: int = int(os.environ.get("OBSIGATE_PDF_MAX_SIZE_MB", "50"))
PDF_EXTRACT_TIMEOUT: float = float(os.environ.get("OBSIGATE_PDF_EXTRACT_TIMEOUT", "30"))
PDF_READER: str = "pypdf"
PdfReader = None # type: ignore
try:
@@ -20,15 +29,51 @@ except ImportError:
logger.warning("No PDF reader available — install pypdf or pymupdf")
def pdf_exceeds_size_limit(file_path: Path) -> bool:
"""True if the PDF is larger than OBSIGATE_PDF_MAX_SIZE_MB (skip text extraction)."""
try:
return file_path.stat().st_size > PDF_MAX_SIZE_MB * 1024 * 1024
except OSError:
return False
def _run_with_timeout(fn, *args, timeout: float):
"""Run a sync function in a worker thread with a hard timeout.
A timed-out extraction leaves its worker thread running (daemon-style pool
is abandoned), but the request itself is freed — acceptable trade-off for
pathological PDFs on a self-hosted single-user server.
"""
executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="pdf-extract")
try:
future = executor.submit(fn, *args)
return future.result(timeout=timeout)
finally:
executor.shutdown(wait=False)
def extract_pdf_text(file_path: Path, max_chars: int = 100000) -> str:
"""Extract text from a PDF file. Returns empty string on failure."""
"""Extract text from a PDF file. Returns empty string on failure.
Oversized PDFs (> OBSIGATE_PDF_MAX_SIZE_MB) and extractions exceeding
OBSIGATE_PDF_EXTRACT_TIMEOUT seconds return "" instead of blocking.
"""
if PdfReader is None and PDF_READER == "pypdf":
return ""
if pdf_exceeds_size_limit(file_path):
logger.info("PDF too large to index (> %d MB), skipping text extraction: %s",
PDF_MAX_SIZE_MB, file_path)
return ""
try:
if PDF_READER == "pymupdf":
return _extract_pymupdf(file_path, max_chars)
return _run_with_timeout(_extract_pymupdf, file_path, max_chars,
timeout=PDF_EXTRACT_TIMEOUT)
else:
return _extract_pypdf(file_path, max_chars)
return _run_with_timeout(_extract_pypdf, file_path, max_chars,
timeout=PDF_EXTRACT_TIMEOUT)
except FuturesTimeout:
logger.warning("PDF text extraction timed out (%ss): %s", PDF_EXTRACT_TIMEOUT, file_path)
return ""
except Exception as e:
logger.warning("Failed to extract PDF text from %s: %s", file_path, e)
return ""
+138
View File
@@ -124,6 +124,144 @@ class TestPdfReader:
assert isinstance(toc, list)
class TestPdfIncrementalIndexing:
"""_index_single_file_sync (watcher/update path) must handle binary PDFs.
Regression: it used to read_text() every file, producing garbage for PDFs.
"""
def test_index_single_file_sync_extracts_pdf_text(self, pdf_dir: Path):
from backend.indexer import _index_single_file_sync
info = _index_single_file_sync("v", str(pdf_dir), str(pdf_dir / "simple.pdf"))
assert info is not None
assert info["extension"] == ".pdf"
assert "ObsiGate test PDF" in info["content"]
assert "uniqueword0" in info["content"]
def test_index_single_file_sync_pdf_title_from_metadata(self, pdf_dir: Path):
from backend.indexer import _index_single_file_sync
info = _index_single_file_sync("v", str(pdf_dir), str(pdf_dir / "simple.pdf"))
# title metadata was set at generation time
assert info["title"] == "Simple Test"
class TestPdfSizeLimit:
def test_oversized_pdf_skips_text_extraction(self, pdf_dir: Path, monkeypatch):
import backend.pdf_reader as pr
monkeypatch.setattr(pr, "PDF_MAX_SIZE_MB", 0) # everything is "too large"
text = pr.extract_pdf_text(pdf_dir / "simple.pdf")
assert text == ""
def test_normal_pdf_within_limit_extracts(self, pdf_dir: Path, monkeypatch):
import backend.pdf_reader as pr
monkeypatch.setattr(pr, "PDF_MAX_SIZE_MB", 50)
text = pr.extract_pdf_text(pdf_dir / "simple.pdf")
assert "ObsiGate test PDF" in text
# ── API: /pdf/stream (206 Range) + /pdf/info ───────────────────────────────
@pytest.fixture
def pdf_client():
"""TestClient (auth disabled) over a vault containing one generated PDF."""
tmp = Path(tempfile.mkdtemp())
vault = tmp / "PdfVault"
vault.mkdir()
make_simple_pdf(vault / "doc.pdf", pages=2, title="Doc Test", author="Bruno")
os.environ["VAULT_1_NAME"] = "PdfVault"
os.environ["VAULT_1_PATH"] = str(vault)
os.environ["OBSIGATE_AUTH_ENABLED"] = "false"
os.environ["OBSIGATE_WATCHER_ENABLED"] = "false"
import backend.main
backend.main._load_config = lambda: {"watcher_enabled": False}
from backend.indexer import build_index, index
for key in list(index.keys()):
del index[key]
import asyncio
loop = asyncio.new_event_loop()
asyncio.set_event_loop(loop)
loop.run_until_complete(build_index())
from fastapi.testclient import TestClient
client = TestClient(backend.main.app)
yield client
client.close()
shutil.rmtree(str(tmp), ignore_errors=True)
for k in ["VAULT_1_NAME", "VAULT_1_PATH", "OBSIGATE_AUTH_ENABLED", "OBSIGATE_WATCHER_ENABLED"]:
os.environ.pop(k, None)
class TestPdfStreamApi:
def test_stream_returns_200_application_pdf(self, pdf_client):
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf")
assert r.status_code == 200
assert r.headers["content-type"] == "application/pdf"
assert r.headers.get("accept-ranges") == "bytes"
assert r.content[:4] == b"%PDF"
def test_stream_full_range_returns_whole_file(self, pdf_client):
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
headers={"Range": "bytes=0-99999999"})
assert r.status_code == 206
assert r.headers["content-range"].startswith("bytes 0-")
assert r.content[:4] == b"%PDF"
def test_stream_partial_range(self, pdf_client):
full = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf").content
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
headers={"Range": "bytes=10-19"})
assert r.status_code == 206
assert r.content == full[10:20]
assert len(r.content) == 10
def test_stream_open_ended_range(self, pdf_client):
full = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf").content
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
headers={"Range": "bytes=100-"})
assert r.status_code == 206
assert r.content == full[100:]
def test_stream_bad_range_returns_416(self, pdf_client):
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.pdf",
headers={"Range": "bytes=999999999-"})
assert r.status_code == 416
assert "bytes */" in r.headers.get("content-range", "")
def test_stream_rejects_non_pdf(self, pdf_client):
r = pdf_client.get("/api/file/PdfVault/pdf/stream?path=doc.md")
assert r.status_code == 404 or r.status_code == 400
class TestPdfInfoApi:
def test_info_returns_metadata_without_content(self, pdf_client):
r = pdf_client.get("/api/file/PdfVault/pdf/info?path=doc.pdf")
assert r.status_code == 200
body = r.json()
assert body["pages"] == 2
assert body["title"] in ("Doc Test", "doc.pdf")
assert body["size_bytes"] > 0
assert "path" in body and body["vault"] == "PdfVault"
# no heavy content in the payload
assert "html" not in body
def test_info_missing_file_404(self, pdf_client):
r = pdf_client.get("/api/file/PdfVault/pdf/info?path=nope.pdf")
assert r.status_code == 404
def test_info_missing_non_pdf_404(self, pdf_client):
r = pdf_client.get("/api/file/PdfVault/pdf/info?path=readme.txt")
assert r.status_code == 404 # missing file checked before extension
# ── backend/indexer.py ─────────────────────────────────────────────────────