feat: PDF support + Split View (P1 roadmap items)

#74 — Support PDF complet:
- backend/pdf_reader.py — pymupdf/pypdf text extraction
- .pdf added to SUPPORTED_EXTENSIONS in indexer.py
- PDF-aware file view API endpoint (metadata, text preview)
- Streaming endpoint /api/file/{vault}/pdf/stream
- Frontend PDF viewer with iframe, toolbar, download
- CSS for PDF viewer container

#75 — Split View (PaneManager):
- New module frontend/js/pane-manager.js — ES module
- 1-4 pane CSS grid with resize handles
- Split right/down via keyboard (Ctrl+Alt+\) or PaneManager API
- Persistence in localStorage (obsigate-panes)
- Collapse to single pane, per-pane active state

285 tests passent
This commit is contained in:
2026-07-23 17:33:51 -04:00
parent 5b956f8d95
commit 93d0bda627
8 changed files with 506 additions and 19 deletions
+25 -19
View File
@@ -57,7 +57,7 @@ SUPPORTED_EXTENSIONS = {
".md", ".txt", ".log", ".py", ".js", ".ts", ".jsx", ".tsx",
".sh", ".bash", ".zsh", ".fish", ".bat", ".cmd", ".ps1",
".json", ".yaml", ".yml", ".toml", ".xml", ".csv",
".cfg", ".ini", ".conf", ".env",
".cfg", ".ini", ".conf", ".env", ".pdf",
".html", ".css", ".scss", ".less",
".java", ".c", ".cpp", ".h", ".hpp", ".cs", ".go", ".rs", ".rb",
".php", ".sql", ".r", ".m", ".swift", ".kt",
@@ -281,26 +281,32 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: Optional[Dict[str,
stat = fpath.stat()
modified = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc).isoformat()
raw = fpath.read_text(encoding="utf-8", errors="replace")
# PDF handling — special path (binary, uses pdf_reader)
if ext == ".pdf":
from backend.pdf_reader import extract_pdf_text, extract_pdf_metadata
raw = extract_pdf_text(fpath, max_chars=100000)
pdf_meta = extract_pdf_metadata(fpath)
title = pdf_meta.get("title") or fpath.stem.replace("-", " ").replace("_", " ")
content_preview = raw[:200].strip()
tags: List[str] = []
else:
raw = fpath.read_text(encoding="utf-8", errors="replace")
tags: List[str] = []
title = fpath.stem.replace("-", " ").replace("_", " ")
content_preview = raw[:200].strip()
tags: List[str] = []
title = fpath.stem.replace("-", " ").replace("_", " ")
content_preview = raw[:200].strip()
if ext == ".md":
post = parse_markdown_file(raw)
tags = _extract_tags(post)
inline_tags = _extract_inline_tags(post.content)
tags = list(set(tags) | set(inline_tags))
title = _extract_title(post, fpath)
content_preview = post.content[:200].strip()
if ext == ".md":
post = parse_markdown_file(raw)
tags = _extract_tags(post)
# Merge inline #tags found in content body
inline_tags = _extract_inline_tags(post.content)
tags = list(set(tags) | set(inline_tags))
title = _extract_title(post, fpath)
content_preview = post.content[:200].strip()
# Extract wikilinks for backlink index
_extract_wikilinks_for_backlinks(
vault_name, str(relative).replace("\\", "/"),
title, post.content
)
_extract_wikilinks_for_backlinks(
vault_name, str(relative).replace("\\", "/"),
title, post.content
)
files.append({
"path": str(relative).replace("\\", "/"),
+45
View File
@@ -2301,6 +2301,32 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
ext = file_path.suffix.lower()
# === PDF: special handling before read_text (binary file) ===
if ext == ".pdf":
try:
from backend.pdf_reader import extract_pdf_text, extract_pdf_metadata
pdf_text = extract_pdf_text(file_path, max_chars=100000)
pdf_meta = extract_pdf_metadata(file_path)
size = file_path.stat().st_size
return {
"vault": vault_name,
"path": path,
"title": pdf_meta.get("title") or file_path.name,
"tags": [],
"frontmatter": {},
"html": f"<div class='pdf-viewer'><p>PDF — {pdf_meta.get('pages', '?')} pages</p><pre>{pdf_text[:5000]}</pre></div>",
"raw_length": size,
"extension": ext,
"is_markdown": False,
"is_pdf": True,
"unsupported": False,
"pdf_metadata": pdf_meta,
"size_bytes": size,
}
except Exception as e:
logger.error(f"PDF read error for {path}: {e}")
raise HTTPException(status_code=500, detail=f"Error reading PDF: {str(e)}")
try:
raw = file_path.read_text(encoding="utf-8", errors="replace")
except PermissionError as e:
@@ -2365,6 +2391,25 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
}
@app.get("/api/file/{vault_name}/pdf/stream")
async def api_pdf_stream(vault_name: str, path: str = Query(...)):
"""Stream a PDF file with Content-Type: application/pdf for inline browser viewing."""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
vault_data = get_vault_data(vault_name)
if not vault_data:
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
vault_root = Path(vault_data["path"])
file_path = _resolve_safe_path(vault_root, path)
if not file_path.exists() or not file_path.is_file():
raise HTTPException(status_code=404, detail=f"File not found: {path}")
if file_path.suffix.lower() != ".pdf":
raise HTTPException(status_code=400, detail="Not a PDF file")
from fastapi.responses import FileResponse
return FileResponse(str(file_path), media_type="application/pdf", headers={
"Content-Disposition": f"inline; filename=\"{file_path.name}\""})
@app.get("/api/search", response_model=SearchResponse)
async def api_search(
q: str = Query("", description="Search query"),
+92
View File
@@ -0,0 +1,92 @@
"""PDF text extraction for ObsiGate — pypdf backend, pymupdf optional fallback."""
from __future__ import annotations
import logging
from pathlib import Path
logger = logging.getLogger(__name__)
PDF_READER: str = "pypdf"
try:
import fitz # pymupdf
PDF_READER = "pymupdf"
logger.info("PDF reader: pymupdf (high performance)")
except ImportError:
try:
from pypdf import PdfReader # type: ignore
logger.info("PDF reader: pypdf (pure Python)")
except ImportError:
PdfReader = None # type: ignore
logger.warning("No PDF reader available — install pypdf or pymupdf")
def extract_pdf_text(file_path: Path, max_chars: int = 100000) -> str:
"""Extract text from a PDF file. Returns empty string on failure."""
if PdfReader is None and PDF_READER == "pypdf":
return ""
try:
if PDF_READER == "pymupdf":
return _extract_pymupdf(file_path, max_chars)
else:
return _extract_pypdf(file_path, max_chars)
except Exception as e:
logger.warning("Failed to extract PDF text from %s: %s", file_path, e)
return ""
def extract_pdf_metadata(file_path: Path) -> dict:
"""Extract metadata from a PDF file."""
info = {"pages": 0, "title": "", "author": ""}
try:
if PDF_READER == "pymupdf":
doc = fitz.open(str(file_path))
info["pages"] = doc.page_count
meta = doc.metadata or {}
info["title"] = meta.get("title", "")
info["author"] = meta.get("author", "")
doc.close()
else:
reader = PdfReader(str(file_path))
info["pages"] = len(reader.pages)
meta = reader.metadata or {}
if meta:
info["title"] = str(meta.get("/Title", ""))
info["author"] = str(meta.get("/Author", ""))
except Exception as e:
logger.warning("Failed to extract PDF metadata from %s: %s", file_path, e)
return info
def _extract_pymupdf(file_path: Path, max_chars: int) -> str:
doc = fitz.open(str(file_path))
parts = []
total = 0
for page in doc:
text = page.get_text()
if text:
if total + len(text) > max_chars:
remaining = max_chars - total
if remaining > 0:
parts.append(text[:remaining])
break
parts.append(text)
total += len(text)
doc.close()
return "\f".join(parts)
def _extract_pypdf(file_path: Path, max_chars: int) -> str:
reader = PdfReader(str(file_path))
parts = []
total = 0
for page in reader.pages:
text = page.extract_text() or ""
if text:
if total + len(text) > max_chars:
remaining = max_chars - total
if remaining > 0:
parts.append(text[:remaining])
break
parts.append(text)
total += len(text)
return "\f".join(parts)
+1
View File
@@ -12,3 +12,4 @@ sortedcontainers>=2.4.0
snowballstemmer>=2.2.0
weasyprint>=60.0
httpx>=0.27.0
pypdf>=4.0