feat: PDF support + Split View (P1 roadmap items)
#74 — Support PDF complet: - backend/pdf_reader.py — pymupdf/pypdf text extraction - .pdf added to SUPPORTED_EXTENSIONS in indexer.py - PDF-aware file view API endpoint (metadata, text preview) - Streaming endpoint /api/file/{vault}/pdf/stream - Frontend PDF viewer with iframe, toolbar, download - CSS for PDF viewer container #75 — Split View (PaneManager): - New module frontend/js/pane-manager.js — ES module - 1-4 pane CSS grid with resize handles - Split right/down via keyboard (Ctrl+Alt+\) or PaneManager API - Persistence in localStorage (obsigate-panes) - Collapse to single pane, per-pane active state 285 tests passent
This commit is contained in:
+25
-19
@@ -57,7 +57,7 @@ SUPPORTED_EXTENSIONS = {
|
||||
".md", ".txt", ".log", ".py", ".js", ".ts", ".jsx", ".tsx",
|
||||
".sh", ".bash", ".zsh", ".fish", ".bat", ".cmd", ".ps1",
|
||||
".json", ".yaml", ".yml", ".toml", ".xml", ".csv",
|
||||
".cfg", ".ini", ".conf", ".env",
|
||||
".cfg", ".ini", ".conf", ".env", ".pdf",
|
||||
".html", ".css", ".scss", ".less",
|
||||
".java", ".c", ".cpp", ".h", ".hpp", ".cs", ".go", ".rs", ".rb",
|
||||
".php", ".sql", ".r", ".m", ".swift", ".kt",
|
||||
@@ -281,26 +281,32 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: Optional[Dict[str,
|
||||
stat = fpath.stat()
|
||||
modified = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc).isoformat()
|
||||
|
||||
raw = fpath.read_text(encoding="utf-8", errors="replace")
|
||||
# PDF handling — special path (binary, uses pdf_reader)
|
||||
if ext == ".pdf":
|
||||
from backend.pdf_reader import extract_pdf_text, extract_pdf_metadata
|
||||
raw = extract_pdf_text(fpath, max_chars=100000)
|
||||
pdf_meta = extract_pdf_metadata(fpath)
|
||||
title = pdf_meta.get("title") or fpath.stem.replace("-", " ").replace("_", " ")
|
||||
content_preview = raw[:200].strip()
|
||||
tags: List[str] = []
|
||||
else:
|
||||
raw = fpath.read_text(encoding="utf-8", errors="replace")
|
||||
tags: List[str] = []
|
||||
title = fpath.stem.replace("-", " ").replace("_", " ")
|
||||
content_preview = raw[:200].strip()
|
||||
|
||||
tags: List[str] = []
|
||||
title = fpath.stem.replace("-", " ").replace("_", " ")
|
||||
content_preview = raw[:200].strip()
|
||||
if ext == ".md":
|
||||
post = parse_markdown_file(raw)
|
||||
tags = _extract_tags(post)
|
||||
inline_tags = _extract_inline_tags(post.content)
|
||||
tags = list(set(tags) | set(inline_tags))
|
||||
title = _extract_title(post, fpath)
|
||||
content_preview = post.content[:200].strip()
|
||||
|
||||
if ext == ".md":
|
||||
post = parse_markdown_file(raw)
|
||||
tags = _extract_tags(post)
|
||||
# Merge inline #tags found in content body
|
||||
inline_tags = _extract_inline_tags(post.content)
|
||||
tags = list(set(tags) | set(inline_tags))
|
||||
title = _extract_title(post, fpath)
|
||||
content_preview = post.content[:200].strip()
|
||||
|
||||
# Extract wikilinks for backlink index
|
||||
_extract_wikilinks_for_backlinks(
|
||||
vault_name, str(relative).replace("\\", "/"),
|
||||
title, post.content
|
||||
)
|
||||
_extract_wikilinks_for_backlinks(
|
||||
vault_name, str(relative).replace("\\", "/"),
|
||||
title, post.content
|
||||
)
|
||||
|
||||
files.append({
|
||||
"path": str(relative).replace("\\", "/"),
|
||||
|
||||
@@ -2301,6 +2301,32 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
|
||||
|
||||
ext = file_path.suffix.lower()
|
||||
|
||||
# === PDF: special handling before read_text (binary file) ===
|
||||
if ext == ".pdf":
|
||||
try:
|
||||
from backend.pdf_reader import extract_pdf_text, extract_pdf_metadata
|
||||
pdf_text = extract_pdf_text(file_path, max_chars=100000)
|
||||
pdf_meta = extract_pdf_metadata(file_path)
|
||||
size = file_path.stat().st_size
|
||||
return {
|
||||
"vault": vault_name,
|
||||
"path": path,
|
||||
"title": pdf_meta.get("title") or file_path.name,
|
||||
"tags": [],
|
||||
"frontmatter": {},
|
||||
"html": f"<div class='pdf-viewer'><p>PDF — {pdf_meta.get('pages', '?')} pages</p><pre>{pdf_text[:5000]}</pre></div>",
|
||||
"raw_length": size,
|
||||
"extension": ext,
|
||||
"is_markdown": False,
|
||||
"is_pdf": True,
|
||||
"unsupported": False,
|
||||
"pdf_metadata": pdf_meta,
|
||||
"size_bytes": size,
|
||||
}
|
||||
except Exception as e:
|
||||
logger.error(f"PDF read error for {path}: {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Error reading PDF: {str(e)}")
|
||||
|
||||
try:
|
||||
raw = file_path.read_text(encoding="utf-8", errors="replace")
|
||||
except PermissionError as e:
|
||||
@@ -2365,6 +2391,25 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/file/{vault_name}/pdf/stream")
|
||||
async def api_pdf_stream(vault_name: str, path: str = Query(...)):
|
||||
"""Stream a PDF file with Content-Type: application/pdf for inline browser viewing."""
|
||||
if not check_vault_access(vault_name, current_user):
|
||||
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
||||
vault_data = get_vault_data(vault_name)
|
||||
if not vault_data:
|
||||
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
||||
vault_root = Path(vault_data["path"])
|
||||
file_path = _resolve_safe_path(vault_root, path)
|
||||
if not file_path.exists() or not file_path.is_file():
|
||||
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
||||
if file_path.suffix.lower() != ".pdf":
|
||||
raise HTTPException(status_code=400, detail="Not a PDF file")
|
||||
from fastapi.responses import FileResponse
|
||||
return FileResponse(str(file_path), media_type="application/pdf", headers={
|
||||
"Content-Disposition": f"inline; filename=\"{file_path.name}\""})
|
||||
|
||||
|
||||
@app.get("/api/search", response_model=SearchResponse)
|
||||
async def api_search(
|
||||
q: str = Query("", description="Search query"),
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
"""PDF text extraction for ObsiGate — pypdf backend, pymupdf optional fallback."""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PDF_READER: str = "pypdf"
|
||||
try:
|
||||
import fitz # pymupdf
|
||||
PDF_READER = "pymupdf"
|
||||
logger.info("PDF reader: pymupdf (high performance)")
|
||||
except ImportError:
|
||||
try:
|
||||
from pypdf import PdfReader # type: ignore
|
||||
logger.info("PDF reader: pypdf (pure Python)")
|
||||
except ImportError:
|
||||
PdfReader = None # type: ignore
|
||||
logger.warning("No PDF reader available — install pypdf or pymupdf")
|
||||
|
||||
|
||||
def extract_pdf_text(file_path: Path, max_chars: int = 100000) -> str:
|
||||
"""Extract text from a PDF file. Returns empty string on failure."""
|
||||
if PdfReader is None and PDF_READER == "pypdf":
|
||||
return ""
|
||||
try:
|
||||
if PDF_READER == "pymupdf":
|
||||
return _extract_pymupdf(file_path, max_chars)
|
||||
else:
|
||||
return _extract_pypdf(file_path, max_chars)
|
||||
except Exception as e:
|
||||
logger.warning("Failed to extract PDF text from %s: %s", file_path, e)
|
||||
return ""
|
||||
|
||||
|
||||
def extract_pdf_metadata(file_path: Path) -> dict:
|
||||
"""Extract metadata from a PDF file."""
|
||||
info = {"pages": 0, "title": "", "author": ""}
|
||||
try:
|
||||
if PDF_READER == "pymupdf":
|
||||
doc = fitz.open(str(file_path))
|
||||
info["pages"] = doc.page_count
|
||||
meta = doc.metadata or {}
|
||||
info["title"] = meta.get("title", "")
|
||||
info["author"] = meta.get("author", "")
|
||||
doc.close()
|
||||
else:
|
||||
reader = PdfReader(str(file_path))
|
||||
info["pages"] = len(reader.pages)
|
||||
meta = reader.metadata or {}
|
||||
if meta:
|
||||
info["title"] = str(meta.get("/Title", ""))
|
||||
info["author"] = str(meta.get("/Author", ""))
|
||||
except Exception as e:
|
||||
logger.warning("Failed to extract PDF metadata from %s: %s", file_path, e)
|
||||
return info
|
||||
|
||||
|
||||
def _extract_pymupdf(file_path: Path, max_chars: int) -> str:
|
||||
doc = fitz.open(str(file_path))
|
||||
parts = []
|
||||
total = 0
|
||||
for page in doc:
|
||||
text = page.get_text()
|
||||
if text:
|
||||
if total + len(text) > max_chars:
|
||||
remaining = max_chars - total
|
||||
if remaining > 0:
|
||||
parts.append(text[:remaining])
|
||||
break
|
||||
parts.append(text)
|
||||
total += len(text)
|
||||
doc.close()
|
||||
return "\f".join(parts)
|
||||
|
||||
|
||||
def _extract_pypdf(file_path: Path, max_chars: int) -> str:
|
||||
reader = PdfReader(str(file_path))
|
||||
parts = []
|
||||
total = 0
|
||||
for page in reader.pages:
|
||||
text = page.extract_text() or ""
|
||||
if text:
|
||||
if total + len(text) > max_chars:
|
||||
remaining = max_chars - total
|
||||
if remaining > 0:
|
||||
parts.append(text[:remaining])
|
||||
break
|
||||
parts.append(text)
|
||||
total += len(text)
|
||||
return "\f".join(parts)
|
||||
@@ -12,3 +12,4 @@ sortedcontainers>=2.4.0
|
||||
snowballstemmer>=2.2.0
|
||||
weasyprint>=60.0
|
||||
httpx>=0.27.0
|
||||
pypdf>=4.0
|
||||
|
||||
Reference in New Issue
Block a user