feat: affichage PDF avec TOC + frame-src CSP
CI / lint (push) Successful in 23s
CI / security (push) Successful in 14s
CI / test (push) Successful in 27s
CI / build (push) Successful in 8s
CI / e2e (push) Failing after 1m29s
Desktop Build / build-windows (push) Has been cancelled
Desktop Build / build-linux (push) Has been cancelled
CI / lint (push) Successful in 23s
CI / security (push) Successful in 14s
CI / test (push) Successful in 27s
CI / build (push) Successful in 8s
CI / e2e (push) Failing after 1m29s
Desktop Build / build-windows (push) Has been cancelled
Desktop Build / build-linux (push) Has been cancelled
- CSP: ajout frame-src 'self' http://127.0.0.1:* (permet iframe PDF) - pdf_reader.py: extract_pdf_toc() — extrait les bookmarks/outline (pypdf + pymupdf) - main.py: inclut pdf_toc dans la réponse /api/file pour les PDF - viewer.js: sidebar TOC avec liens vers les pages (si bookmarks présents) - style.css: layout .pdf-body flex + .pdf-toc sidebar 240px
This commit is contained in:
+3
-1
@@ -2339,9 +2339,10 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
|
||||
# === PDF: special handling before read_text (binary file) ===
|
||||
if ext == ".pdf":
|
||||
try:
|
||||
from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text
|
||||
from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text, extract_pdf_toc
|
||||
pdf_text = extract_pdf_text(file_path, max_chars=100000)
|
||||
pdf_meta = extract_pdf_metadata(file_path)
|
||||
pdf_toc = extract_pdf_toc(file_path)
|
||||
size = file_path.stat().st_size
|
||||
return {
|
||||
"vault": vault_name,
|
||||
@@ -2356,6 +2357,7 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
|
||||
"is_pdf": True,
|
||||
"unsupported": False,
|
||||
"pdf_metadata": pdf_meta,
|
||||
"pdf_toc": pdf_toc,
|
||||
"size_bytes": size,
|
||||
}
|
||||
except Exception as e:
|
||||
|
||||
@@ -57,6 +57,44 @@ def extract_pdf_metadata(file_path: Path) -> dict:
|
||||
return info
|
||||
|
||||
|
||||
def extract_pdf_toc(file_path: Path) -> list[dict]:
|
||||
"""Extract table of contents (bookmarks/outline) from a PDF file.
|
||||
|
||||
Returns a list of {title, page, level} dicts, or empty list on failure.
|
||||
"""
|
||||
toc: list[dict] = []
|
||||
try:
|
||||
if PDF_READER == "pymupdf":
|
||||
doc = fitz.open(str(file_path))
|
||||
raw = doc.get_toc(simple=False)
|
||||
for item in raw:
|
||||
toc.append({
|
||||
"title": str(item[1]),
|
||||
"page": item[2],
|
||||
"level": item[0],
|
||||
})
|
||||
doc.close()
|
||||
elif PdfReader is not None:
|
||||
reader = PdfReader(str(file_path))
|
||||
outline = reader.outline
|
||||
if outline:
|
||||
def _flatten(items, level=1):
|
||||
for item in items:
|
||||
if isinstance(item, list):
|
||||
_flatten(item, level + 1)
|
||||
elif hasattr(item, 'title') and hasattr(item, 'page'):
|
||||
page_num = reader.get_page_number(item.page) + 1 if hasattr(item, 'page') and item.page else 1
|
||||
toc.append({
|
||||
"title": str(item.title),
|
||||
"page": page_num,
|
||||
"level": level,
|
||||
})
|
||||
_flatten(outline)
|
||||
except Exception as e:
|
||||
logger.warning("Failed to extract PDF TOC from %s: %s", file_path, e)
|
||||
return toc
|
||||
|
||||
|
||||
def _extract_pymupdf(file_path: Path, max_chars: int) -> str:
|
||||
doc = fitz.open(str(file_path))
|
||||
parts = []
|
||||
|
||||
Reference in New Issue
Block a user