526 lines
20 KiB
Python
526 lines
20 KiB
Python
"""File browsing & reading endpoints (ROADMAP #85, tranche 6a).
|
|
|
|
Handlers déplacés depuis :mod:`backend.main` sans changement de
|
|
comportement : mêmes chemins (``/api/browse/*``, ``/api/file/*`` en
|
|
lecture), mêmes modèles de réponse (déménagés dans
|
|
:mod:`backend.schemas`), mêmes dépendances d'authentification.
|
|
|
|
Adaptations strictement équivalentes :
|
|
- ``_resolve_safe_path`` → :mod:`backend.services.paths` (pass-through).
|
|
- ``_render_markdown`` reste dans ``main`` (import différé, extraction
|
|
prévue dans une tranche ultérieure).
|
|
- ``_content_disposition`` / ``_media_max_inline_bytes`` / ``EXT_TO_LANG``
|
|
ont déménagé : helpers partagés dans :mod:`backend.routers.helpers`
|
|
(``EXT_TO_LANG`` n'était utilisé que par la vue fichier).
|
|
"""
|
|
|
|
import html as html_mod
|
|
import logging
|
|
from pathlib import Path
|
|
from urllib.parse import quote
|
|
|
|
from fastapi import APIRouter, Depends, HTTPException, Query
|
|
from fastapi.responses import FileResponse
|
|
|
|
from backend.auth.middleware import check_vault_access, require_auth
|
|
from backend.history import record_open
|
|
from backend.indexer import (
|
|
_extract_tags,
|
|
get_backlinks,
|
|
get_vault_data,
|
|
parse_markdown_file,
|
|
)
|
|
from backend.media_types import is_audio, is_image, is_video, media_mime_type
|
|
from backend.routers.helpers import media_max_inline_bytes
|
|
from backend.schemas import (
|
|
BacklinksResponse,
|
|
BrowseResponse,
|
|
FileContentResponse,
|
|
FileRawResponse,
|
|
)
|
|
from backend.services.files import read_raw_file
|
|
from backend.services.paths import resolve_safe_path
|
|
from backend.services.vaults import browse_directory
|
|
|
|
logger = logging.getLogger("obsigate")
|
|
|
|
# Map file extensions to highlight.js language hints
|
|
EXT_TO_LANG = {
|
|
".py": "python", ".js": "javascript", ".ts": "typescript",
|
|
".jsx": "jsx", ".tsx": "tsx", ".sh": "bash", ".bash": "bash",
|
|
".zsh": "bash", ".fish": "fish", ".bat": "batch", ".cmd": "batch",
|
|
".ps1": "powershell", ".json": "json", ".yaml": "yaml", ".yml": "yaml",
|
|
".toml": "toml", ".xml": "xml", ".csv": "plaintext",
|
|
".cfg": "ini", ".ini": "ini", ".conf": "ini", ".env": "bash",
|
|
".html": "html", ".css": "css", ".scss": "scss", ".less": "less",
|
|
".java": "java", ".c": "c", ".cpp": "cpp", ".h": "c", ".hpp": "cpp",
|
|
".cs": "csharp", ".go": "go", ".rs": "rust", ".rb": "ruby",
|
|
".php": "php", ".sql": "sql", ".r": "r", ".swift": "swift",
|
|
".kt": "kotlin", ".txt": "plaintext", ".log": "plaintext",
|
|
".lua": "lua", ".pl": "perl", ".pm": "perl", ".ex": "elixir", ".exs": "elixir",
|
|
".dart": "dart", ".tf": "haskell", ".gradle": "groovy", ".groovy": "groovy",
|
|
".graphql": "graphql", ".gql": "graphql", ".prisma": "sql", ".proto": "c",
|
|
".vb": "basic", ".asm": "x86asm", ".s": "armasm",
|
|
".vue": "xml", ".svelte": "xml", ".astro": "xml",
|
|
".properties": "ini", ".service": "ini", ".hosts": "ini",
|
|
".ksh": "bash", ".dockerfile": "dockerfile",
|
|
".makefile": "makefile", ".cmake": "cmake",
|
|
}
|
|
|
|
router = APIRouter(tags=["files"])
|
|
|
|
|
|
@router.get("/api/browse/{vault_name}", response_model=BrowseResponse)
|
|
async def api_browse(vault_name: str, path: str = "", current_user=Depends(require_auth)):
|
|
"""Browse directories and files in a vault at a given path level.
|
|
|
|
Returns sorted entries (directories first, then files) with metadata.
|
|
Hidden files/directories (starting with ``"."`` ) are excluded.
|
|
|
|
Args:
|
|
vault_name: Name of the vault to browse.
|
|
path: Relative directory path within the vault (empty = root).
|
|
|
|
Returns:
|
|
``BrowseResponse`` with vault name, path, and item list.
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
return browse_directory(vault_name, path)
|
|
|
|
|
|
@router.get("/api/file/{vault_name}/raw", response_model=FileRawResponse)
|
|
async def api_file_raw(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
|
|
"""Return raw file content as plain text.
|
|
|
|
Args:
|
|
vault_name: Name of the vault.
|
|
path: Relative file path within the vault.
|
|
|
|
Returns:
|
|
``FileRawResponse`` with vault, path, and raw text content.
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
return read_raw_file(vault_name, path)
|
|
|
|
|
|
@router.get("/api/file/{vault_name}/download", response_class=FileResponse)
|
|
async def api_file_download(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
|
|
"""Download a file as an attachment.
|
|
|
|
Args:
|
|
vault_name: Name of the vault.
|
|
path: Relative file path within the vault.
|
|
|
|
Returns:
|
|
``FileResponse`` with ``application/octet-stream`` content-type.
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
vault_data = get_vault_data(vault_name)
|
|
if not vault_data:
|
|
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
|
|
|
vault_root = Path(vault_data["path"])
|
|
file_path = resolve_safe_path(vault_root, path)
|
|
|
|
if not file_path.exists() or not file_path.is_file():
|
|
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
|
|
|
# Record history
|
|
record_open(current_user.get("username"), vault_name, path)
|
|
|
|
return FileResponse(
|
|
path=str(file_path),
|
|
filename=file_path.name,
|
|
media_type="application/octet-stream",
|
|
)
|
|
|
|
|
|
@router.get("/api/file/{vault_name}/backlinks", response_model=BacklinksResponse)
|
|
async def api_file_backlinks(
|
|
vault_name: str,
|
|
path: str = Query(..., description="Relative path to file"),
|
|
current_user=Depends(require_auth),
|
|
):
|
|
"""Get backlinks (files linking to this file via wikilinks).
|
|
|
|
Returns a list of files that contain `[[wikilinks]]` pointing
|
|
to the requested file, across all accessible vaults.
|
|
|
|
Args:
|
|
vault_name: Name of the vault containing the target file.
|
|
path: Relative path of the target file within the vault.
|
|
|
|
Returns:
|
|
``{"vault": str, "path": str, "backlinks": [...]}``
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
vault_data = get_vault_data(vault_name)
|
|
if not vault_data:
|
|
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
|
|
|
user_vaults = current_user.get("_token_vaults") or current_user.get("vaults", [])
|
|
backlinks = get_backlinks(vault_name, path)
|
|
|
|
# Filter by user-accessible vaults
|
|
if "*" not in user_vaults:
|
|
backlinks = [b for b in backlinks if b["vault"] in user_vaults]
|
|
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"backlinks": backlinks,
|
|
"total": len(backlinks),
|
|
}
|
|
|
|
|
|
@router.get("/api/file/{vault_name}", response_model=FileContentResponse)
|
|
async def api_file(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
|
|
"""Return rendered HTML and metadata for a file.
|
|
|
|
Markdown files are parsed for frontmatter, rendered with wikilink
|
|
support, and returned with extracted tags. Other supported file
|
|
types are syntax-highlighted as code blocks.
|
|
|
|
Args:
|
|
vault_name: Name of the vault.
|
|
path: Relative file path within the vault.
|
|
|
|
Returns:
|
|
``FileContentResponse`` with HTML, metadata, and tags.
|
|
"""
|
|
from backend.main import _render_markdown # différé : évite l'import circulaire (#85)
|
|
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
vault_data = get_vault_data(vault_name)
|
|
if not vault_data:
|
|
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
|
|
|
vault_root = Path(vault_data["path"])
|
|
file_path = resolve_safe_path(vault_root, path)
|
|
|
|
if not file_path.exists() or not file_path.is_file():
|
|
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
|
|
|
# Record history
|
|
record_open(current_user.get("username"), vault_name, path, title=file_path.name)
|
|
|
|
ext = file_path.suffix.lower()
|
|
|
|
# === PDF: special handling before read_text (binary file) ===
|
|
if ext == ".pdf":
|
|
try:
|
|
from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text, extract_pdf_toc
|
|
pdf_text = extract_pdf_text(file_path, max_chars=100000)
|
|
pdf_meta = extract_pdf_metadata(file_path)
|
|
pdf_toc = extract_pdf_toc(file_path)
|
|
size = file_path.stat().st_size
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": pdf_meta.get("title") or file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": f"<div class='pdf-viewer'><p>PDF — {pdf_meta.get('pages', '?')} pages</p><pre>{pdf_text[:5000]}</pre></div>",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_pdf": True,
|
|
"unsupported": False,
|
|
"pdf_metadata": pdf_meta,
|
|
"pdf_toc": pdf_toc,
|
|
"size_bytes": size,
|
|
}
|
|
except Exception as e:
|
|
logger.error(f"PDF read error for {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading PDF: {e!s}")
|
|
|
|
# === Excel .xlsx: render sheets as HTML tables (binary, before read_text) ===
|
|
if ext == ".xlsx":
|
|
try:
|
|
from backend.xlsx_reader import render_sheets
|
|
|
|
sheets = render_sheets(file_path)
|
|
size = file_path.stat().st_size
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": sheets[0]["html"] if sheets else "",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_xlsx": True,
|
|
"xlsx_sheets": sheets,
|
|
"unsupported": False,
|
|
"size_bytes": size,
|
|
}
|
|
except Exception as e:
|
|
logger.error(f"XLSX read error for {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
|
|
|
|
# === Images: return as viewable image ===
|
|
if is_image(ext):
|
|
size = file_path.stat().st_size
|
|
mime = media_mime_type(str(file_path))
|
|
# #108-B1 — the raw endpoint returns JSON (FileRawResponse), so the
|
|
# standalone <img> must point to /api/image, which serves the bytes
|
|
# with the right MIME type. Paths are URL-encoded (accents, spaces).
|
|
img_url = f"/api/image/{quote(vault_name, safe='')}?path={quote(path, safe='')}"
|
|
html = (
|
|
f'<div class="image-viewer">'
|
|
f'<img src="{img_url}" '
|
|
f'alt="{html_mod.escape(file_path.name, quote=True)}" '
|
|
f'style="max-width:100%;max-height:80vh;object-fit:contain" />'
|
|
f'</div>'
|
|
)
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": html,
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_image": True,
|
|
"image_mime": mime,
|
|
"size_bytes": size,
|
|
}
|
|
|
|
# === Audio / Video: HTML5 players streamed from /api/media (roadmap #109) ===
|
|
if is_audio(ext) or is_video(ext):
|
|
size = file_path.stat().st_size
|
|
mime = media_mime_type(str(file_path))
|
|
media_kind = "audio" if is_audio(ext) else "video"
|
|
|
|
# #109-A3 — beyond the inline limit the viewer falls back to download
|
|
# (a single uvicorn worker must not be pinned by multi-GB media).
|
|
if size > media_max_inline_bytes():
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": "",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"unsupported": True,
|
|
"media_too_large": True,
|
|
"size_bytes": size,
|
|
}
|
|
|
|
# #109-A2 — byte-range endpoint: enables scrub and is required by Safari.
|
|
stream_url = f"/api/media/{quote(vault_name, safe='')}?path={quote(path, safe='')}"
|
|
if media_kind == "audio":
|
|
html = (
|
|
f'<div class="audio-viewer">'
|
|
f'<audio controls preload="metadata" src="{stream_url}"></audio>'
|
|
f'</div>'
|
|
)
|
|
else:
|
|
html = (
|
|
f'<div class="video-viewer">'
|
|
f'<video controls playsinline preload="metadata" src="{stream_url}"></video>'
|
|
f'</div>'
|
|
)
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": html,
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_audio": media_kind == "audio",
|
|
"is_video": media_kind == "video",
|
|
"media_mime": mime,
|
|
"stream_url": stream_url,
|
|
"size_bytes": size,
|
|
}
|
|
|
|
try:
|
|
raw = file_path.read_text(encoding="utf-8", errors="replace")
|
|
except PermissionError as e:
|
|
logger.error(f"Permission denied reading file {path}: {e}")
|
|
raise HTTPException(status_code=403, detail=f"Permission denied: cannot read file {path}")
|
|
except UnicodeDecodeError:
|
|
# Binary / unsupported file — return structured info with download option
|
|
size = file_path.stat().st_size
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": "",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"unsupported": True,
|
|
"size_bytes": size,
|
|
}
|
|
except Exception as e:
|
|
logger.error(f"Unexpected error reading file {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading file: {e!s}")
|
|
|
|
# === CSV: render as HTML table ===
|
|
if ext == ".csv":
|
|
import csv
|
|
import io as csv_io
|
|
reader = csv.reader(csv_io.StringIO(raw))
|
|
rows = list(reader)
|
|
if not rows:
|
|
html = "<p><em>Fichier CSV vide</em></p>"
|
|
else:
|
|
headers = rows[0]
|
|
data_rows = rows[1:]
|
|
html = '<div class="csv-table-wrapper"><table class="csv-table"><thead><tr>'
|
|
for h in headers:
|
|
html += f"<th>{h}</th>"
|
|
html += "</tr></thead><tbody>"
|
|
for row in data_rows:
|
|
html += "<tr>"
|
|
for cell in row:
|
|
html += f"<td>{cell}</td>"
|
|
html += "</tr>"
|
|
html += "</tbody></table></div>"
|
|
return {
|
|
"vault": vault_name, "path": path,
|
|
"title": file_path.name, "tags": [], "frontmatter": {},
|
|
"html": html, "raw_length": len(raw), "extension": ext,
|
|
"is_markdown": False, "is_csv": True,
|
|
}
|
|
|
|
# === JSON: syntax-highlighted display ===
|
|
if ext == ".json":
|
|
import json as json_mod
|
|
try:
|
|
parsed = json_mod.loads(raw)
|
|
formatted = json_mod.dumps(parsed, indent=2, ensure_ascii=False)
|
|
except json_mod.JSONDecodeError:
|
|
formatted = raw
|
|
html = f"<pre class='json-viewer'><code>{html_mod.escape(formatted)}</code></pre>"
|
|
return {
|
|
"vault": vault_name, "path": path,
|
|
"title": file_path.name, "tags": [], "frontmatter": {},
|
|
"html": html, "raw_length": len(raw), "extension": ext,
|
|
"is_markdown": False, "is_json": True,
|
|
}
|
|
|
|
# === Excalidraw .excalidraw.md (Obsidian plugin format) ===
|
|
if path.lower().endswith(".excalidraw.md"):
|
|
import re as re_mod
|
|
raw_lower = file_path.read_text(encoding="utf-8", errors="replace")
|
|
# Check for excalidraw-plugin in frontmatter or body
|
|
if "excalidraw-plugin:" in raw_lower:
|
|
# Extract compressed JSON block
|
|
match = re_mod.search(r'```compressed-json\n(.*?)\n```', raw_lower, re_mod.DOTALL)
|
|
if match:
|
|
compressed = match.group(1).strip()
|
|
return {
|
|
"vault": vault_name, "path": path,
|
|
"title": file_path.name.replace(".excalidraw.md", ""),
|
|
"tags": [], "frontmatter": {},
|
|
"html": "", "raw_length": len(raw_lower),
|
|
"extension": ".excalidraw.md",
|
|
"is_markdown": False,
|
|
"is_excalidraw": True,
|
|
"excalidraw_data_compressed": compressed,
|
|
}
|
|
# Fallback: treat as regular markdown
|
|
raw = raw_lower
|
|
if ext == ".excalidraw":
|
|
import json as json_mod
|
|
try:
|
|
parsed = json_mod.loads(raw)
|
|
except json_mod.JSONDecodeError:
|
|
parsed = None
|
|
if parsed and parsed.get("type") == "excalidraw":
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": parsed.get("appState", {}).get("name") or file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": "",
|
|
"raw_length": len(raw),
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_excalidraw": True,
|
|
"excalidraw_data": {
|
|
"elements": parsed.get("elements", []),
|
|
"appState": parsed.get("appState", {}),
|
|
"files": parsed.get("files", {}),
|
|
},
|
|
}
|
|
else:
|
|
# Not a valid Excalidraw file — fall through to text viewer
|
|
pass
|
|
|
|
# === Plain text / other readable files ===
|
|
TEXT_EXTENSIONS = {".txt", ".log", ".yml", ".yaml", ".toml", ".ini", ".cfg",
|
|
".sh", ".bash", ".py", ".js", ".ts", ".html", ".css",
|
|
".xml", ".rst", ".tex", ".sql", ".conf", ".env"}
|
|
if ext in TEXT_EXTENSIONS or ext == ".md":
|
|
pass # handled below or by markdown section
|
|
|
|
if ext == ".md":
|
|
post = parse_markdown_file(raw)
|
|
|
|
# Extract metadata using shared indexer logic
|
|
tags = _extract_tags(post)
|
|
|
|
title = post.metadata.get("title", file_path.stem.replace("-", " ").replace("_", " "))
|
|
html_content = _render_markdown(post.content, vault_name, file_path)
|
|
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": str(title),
|
|
"tags": tags,
|
|
"frontmatter": dict(post.metadata) if post.metadata else {},
|
|
"html": html_content,
|
|
"raw_length": len(raw),
|
|
"extension": ext,
|
|
"is_markdown": True,
|
|
}
|
|
else:
|
|
# Non-markdown: wrap in syntax-highlighted code block
|
|
lang = EXT_TO_LANG.get(ext, "")
|
|
if not lang:
|
|
# Fichiers sans extension usuels (Dockerfile, Makefile, etc.)
|
|
NAME_TO_LANG = {
|
|
"dockerfile": "dockerfile", "makefile": "makefile",
|
|
"cmakelists.txt": "cmake", "jenkinsfile": "groovy",
|
|
"vagrantfile": "ruby", "rakefile": "ruby", "gemfile": "ruby",
|
|
"procfile": "plaintext", "bashrc": "bash", "bash_profile": "bash",
|
|
"zshrc": "bash", "profile": "bash", "gitignore": "plaintext",
|
|
}
|
|
lang = NAME_TO_LANG.get(file_path.name.lower(), "plaintext")
|
|
escaped = html_mod.escape(raw)
|
|
html_content = f'<pre><code class="language-{lang}">{escaped}</code></pre>'
|
|
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": html_content,
|
|
"raw_length": len(raw),
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
}
|