Files
ObsiGate/backend/routers/files_read.py
T
bruno c4b8e66206 feat: tableau de bord classeur - plages nommees, TCD/graphiques et KPI par feuille #153
A17 — nouveau endpoint GET /api/file/{vault}/xlsx/dashboard (read_workbook_
dashboard : plages nommees avec portee depuis defined_names read_only,
comptage graphiques/TCD par parts OPC, stats par feuille bornées 500x40 :
cellules/lignes/colonnes/formules/numerique + 8 premieres valeurs en cartes
KPI) et panneau frontend toggled depuis la toolbar (table des plages,
cartes KPI par feuille, hint actions IA). Bouton absent pour .csv et
formats en lecture seule ; le menu structure est saute quand le bouton
n'existe pas. i18n FR/EN (xlsx.dashboard_*), 8 tests backend + 2 tests
JSDOM + contre-preuve (5 echecs sur neutralisation), ruff/mypy 0.

🤖 Generated with Codebuff
Co-Authored-By: Codebuff <[email protected]>
2026-09-28 16:40:09 -04:00

679 lines
27 KiB
Python

"""File browsing & reading endpoints (ROADMAP #85, tranche 6a).
Handlers déplacés depuis :mod:`backend.main` sans changement de
comportement : mêmes chemins (``/api/browse/*``, ``/api/file/*`` en
lecture), mêmes modèles de réponse (déménagés dans
:mod:`backend.schemas`), mêmes dépendances d'authentification.
Adaptations strictement équivalentes :
- ``_resolve_safe_path`` → :mod:`backend.services.paths` (pass-through).
- ``_render_markdown`` vient de :mod:`backend.render` (#85 T9, sans cycle
d'import).
- ``_content_disposition`` / ``_media_max_inline_bytes`` / ``EXT_TO_LANG``
ont déménagé : helpers partagés dans :mod:`backend.routers.helpers`
(``EXT_TO_LANG`` n'était utilisé que par la vue fichier).
"""
import html as html_mod
import logging
from pathlib import Path
from urllib.parse import quote
from fastapi import APIRouter, Depends, HTTPException, Query
from fastapi.responses import FileResponse
from backend.auth.middleware import check_vault_access, require_auth
from backend.history import record_open
from backend.indexer import (
_extract_tags,
get_backlinks,
get_vault_data,
parse_markdown_file,
)
from backend.media_types import is_audio, is_image, is_video, media_mime_type
from backend.render import _render_markdown
from backend.routers.helpers import media_max_inline_bytes
from backend.schemas import (
BacklinksResponse,
BrowseResponse,
FileContentResponse,
FileRawResponse,
XlsxDashboardResponse,
XlsxSheetWindowResponse,
)
from backend.services.files import read_raw_file
from backend.services.paths import resolve_safe_path
from backend.services.vaults import browse_directory, get_vault_root
logger = logging.getLogger("obsigate")
# Map file extensions to highlight.js language hints
EXT_TO_LANG = {
".py": "python", ".js": "javascript", ".ts": "typescript",
".jsx": "jsx", ".tsx": "tsx", ".sh": "bash", ".bash": "bash",
".zsh": "bash", ".fish": "fish", ".bat": "batch", ".cmd": "batch",
".ps1": "powershell", ".json": "json", ".yaml": "yaml", ".yml": "yaml",
".toml": "toml", ".xml": "xml", ".csv": "plaintext",
".cfg": "ini", ".ini": "ini", ".conf": "ini", ".env": "bash",
".html": "html", ".css": "css", ".scss": "scss", ".less": "less",
".java": "java", ".c": "c", ".cpp": "cpp", ".h": "c", ".hpp": "cpp",
".cs": "csharp", ".go": "go", ".rs": "rust", ".rb": "ruby",
".php": "php", ".sql": "sql", ".r": "r", ".swift": "swift",
".kt": "kotlin", ".txt": "plaintext", ".log": "plaintext",
".lua": "lua", ".pl": "perl", ".pm": "perl", ".ex": "elixir", ".exs": "elixir",
".dart": "dart", ".tf": "haskell", ".gradle": "groovy", ".groovy": "groovy",
".graphql": "graphql", ".gql": "graphql", ".prisma": "sql", ".proto": "c",
".vb": "basic", ".asm": "x86asm", ".s": "armasm",
".vue": "xml", ".svelte": "xml", ".astro": "xml",
".properties": "ini", ".service": "ini", ".hosts": "ini",
".ksh": "bash", ".dockerfile": "dockerfile",
".makefile": "makefile", ".cmake": "cmake",
}
router = APIRouter(tags=["files"])
@router.get("/api/browse/{vault_name}", response_model=BrowseResponse)
async def api_browse(vault_name: str, path: str = "", current_user=Depends(require_auth)):
"""Browse directories and files in a vault at a given path level.
Returns sorted entries (directories first, then files) with metadata.
Hidden files/directories (starting with ``"."`` ) are excluded.
Args:
vault_name: Name of the vault to browse.
path: Relative directory path within the vault (empty = root).
Returns:
``BrowseResponse`` with vault name, path, and item list.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
return browse_directory(vault_name, path)
@router.get("/api/file/{vault_name}/raw", response_model=FileRawResponse)
async def api_file_raw(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
"""Return raw file content as plain text.
Args:
vault_name: Name of the vault.
path: Relative file path within the vault.
Returns:
``FileRawResponse`` with vault, path, and raw text content.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
return read_raw_file(vault_name, path)
@router.get("/api/file/{vault_name}/download", response_class=FileResponse)
async def api_file_download(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
"""Download a file as an attachment.
Args:
vault_name: Name of the vault.
path: Relative file path within the vault.
Returns:
``FileResponse`` with ``application/octet-stream`` content-type.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
vault_data = get_vault_data(vault_name)
if not vault_data:
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
vault_root = Path(vault_data["path"])
file_path = resolve_safe_path(vault_root, path)
if not file_path.exists() or not file_path.is_file():
raise HTTPException(status_code=404, detail=f"File not found: {path}")
# Record history
record_open(current_user.get("username"), vault_name, path)
return FileResponse(
path=str(file_path),
filename=file_path.name,
media_type="application/octet-stream",
)
@router.get("/api/file/{vault_name}/backlinks", response_model=BacklinksResponse)
async def api_file_backlinks(
vault_name: str,
path: str = Query(..., description="Relative path to file"),
current_user=Depends(require_auth),
):
"""Get backlinks (files linking to this file via wikilinks).
Returns a list of files that contain `[[wikilinks]]` pointing
to the requested file, across all accessible vaults.
Args:
vault_name: Name of the vault containing the target file.
path: Relative path of the target file within the vault.
Returns:
``{"vault": str, "path": str, "backlinks": [...]}``
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
vault_data = get_vault_data(vault_name)
if not vault_data:
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
user_vaults = current_user.get("_token_vaults") or current_user.get("vaults", [])
backlinks = get_backlinks(vault_name, path)
# Filter by user-accessible vaults
if "*" not in user_vaults:
backlinks = [b for b in backlinks if b["vault"] in user_vaults]
return {
"vault": vault_name,
"path": path,
"backlinks": backlinks,
"total": len(backlinks),
}
@router.get(
"/api/file/{vault_name}/xlsx/dashboard", response_model=XlsxDashboardResponse
)
def api_file_xlsx_dashboard(
vault_name: str,
path: str = Query(..., description="Relative path to the .xlsx workbook"),
current_user=Depends(require_auth),
):
"""Return the dashboard metadata of an .xlsx workbook (#153 A17).
Named ranges (workbook- or sheet-scoped), chart/pivot object counts and
per-sheet KPI stats (non-empty cells, rows/cols coverage, formulas,
numeric cells, first numeric values as KPI cards). Read-only, bounded by
the 500x40 render caps; never raises for an unreadable workbook — an
empty payload comes back and the viewer hides the panel.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
_vault_root = get_vault_root(vault_name)
file_path = resolve_safe_path(_vault_root, path)
if not file_path.is_file():
raise HTTPException(status_code=404, detail=f"File not found: {path}")
if file_path.suffix.lower() not in (".xlsx", ".xlsm"):
raise HTTPException(
status_code=415, detail="Le fichier n'est pas un classeur .xlsx/.xlsm"
)
from backend.xlsx_reader import read_workbook_dashboard
try:
dashboard = read_workbook_dashboard(file_path)
except Exception as e:
logger.error(f"XLSX dashboard read error for {path}: {e}")
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
return {
"vault": vault_name,
"path": path,
**dashboard,
}
@router.get(
"/api/file/{vault_name}/xlsx/sheet", response_model=XlsxSheetWindowResponse
)
def api_file_xlsx_sheet(
vault_name: str,
path: str = Query(..., description="Relative path to the .xlsx file"),
sheet: str = Query(..., description="Sheet name (as shown in the viewer tab)"),
offset: int = Query(0, ge=0, description="0-based index of the first row to return"),
limit: int = Query(
200, ge=1, le=1000, description="Rows to return (server-capped)"
),
current_user=Depends(require_auth),
):
"""Return a window of rows of one sheet of an .xlsx workbook (#153 A9).
Backs the viewer's lazy loading: instead of every sheet in a single JSON
payload, the client asks for the block it is about to display. The row
numbers and the ``data-cell`` references are the real A1 coordinates of the
sheet, so a window behaves like the full render (editing a cell in it
targets the right cell).
The response also carries ``total_rows``/``total_cols`` and the ``truncated``
flag, so the client can say what is hidden behind the 500x40 render caps
instead of silently hiding it.
Args:
vault_name: Name of the vault.
path: Relative path of the .xlsx file within the vault.
sheet: Sheet name; **404** if the workbook has no such sheet.
offset: 0-based index of the first row to return.
limit: Rows to return, capped server-side at 1000.
Returns:
``XlsxSheetWindowResponse`` with the rendered ``html`` of the window.
Raises:
HTTPException: 403 (vault access), 404 (vault, file or sheet unknown),
415 (not an .xlsx file), 500 (unreadable workbook).
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
vault_data = get_vault_data(vault_name)
if not vault_data:
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
file_path = resolve_safe_path(Path(vault_data["path"]), path)
if not file_path.is_file():
raise HTTPException(status_code=404, detail=f"File not found: {path}")
if file_path.suffix.lower() != ".xlsx":
raise HTTPException(status_code=415, detail="Le fichier n'est pas un classeur .xlsx")
# Import tardif : openpyxl n'est chargé que si un .xlsx est réellement demandé.
from backend.xlsx_reader import read_sheet_window
try:
window = read_sheet_window(file_path, sheet, offset=offset, limit=limit)
except Exception as e:
logger.error(f"XLSX sheet read error for {path}: {e}")
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
if window is None:
raise HTTPException(status_code=404, detail=f"Feuille introuvable: {sheet}")
return {"vault": vault_name, "path": path, **window}
@router.get("/api/file/{vault_name}", response_model=FileContentResponse)
async def api_file(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
"""Return rendered HTML and metadata for a file.
Markdown files are parsed for frontmatter, rendered with wikilink
support, and returned with extracted tags. Other supported file
types are syntax-highlighted as code blocks.
Args:
vault_name: Name of the vault.
path: Relative file path within the vault.
Returns:
``FileContentResponse`` with HTML, metadata, and tags.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
vault_data = get_vault_data(vault_name)
if not vault_data:
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
vault_root = Path(vault_data["path"])
file_path = resolve_safe_path(vault_root, path)
if not file_path.exists() or not file_path.is_file():
raise HTTPException(status_code=404, detail=f"File not found: {path}")
# Record history
record_open(current_user.get("username"), vault_name, path, title=file_path.name)
ext = file_path.suffix.lower()
# === PDF: special handling before read_text (binary file) ===
if ext == ".pdf":
try:
from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text, extract_pdf_toc
pdf_text = extract_pdf_text(file_path, max_chars=100000)
pdf_meta = extract_pdf_metadata(file_path)
pdf_toc = extract_pdf_toc(file_path)
size = file_path.stat().st_size
return {
"vault": vault_name,
"path": path,
"title": pdf_meta.get("title") or file_path.name,
"tags": [],
"frontmatter": {},
"html": f"<div class='pdf-viewer'><p>PDF — {pdf_meta.get('pages', '?')} pages</p><pre>{pdf_text[:5000]}</pre></div>",
"raw_length": size,
"extension": ext,
"is_markdown": False,
"is_pdf": True,
"unsupported": False,
"pdf_metadata": pdf_meta,
"pdf_toc": pdf_toc,
"size_bytes": size,
}
except Exception as e:
logger.error(f"PDF read error for {path}: {e}")
raise HTTPException(status_code=500, detail=f"Error reading PDF: {e!s}")
# === Excel .xlsx: render sheets as HTML tables (binary, before read_text) ===
if ext == ".xlsx":
try:
from backend.xlsx_reader import inspect_workbook, render_sheets
# #153 A15 — every sheet dict already carries its styles, aligns,
# merges and freeze anchor (read_workbook_meta, one normal-mode
# load inside render_sheets).
sheets = render_sheets(file_path)
size = file_path.stat().st_size
return {
"vault": vault_name,
"path": path,
"title": file_path.name,
"tags": [],
"frontmatter": {},
"html": sheets[0]["html"] if sheets else "",
"raw_length": size,
"extension": ext,
"is_markdown": False,
"is_xlsx": True,
"xlsx_sheets": sheets,
# #153 A1 — parts a save would drop; the viewer warns and asks
# for an explicit confirmation before forcing the write.
"xlsx_lossy_features": inspect_workbook(file_path),
"unsupported": False,
"size_bytes": size,
}
except Exception as e:
logger.error(f"XLSX read error for {path}: {e}")
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
# === Images: return as viewable image ===
if is_image(ext):
size = file_path.stat().st_size
mime = media_mime_type(str(file_path))
# #108-B1 — the raw endpoint returns JSON (FileRawResponse), so the
# standalone <img> must point to /api/image, which serves the bytes
# with the right MIME type. Paths are URL-encoded (accents, spaces).
img_url = f"/api/image/{quote(vault_name, safe='')}?path={quote(path, safe='')}"
html = (
f'<div class="image-viewer">'
f'<img src="{img_url}" '
f'alt="{html_mod.escape(file_path.name, quote=True)}" '
f'style="max-width:100%;max-height:80vh;object-fit:contain" />'
f'</div>'
)
return {
"vault": vault_name,
"path": path,
"title": file_path.name,
"tags": [],
"frontmatter": {},
"html": html,
"raw_length": size,
"extension": ext,
"is_markdown": False,
"is_image": True,
"image_mime": mime,
"size_bytes": size,
}
# === Audio / Video: HTML5 players streamed from /api/media (roadmap #109) ===
if is_audio(ext) or is_video(ext):
size = file_path.stat().st_size
mime = media_mime_type(str(file_path))
media_kind = "audio" if is_audio(ext) else "video"
# #109-A3 — beyond the inline limit the viewer falls back to download
# (a single uvicorn worker must not be pinned by multi-GB media).
if size > media_max_inline_bytes():
return {
"vault": vault_name,
"path": path,
"title": file_path.name,
"tags": [],
"frontmatter": {},
"html": "",
"raw_length": size,
"extension": ext,
"is_markdown": False,
"unsupported": True,
"media_too_large": True,
"size_bytes": size,
}
# #109-A2 — byte-range endpoint: enables scrub and is required by Safari.
stream_url = f"/api/media/{quote(vault_name, safe='')}?path={quote(path, safe='')}"
if media_kind == "audio":
html = (
f'<div class="audio-viewer">'
f'<audio controls preload="metadata" src="{stream_url}"></audio>'
f'</div>'
)
else:
html = (
f'<div class="video-viewer">'
f'<video controls playsinline preload="metadata" src="{stream_url}"></video>'
f'</div>'
)
return {
"vault": vault_name,
"path": path,
"title": file_path.name,
"tags": [],
"frontmatter": {},
"html": html,
"raw_length": size,
"extension": ext,
"is_markdown": False,
"is_audio": media_kind == "audio",
"is_video": media_kind == "video",
"media_mime": mime,
"stream_url": stream_url,
"size_bytes": size,
}
try:
raw = file_path.read_text(encoding="utf-8", errors="replace")
except PermissionError as e:
logger.error(f"Permission denied reading file {path}: {e}")
raise HTTPException(status_code=403, detail=f"Permission denied: cannot read file {path}")
except UnicodeDecodeError:
# Binary / unsupported file — return structured info with download option
size = file_path.stat().st_size
return {
"vault": vault_name,
"path": path,
"title": file_path.name,
"tags": [],
"frontmatter": {},
"html": "",
"raw_length": size,
"extension": ext,
"is_markdown": False,
"unsupported": True,
"size_bytes": size,
}
except Exception as e:
logger.error(f"Unexpected error reading file {path}: {e}")
raise HTTPException(status_code=500, detail=f"Error reading file: {e!s}")
# === Excel .xlsm: same editable viewer as .xlsx, macros preserved on save ===
if ext == ".xlsm":
try:
from backend.xlsx_reader import inspect_workbook, render_sheets
sheets = render_sheets(file_path)
size = file_path.stat().st_size
return {
"vault": vault_name,
"path": path,
"title": file_path.name,
"tags": [],
"frontmatter": {},
"html": sheets[0]["html"] if sheets else "",
"raw_length": size,
"extension": ext,
"is_markdown": False,
"is_xlsx": True,
"xlsx_sheets": sheets,
# Macros are NOT lossy for .xlsm: keep_vba re-serializes them
# (an empty LOSSY probe is what makes the save gate pass).
"xlsx_lossy_features": [],
"unsupported": False,
"size_bytes": size,
}
except Exception as e:
logger.error(f"XLSX read error for {path}: {e}")
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
# === Legacy/ODF spreadsheets (.xls, .ods): read-only table view ===
if ext in (".xls", ".ods"):
try:
from backend.xlsx_reader import render_legacy_workbook
sheets = render_legacy_workbook(file_path, ext)
size = file_path.stat().st_size
return {
"vault": vault_name,
"path": path,
"title": file_path.name,
"tags": [],
"frontmatter": {},
"html": sheets[0]["html"] if sheets else "",
"raw_length": size,
"extension": ext,
"is_markdown": False,
"is_xlsx": True,
"xlsx_readonly": True,
"xlsx_sheets": sheets,
"unsupported": False,
"size_bytes": size,
}
except Exception as e:
logger.error(f"Spreadsheet read error for {path}: {e}")
raise HTTPException(status_code=500, detail=f"Error reading spreadsheet: {e!s}")
# === CSV: spreadsheet-style table (same shape as the xlsx viewer) ===
if ext == ".csv":
from backend.xlsx_reader import render_csv_table
html = render_csv_table(raw)
return {
"vault": vault_name, "path": path,
"title": file_path.name, "tags": [], "frontmatter": {},
"html": html, "raw_length": len(raw), "extension": ext,
"is_markdown": False, "is_csv": True,
}
# === JSON: syntax-highlighted display ===
if ext == ".json":
import json as json_mod
try:
parsed = json_mod.loads(raw)
formatted = json_mod.dumps(parsed, indent=2, ensure_ascii=False)
except json_mod.JSONDecodeError:
formatted = raw
html = f"<pre class='json-viewer'><code>{html_mod.escape(formatted)}</code></pre>"
return {
"vault": vault_name, "path": path,
"title": file_path.name, "tags": [], "frontmatter": {},
"html": html, "raw_length": len(raw), "extension": ext,
"is_markdown": False, "is_json": True,
}
# === Excalidraw .excalidraw.md (Obsidian plugin format) ===
if path.lower().endswith(".excalidraw.md"):
import re as re_mod
raw_lower = file_path.read_text(encoding="utf-8", errors="replace")
# Check for excalidraw-plugin in frontmatter or body
if "excalidraw-plugin:" in raw_lower:
# Extract compressed JSON block
match = re_mod.search(r'```compressed-json\n(.*?)\n```', raw_lower, re_mod.DOTALL)
if match:
compressed = match.group(1).strip()
return {
"vault": vault_name, "path": path,
"title": file_path.name.replace(".excalidraw.md", ""),
"tags": [], "frontmatter": {},
"html": "", "raw_length": len(raw_lower),
"extension": ".excalidraw.md",
"is_markdown": False,
"is_excalidraw": True,
"excalidraw_data_compressed": compressed,
}
# Fallback: treat as regular markdown
raw = raw_lower
if ext == ".excalidraw":
import json as json_mod
try:
parsed = json_mod.loads(raw)
except json_mod.JSONDecodeError:
parsed = None
if parsed and parsed.get("type") == "excalidraw":
return {
"vault": vault_name,
"path": path,
"title": parsed.get("appState", {}).get("name") or file_path.name,
"tags": [],
"frontmatter": {},
"html": "",
"raw_length": len(raw),
"extension": ext,
"is_markdown": False,
"is_excalidraw": True,
"excalidraw_data": {
"elements": parsed.get("elements", []),
"appState": parsed.get("appState", {}),
"files": parsed.get("files", {}),
},
}
else:
# Not a valid Excalidraw file — fall through to text viewer
pass
# === Plain text / other readable files ===
TEXT_EXTENSIONS = {".txt", ".log", ".yml", ".yaml", ".toml", ".ini", ".cfg",
".sh", ".bash", ".py", ".js", ".ts", ".html", ".css",
".xml", ".rst", ".tex", ".sql", ".conf", ".env"}
if ext in TEXT_EXTENSIONS or ext == ".md":
pass # handled below or by markdown section
if ext == ".md":
post = parse_markdown_file(raw)
# Extract metadata using shared indexer logic
tags = _extract_tags(post)
title = post.metadata.get("title", file_path.stem.replace("-", " ").replace("_", " "))
html_content = _render_markdown(post.content, vault_name, file_path)
return {
"vault": vault_name,
"path": path,
"title": str(title),
"tags": tags,
"frontmatter": dict(post.metadata) if post.metadata else {},
"html": html_content,
"raw_length": len(raw),
"extension": ext,
"is_markdown": True,
}
else:
# Non-markdown: wrap in syntax-highlighted code block
lang = EXT_TO_LANG.get(ext, "")
if not lang:
# Fichiers sans extension usuels (Dockerfile, Makefile, etc.)
NAME_TO_LANG = {
"dockerfile": "dockerfile", "makefile": "makefile",
"cmakelists.txt": "cmake", "jenkinsfile": "groovy",
"vagrantfile": "ruby", "rakefile": "ruby", "gemfile": "ruby",
"procfile": "plaintext", "bashrc": "bash", "bash_profile": "bash",
"zshrc": "bash", "profile": "bash", "gitignore": "plaintext",
}
lang = NAME_TO_LANG.get(file_path.name.lower(), "plaintext")
escaped = html_mod.escape(raw)
html_content = f'<pre><code class="language-{lang}">{escaped}</code></pre>'
return {
"vault": vault_name,
"path": path,
"title": file_path.name,
"tags": [],
"frontmatter": {},
"html": html_content,
"raw_length": len(raw),
"extension": ext,
"is_markdown": False,
}