"""File browsing & reading endpoints (ROADMAP #85, tranche 6a). Handlers déplacés depuis :mod:`backend.main` sans changement de comportement : mêmes chemins (``/api/browse/*``, ``/api/file/*`` en lecture), mêmes modèles de réponse (déménagés dans :mod:`backend.schemas`), mêmes dépendances d'authentification. Adaptations strictement équivalentes : - ``_resolve_safe_path`` → :mod:`backend.services.paths` (pass-through). - ``_render_markdown`` vient de :mod:`backend.render` (#85 T9, sans cycle d'import). - ``_content_disposition`` / ``_media_max_inline_bytes`` / ``EXT_TO_LANG`` ont déménagé : helpers partagés dans :mod:`backend.routers.helpers` (``EXT_TO_LANG`` n'était utilisé que par la vue fichier). """ import html as html_mod import logging from pathlib import Path from urllib.parse import quote from fastapi import APIRouter, Depends, HTTPException, Query from fastapi.responses import FileResponse from backend.auth.middleware import check_vault_access, require_auth from backend.history import record_open from backend.indexer import ( _extract_tags, get_backlinks, get_vault_data, parse_markdown_file, ) from backend.media_types import is_audio, is_image, is_video, media_mime_type from backend.render import _render_markdown from backend.routers.helpers import media_max_inline_bytes from backend.schemas import ( BacklinksResponse, BrowseResponse, FileContentResponse, FileRawResponse, XlsxDashboardResponse, XlsxSheetWindowResponse, ) from backend.services.files import read_raw_file from backend.services.mutations import file_revision from backend.services.paths import resolve_safe_path from backend.services.vaults import browse_directory, get_vault_root from backend.share import list_shares logger = logging.getLogger("obsigate") def _resolve_shared_file(vault_name: str, path: str, username: str | None): """#196 — resolve ``home-/Partage/`` to the real source file. Returns ``(real_vault, real_path)`` when *path* is a received share mounted in the user's personal folder, else ``None``. Read-only: only a recipient (or the share creator, whose own file is already accessible) gets a mapping — a share directed to someone else never resolves here. Chemin canonique : ``Partage//`` (token = lève l'ambiguïté de deux partages au même nom). Fallback par nom seul pour les liens ne transportant pas le token. """ if not username or not vault_name.startswith("home-") or not path.startswith("Partage/"): return None owner = vault_name[len("home-"):] if owner != username: return None # Chemin virtuel canonique : Partage// — le token lève # l'ambiguïté (deux partages, même nom de base). Fallback : match par # nom pour les liens ne portant pas le token. rest = path.split("/", 1)[1] parts = rest.split("/", 1) if len(parts) == 2 and len(parts[0]) >= 20: # token (64 hex) vs nom de fichier token, _name = parts for s in list_shares(user=username): if s.get("token") == token and s.get("created_by") != username: return s["vault"], s["path"] return None for s in list_shares(user=username): if s.get("created_by") == username: continue if (s.get("path") or "").split("/")[-1] == rest: return s["vault"], s["path"] return None # Map file extensions to highlight.js language hints EXT_TO_LANG = { ".py": "python", ".js": "javascript", ".ts": "typescript", ".jsx": "jsx", ".tsx": "tsx", ".sh": "bash", ".bash": "bash", ".zsh": "bash", ".fish": "fish", ".bat": "batch", ".cmd": "batch", ".ps1": "powershell", ".json": "json", ".yaml": "yaml", ".yml": "yaml", ".toml": "toml", ".xml": "xml", ".csv": "plaintext", ".cfg": "ini", ".ini": "ini", ".conf": "ini", ".env": "bash", ".html": "html", ".css": "css", ".scss": "scss", ".less": "less", ".java": "java", ".c": "c", ".cpp": "cpp", ".h": "c", ".hpp": "cpp", ".cs": "csharp", ".go": "go", ".rs": "rust", ".rb": "ruby", ".php": "php", ".sql": "sql", ".r": "r", ".swift": "swift", ".kt": "kotlin", ".txt": "plaintext", ".log": "plaintext", ".lua": "lua", ".pl": "perl", ".pm": "perl", ".ex": "elixir", ".exs": "elixir", ".dart": "dart", ".tf": "haskell", ".gradle": "groovy", ".groovy": "groovy", ".graphql": "graphql", ".gql": "graphql", ".prisma": "sql", ".proto": "c", ".vb": "basic", ".asm": "x86asm", ".s": "armasm", ".vue": "xml", ".svelte": "xml", ".astro": "xml", ".properties": "ini", ".service": "ini", ".hosts": "ini", ".ksh": "bash", ".dockerfile": "dockerfile", ".makefile": "makefile", ".cmake": "cmake", } router = APIRouter(tags=["files"]) @router.get("/api/browse/{vault_name}", response_model=BrowseResponse) async def api_browse(vault_name: str, path: str = "", current_user=Depends(require_auth)): """Browse directories and files in a vault at a given path level. Returns sorted entries (directories first, then files) with metadata. Hidden files/directories (starting with ``"."`` ) are excluded. Args: vault_name: Name of the vault to browse. path: Relative directory path within the vault (empty = root). Returns: ``BrowseResponse`` with vault name, path, and item list. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") return browse_directory(vault_name, path) @router.get("/api/file/{vault_name}/raw", response_model=FileRawResponse) async def api_file_raw(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)): """Return raw file content as plain text. Args: vault_name: Name of the vault. path: Relative file path within the vault. Returns: ``FileRawResponse`` with vault, path, and raw text content. """ shared = _resolve_shared_file(vault_name, path, current_user.get("username")) if shared: # Partage dirigé = autorisation (cf. api_file). vault_name, path = shared elif not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") return read_raw_file(vault_name, path) @router.get("/api/file/{vault_name}/download", response_class=FileResponse) async def api_file_download(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)): """Download a file as an attachment. Args: vault_name: Name of the vault. path: Relative file path within the vault. Returns: ``FileResponse`` with ``application/octet-stream`` content-type. """ shared = _resolve_shared_file(vault_name, path, current_user.get("username")) if shared: # Partage dirigé = autorisation (cf. api_file). vault_name, path = shared elif not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) file_path = resolve_safe_path(vault_root, path) if not file_path.exists() or not file_path.is_file(): raise HTTPException(status_code=404, detail=f"File not found: {path}") # Record history record_open(current_user.get("username"), vault_name, path) return FileResponse( path=str(file_path), filename=file_path.name, media_type="application/octet-stream", ) @router.get("/api/file/{vault_name}/backlinks", response_model=BacklinksResponse) async def api_file_backlinks( vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth), ): """Get backlinks (files linking to this file via wikilinks). Returns a list of files that contain `[[wikilinks]]` pointing to the requested file, across all accessible vaults. Args: vault_name: Name of the vault containing the target file. path: Relative path of the target file within the vault. Returns: ``{"vault": str, "path": str, "backlinks": [...]}`` """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") backlinks = get_backlinks(vault_name, path) # Filter by user-accessible vaults (#194 : check_vault_access, "*" sans # les dossiers persos). backlinks = [b for b in backlinks if check_vault_access(b["vault"], current_user)] return { "vault": vault_name, "path": path, "backlinks": backlinks, "total": len(backlinks), } @router.get( "/api/file/{vault_name}/xlsx/dashboard", response_model=XlsxDashboardResponse ) def api_file_xlsx_dashboard( vault_name: str, path: str = Query(..., description="Relative path to the .xlsx workbook"), current_user=Depends(require_auth), ): """Return the dashboard metadata of an .xlsx workbook (#153 A17). Named ranges (workbook- or sheet-scoped), chart/pivot object counts and per-sheet KPI stats (non-empty cells, rows/cols coverage, formulas, numeric cells, first numeric values as KPI cards). Read-only, bounded by the 500x40 render caps; never raises for an unreadable workbook — an empty payload comes back and the viewer hides the panel. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") _vault_root = get_vault_root(vault_name) file_path = resolve_safe_path(_vault_root, path) if not file_path.is_file(): raise HTTPException(status_code=404, detail=f"File not found: {path}") if file_path.suffix.lower() not in (".xlsx", ".xlsm"): raise HTTPException( status_code=415, detail="Le fichier n'est pas un classeur .xlsx/.xlsm" ) from backend.xlsx_reader import read_workbook_dashboard try: dashboard = read_workbook_dashboard(file_path) except Exception as e: logger.error(f"XLSX dashboard read error for {path}: {e}") raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}") return { "vault": vault_name, "path": path, **dashboard, } @router.get( "/api/file/{vault_name}/xlsx/sheet", response_model=XlsxSheetWindowResponse ) def api_file_xlsx_sheet( vault_name: str, path: str = Query(..., description="Relative path to the .xlsx/.xlsm file"), sheet: str = Query(..., description="Sheet name (as shown in the viewer tab)"), offset: int = Query(0, ge=0, description="0-based index of the first row to return"), limit: int = Query( 200, ge=1, le=1000, description="Rows to return (server-capped)" ), current_user=Depends(require_auth), ): """Return a window of rows of one sheet of an .xlsx/.xlsm workbook (#153 A9). Backs the viewer's lazy loading: instead of every sheet in a single JSON payload, the client asks for the block it is about to display. The row numbers and the ``data-cell`` references are the real A1 coordinates of the sheet, so a window behaves like the full render (editing a cell in it targets the right cell). The response also carries ``total_rows``/``total_cols`` and the ``truncated`` flag, so the client can say what is hidden behind the 500x40 render caps instead of silently hiding it. Args: vault_name: Name of the vault. path: Relative path of the .xlsx file within the vault. sheet: Sheet name; **404** if the workbook has no such sheet. offset: 0-based index of the first row to return. limit: Rows to return, capped server-side at 1000. Returns: ``XlsxSheetWindowResponse`` with the rendered ``html`` of the window. Raises: HTTPException: 403 (vault access), 404 (vault, file or sheet unknown), 415 (not an .xlsx/.xlsm file), 500 (unreadable workbook). """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") file_path = resolve_safe_path(Path(vault_data["path"]), path) if not file_path.is_file(): raise HTTPException(status_code=404, detail=f"File not found: {path}") # BUG-097 — a .xlsm rides the same editable viewer (and its lazy loading), # so its row windows must be servable too; other formats stay refused. if file_path.suffix.lower() not in (".xlsx", ".xlsm"): raise HTTPException( status_code=415, detail="Le fichier n'est pas un classeur .xlsx/.xlsm" ) # Import tardif : openpyxl n'est chargé que si un .xlsx est réellement demandé. from backend.xlsx_reader import read_sheet_window try: window = read_sheet_window(file_path, sheet, offset=offset, limit=limit) except Exception as e: logger.error(f"XLSX sheet read error for {path}: {e}") raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}") if window is None: raise HTTPException(status_code=404, detail=f"Feuille introuvable: {sheet}") return {"vault": vault_name, "path": path, **window} @router.get("/api/file/{vault_name}", response_model=FileContentResponse) async def api_file(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)): """Return rendered HTML and metadata for a file. Markdown files are parsed for frontmatter, rendered with wikilink support, and returned with extracted tags. Other supported file types are syntax-highlighted as code blocks. Args: vault_name: Name of the vault. path: Relative file path within the vault. Returns: ``FileContentResponse`` with HTML, metadata, and tags. """ # #196 — fichier reçu par partage dirigé : home-/Partage/ # est résolu vers le fichier source (lecture seule, viewer standard). # Résolu AVANT l'ACL vault : le dossier est virtuel, l'autorisation réelle # est l'appartenance au partage (vérifiée dans le resolver). shared = _resolve_shared_file(vault_name, path, current_user.get("username")) if shared: # Le partage dirigé EST l'autorisation : le destinataire n'a par # définition pas accès au vault source — on ne passe PAS par l'ACL. vault_name, path = shared elif not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) file_path = resolve_safe_path(vault_root, path) if not file_path.exists() or not file_path.is_file(): raise HTTPException(status_code=404, detail=f"File not found: {path}") # Record history record_open(current_user.get("username"), vault_name, path, title=file_path.name) ext = file_path.suffix.lower() # === PDF: special handling before read_text (binary file) === if ext == ".pdf": try: from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text, extract_pdf_toc pdf_text = extract_pdf_text(file_path, max_chars=100000) pdf_meta = extract_pdf_metadata(file_path) pdf_toc = extract_pdf_toc(file_path) size = file_path.stat().st_size return { "vault": vault_name, "path": path, "title": pdf_meta.get("title") or file_path.name, "tags": [], "frontmatter": {}, "html": f"

PDF — {pdf_meta.get('pages', '?')} pages

{pdf_text[:5000]}
", "raw_length": size, "extension": ext, "is_markdown": False, "is_pdf": True, "unsupported": False, "pdf_metadata": pdf_meta, "pdf_toc": pdf_toc, "size_bytes": size, } except Exception as e: logger.error(f"PDF read error for {path}: {e}") raise HTTPException(status_code=500, detail=f"Error reading PDF: {e!s}") # === Excel .xlsx: render sheets as HTML tables (binary, before read_text) === if ext == ".xlsx": try: from backend.xlsx_reader import inspect_workbook, render_sheets # #153 A15 — every sheet dict already carries its styles, aligns, # merges and freeze anchor (read_workbook_meta, one normal-mode # load inside render_sheets). sheets = render_sheets(file_path) size = file_path.stat().st_size return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": sheets[0]["html"] if sheets else "", "raw_length": size, "extension": ext, "is_markdown": False, "is_xlsx": True, "xlsx_sheets": sheets, # #156-A12 — optimistic-concurrency token: the viewer sends it # back as `if_match` so another writer cannot be overwritten in # silence (409 `conflict` instead). "xlsx_revision": file_revision(file_path), # #153 A1 — parts a save would drop; the viewer warns and asks # for an explicit confirmation before forcing the write. "xlsx_lossy_features": inspect_workbook(file_path), "unsupported": False, "size_bytes": size, } except Exception as e: logger.error(f"XLSX read error for {path}: {e}") raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}") # === Images: return as viewable image === if is_image(ext): size = file_path.stat().st_size mime = media_mime_type(str(file_path)) # #108-B1 — the raw endpoint returns JSON (FileRawResponse), so the # standalone must point to /api/image, which serves the bytes # with the right MIME type. Paths are URL-encoded (accents, spaces). img_url = f"/api/image/{quote(vault_name, safe='')}?path={quote(path, safe='')}" html = ( f'
' f'' f'
' ) return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": html, "raw_length": size, "extension": ext, "is_markdown": False, "is_image": True, "image_mime": mime, "size_bytes": size, } # === Audio / Video: HTML5 players streamed from /api/media (roadmap #109) === if is_audio(ext) or is_video(ext): size = file_path.stat().st_size mime = media_mime_type(str(file_path)) media_kind = "audio" if is_audio(ext) else "video" # #109-A3 — beyond the inline limit the viewer falls back to download # (a single uvicorn worker must not be pinned by multi-GB media). if size > media_max_inline_bytes(): return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": "", "raw_length": size, "extension": ext, "is_markdown": False, "unsupported": True, "media_too_large": True, "size_bytes": size, } # #109-A2 — byte-range endpoint: enables scrub and is required by Safari. stream_url = f"/api/media/{quote(vault_name, safe='')}?path={quote(path, safe='')}" if media_kind == "audio": html = ( f'
' f'' f'
' ) else: html = ( f'
' f'' f'
' ) return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": html, "raw_length": size, "extension": ext, "is_markdown": False, "is_audio": media_kind == "audio", "is_video": media_kind == "video", "media_mime": mime, "stream_url": stream_url, "size_bytes": size, } try: raw = file_path.read_text(encoding="utf-8", errors="replace") except PermissionError as e: logger.error(f"Permission denied reading file {path}: {e}") raise HTTPException(status_code=403, detail=f"Permission denied: cannot read file {path}") except UnicodeDecodeError: # Binary / unsupported file — return structured info with download option size = file_path.stat().st_size return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": "", "raw_length": size, "extension": ext, "is_markdown": False, "unsupported": True, "size_bytes": size, } except Exception as e: logger.error(f"Unexpected error reading file {path}: {e}") raise HTTPException(status_code=500, detail=f"Error reading file: {e!s}") # === Excel .xlsm: same editable viewer as .xlsx, macros preserved on save === if ext == ".xlsm": try: from backend.xlsx_reader import inspect_workbook, render_sheets sheets = render_sheets(file_path) size = file_path.stat().st_size return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": sheets[0]["html"] if sheets else "", "raw_length": size, "extension": ext, "is_markdown": False, "is_xlsx": True, "xlsx_sheets": sheets, "xlsx_revision": file_revision(file_path), # Macros are NOT lossy for .xlsm: keep_vba re-serializes them # (an empty LOSSY probe is what makes the save gate pass). "xlsx_lossy_features": [], "unsupported": False, "size_bytes": size, } except Exception as e: logger.error(f"XLSX read error for {path}: {e}") raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}") # === Legacy/ODF spreadsheets (.xls, .ods): read-only table view === if ext in (".xls", ".ods"): try: from backend.xlsx_reader import render_legacy_workbook sheets = render_legacy_workbook(file_path, ext) size = file_path.stat().st_size return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": sheets[0]["html"] if sheets else "", "raw_length": size, "extension": ext, "is_markdown": False, "is_xlsx": True, "xlsx_readonly": True, "xlsx_sheets": sheets, "unsupported": False, "size_bytes": size, } except Exception as e: logger.error(f"Spreadsheet read error for {path}: {e}") raise HTTPException(status_code=500, detail=f"Error reading spreadsheet: {e!s}") # === CSV: spreadsheet-style table (same shape as the xlsx viewer) === if ext == ".csv": from backend.xlsx_reader import render_csv_table html = render_csv_table(raw) return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": html, "raw_length": len(raw), "extension": ext, "is_markdown": False, "is_csv": True, # #156-A12 — same stale-write guard as the workbooks. "xlsx_revision": file_revision(file_path), } # === JSON: syntax-highlighted display === if ext == ".json": import json as json_mod try: parsed = json_mod.loads(raw) formatted = json_mod.dumps(parsed, indent=2, ensure_ascii=False) except json_mod.JSONDecodeError: formatted = raw html = f"
{html_mod.escape(formatted)}
" return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": html, "raw_length": len(raw), "extension": ext, "is_markdown": False, "is_json": True, } # === Excalidraw .excalidraw.md (Obsidian plugin format) === if path.lower().endswith(".excalidraw.md"): import re as re_mod raw_lower = file_path.read_text(encoding="utf-8", errors="replace") # Check for excalidraw-plugin in frontmatter or body if "excalidraw-plugin:" in raw_lower: # Extract compressed JSON block match = re_mod.search(r'```compressed-json\n(.*?)\n```', raw_lower, re_mod.DOTALL) if match: compressed = match.group(1).strip() return { "vault": vault_name, "path": path, "title": file_path.name.replace(".excalidraw.md", ""), "tags": [], "frontmatter": {}, "html": "", "raw_length": len(raw_lower), "extension": ".excalidraw.md", "is_markdown": False, "is_excalidraw": True, "excalidraw_data_compressed": compressed, } # Fallback: treat as regular markdown raw = raw_lower if ext == ".excalidraw": import json as json_mod try: parsed = json_mod.loads(raw) except json_mod.JSONDecodeError: parsed = None if parsed and parsed.get("type") == "excalidraw": return { "vault": vault_name, "path": path, "title": parsed.get("appState", {}).get("name") or file_path.name, "tags": [], "frontmatter": {}, "html": "", "raw_length": len(raw), "extension": ext, "is_markdown": False, "is_excalidraw": True, "excalidraw_data": { "elements": parsed.get("elements", []), "appState": parsed.get("appState", {}), "files": parsed.get("files", {}), }, } else: # Not a valid Excalidraw file — fall through to text viewer pass # === Plain text / other readable files === TEXT_EXTENSIONS = {".txt", ".log", ".yml", ".yaml", ".toml", ".ini", ".cfg", ".sh", ".bash", ".py", ".js", ".ts", ".html", ".css", ".xml", ".rst", ".tex", ".sql", ".conf", ".env"} if ext in TEXT_EXTENSIONS or ext == ".md": pass # handled below or by markdown section if ext == ".md": post = parse_markdown_file(raw) # Extract metadata using shared indexer logic tags = _extract_tags(post) title = post.metadata.get("title", file_path.stem.replace("-", " ").replace("_", " ")) html_content = _render_markdown(post.content, vault_name, file_path, click_to_copy=True) return { "vault": vault_name, "path": path, "title": str(title), "tags": tags, "frontmatter": dict(post.metadata) if post.metadata else {}, "html": html_content, "raw_length": len(raw), "extension": ext, "is_markdown": True, } else: # Non-markdown: wrap in syntax-highlighted code block lang = EXT_TO_LANG.get(ext, "") if not lang: # Fichiers sans extension usuels (Dockerfile, Makefile, etc.) NAME_TO_LANG = { "dockerfile": "dockerfile", "makefile": "makefile", "cmakelists.txt": "cmake", "jenkinsfile": "groovy", "vagrantfile": "ruby", "rakefile": "ruby", "gemfile": "ruby", "procfile": "plaintext", "bashrc": "bash", "bash_profile": "bash", "zshrc": "bash", "profile": "bash", "gitignore": "plaintext", } lang = NAME_TO_LANG.get(file_path.name.lower(), "plaintext") escaped = html_mod.escape(raw) html_content = f'
{escaped}
' return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": html_content, "raw_length": len(raw), "extension": ext, "is_markdown": False, }