import asyncio import html as html_mod import json as _json import logging import mimetypes import os import re import secrets import shutil import string import time import urllib.request from concurrent.futures import ThreadPoolExecutor from contextlib import asynccontextmanager from datetime import datetime, timezone from functools import partial from pathlib import Path from typing import Any import frontmatter import mistune from fastapi import Body, Depends, FastAPI, HTTPException, Query, Request, WebSocket from fastapi.responses import FileResponse, HTMLResponse, JSONResponse, Response, StreamingResponse from fastapi.staticfiles import StaticFiles from pydantic import BaseModel, Field from starlette.middleware.base import BaseHTTPMiddleware from backend.attachment_indexer import get_attachment_stats, rescan_vault_attachments from backend.collab import authenticate_websocket, collab_manager from backend.history import ( get_bookmarks, record_open, remove_recent, toggle_bookmark, update_bookmarks_after_rename, update_history_after_rename, ) from backend.image_processor import preprocess_images from backend.indexer import ( _extract_tags, add_vault_to_index, build_index, find_file_in_index, get_backlinks, get_conflicts, get_vault_data, handle_file_move, index, parse_markdown_file, reload_index, remove_single_file, remove_vault_from_index, update_single_file, ) from backend.openapi_docs import ( API_DESCRIPTION, TAGS_METADATA, enrich_openapi_schema, render_api_landing, ) from backend.schemas import ( AIKeyDeleteResponse, AIKeysResponse, AIModelsResponse, AITestResponse, AllVaultSettingsResponse, AppConfigResponse, AttachmentRescanResponse, AttachmentStatsResponse, BacklinksResponse, BackupContentResponse, BackupsAutoResponse, BackupsCompressResponse, BackupsDeletedResponse, BackupsListResponse, BackupsResponse, BookmarksResponse, BookmarkToggleResponse, ConflictResolveResponse, ConflictsResponse, DashboardResponse, DiagnosticsResponse, PdfInfoResponse, RecentResponse, ReplaceResponse, SavedSearch, ShareModel, StatusResponse, VaultActionResponse, VaultFilesResponse, VaultSettingsResponse, VaultsStatusResponse, VaultStatsResponse, WebhookModel, ) from backend.search import ( init_inverted_index, suggest_tags, suggest_titles, ) from backend.semantic_search import init_semantic_index from backend.services.backups import diff_backup as service_diff_backup from backend.services.backups import get_backup_dir as service_get_backup_dir from backend.services.backups import list_backup_files as service_list_backup_files from backend.services.errors import ServiceError from backend.services.files import read_raw_file from backend.services.graph import get_graph as service_get_graph from backend.services.mutations import ( batch_upload_files as service_batch_upload_files, ) from backend.services.mutations import ( create_directory as service_create_directory, ) from backend.services.mutations import ( create_file as service_create_file, ) from backend.services.mutations import ( delete_directory as service_delete_directory, ) from backend.services.mutations import ( delete_file as service_delete_file, ) from backend.services.mutations import ( edit_file as service_edit_file, ) from backend.services.mutations import ( move_path as service_move_path, ) from backend.services.mutations import ( rename_directory as service_rename_directory, ) from backend.services.mutations import ( rename_file as service_rename_file, ) from backend.services.mutations import ( replace_in_files as service_replace_in_files, ) from backend.services.mutations import ( restore_backup as service_restore_backup, ) from backend.services.recent import humanize_mtime, list_recent from backend.services.sanitizer import sanitize_html from backend.services.search import advanced_search_vaults, list_paths, search_paths, search_vaults from backend.services.search import list_tags as service_list_tags from backend.services.vaults import ( browse_directory, list_accessible_vaults, list_all_files, ) from backend.vault_settings import ( get_vault_setting, update_vault_setting, ) logging.basicConfig( level=logging.INFO, format="%(asctime)s [%(name)s] %(levelname)s: %(message)s", ) logger = logging.getLogger("obsigate") # --------------------------------------------------------------------------- # Pydantic response models # --------------------------------------------------------------------------- class VaultInfo(BaseModel): """Summary information about a configured vault.""" name: str = Field(description="Display name of the vault") file_count: int = Field(description="Number of indexed files") tag_count: int = Field(description="Number of unique tags") type: str = Field(default="VAULT", description="Type of the vault mapping (VAULT or DIR)") class BrowseItem(BaseModel): """A single entry (file or directory) returned by the browse endpoint.""" name: str = Field(description="File or directory name") path: str = Field(description="Relative path within vault") type: str = Field(description="'file' or 'directory'") children_count: int | None = Field(default=None, description="Number of children (directories only)") size: int | None = Field(default=None, description="File size in bytes") extension: str | None = Field(default=None, description="File extension") class BrowseResponse(BaseModel): """Paginated directory listing for a vault.""" vault: str path: str items: list[BrowseItem] class FileContentResponse(BaseModel): """Rendered file content with metadata.""" vault: str = Field(description="Vault name") path: str = Field(description="Relative file path within the vault") title: str = Field(description="File title (from frontmatter or filename)") tags: list[str] = Field(description="Extracted tags from frontmatter and inline #tags") frontmatter: dict[str, Any] = Field(description="YAML frontmatter as key-value dict") html: str = Field(description="Rendered HTML content") raw_length: int = Field(description="Length of raw file content in characters") extension: str = Field(description="File extension (e.g. .md, .txt)") is_markdown: bool = Field(description="Whether the file is markdown") unsupported: bool | None = Field(default=False, description="True for binary/unsupported files") size_bytes: int | None = Field(default=None, description="File size in bytes (for unsupported files)") is_pdf: bool | None = Field(default=None, description="True for PDF files") is_image: bool | None = Field(default=None, description="True for image files") is_csv: bool | None = Field(default=None, description="True for CSV files") is_json: bool | None = Field(default=None, description="True for JSON files") is_excalidraw: bool | None = Field(default=None, description="True for Excalidraw diagram files") excalidraw_data: dict[str, Any] | None = Field(default=None, description="Excalidraw diagram data (elements, appState, files)") excalidraw_data_compressed: str | None = Field(default=None, description="Compressed Excalidraw data for .excalidraw.md files") pdf_metadata: dict[str, Any] | None = Field(default=None, description="PDF metadata") pdf_toc: list[dict[str, Any]] | None = Field(default=None, description="PDF table of contents") image_mime: str | None = Field(default=None, description="MIME type for image files") class FileRawResponse(BaseModel): """Raw text content of a file.""" vault: str = Field(description="Vault name") path: str = Field(description="Relative file path within the vault") raw: str = Field(description="Raw file content as text") class FileSaveResponse(BaseModel): """Confirmation after saving a file.""" status: str = Field(description="Always 'ok'") vault: str = Field(description="Vault name") path: str = Field(description="Relative file path within the vault") size: int = Field(description="Size of saved content in characters") class FileDeleteResponse(BaseModel): """Confirmation after deleting a file.""" status: str = Field(description="Always 'ok'") vault: str = Field(description="Vault name") path: str = Field(description="Relative file path within the vault") class SearchResultItem(BaseModel): """A single search result.""" vault: str = Field(description="Vault name") path: str = Field(description="Relative file path") title: str = Field(description="File title") tags: list[str] = Field(description="File tags") score: int = Field(description="Relevance score") snippet: str = Field(description="Content excerpt with highlights") modified: str = Field(description="ISO 8601 modification timestamp") class SearchResponse(BaseModel): """Full-text search response with optional pagination.""" query: str = Field(description="Original search query") vault_filter: str = Field(description="Vault filter applied ('all' or vault name)") tag_filter: str | None = Field(default=None, description="Tag filter applied") count: int = Field(description="Number of results in this response") total: int = Field(default=0, description="Total results before pagination") offset: int = Field(default=0, description="Current pagination offset") limit: int = Field(default=200, description="Page size") results: list[SearchResultItem] = Field(description="Search result items") class TagsResponse(BaseModel): """Tag aggregation response.""" vault_filter: str | None = Field(default=None, description="Vault filter applied") tags: dict[str, int] = Field(description="Tag name → count mapping") class TreeSearchResult(BaseModel): """A single tree search result item.""" vault: str = Field(description="Vault name") path: str = Field(description="Full relative path") name: str = Field(description="File or directory name") type: str = Field(description="'file' or 'directory'") matched_path: str = Field(description="Path segment that matched the query") class TreeSearchResponse(BaseModel): """Tree search response with matching paths.""" query: str = Field(description="Search query") vault_filter: str = Field(description="Vault filter applied") results: list[TreeSearchResult] = Field(description="Matching files and directories") class VaultPathEntry(BaseModel): """A single indexed path (file or directory) in a vault.""" vault: str = Field(description="Vault name") path: str = Field(description="Full relative path") name: str = Field(description="File or directory name") type: str = Field(description="'file' or 'directory'") class VaultPathsResponse(BaseModel): """Flat list of every indexed path in a vault (capped).""" vault: str = Field(description="Vault name") count: int = Field(description="Number of returned entries") results: list[VaultPathEntry] = Field(description="Indexed files and directories") class AdvancedSearchResultItem(BaseModel): """A single advanced search result with highlighted snippet.""" vault: str = Field(description="Vault name") path: str = Field(description="Relative file path") title: str = Field(description="File title") tags: list[str] = Field(description="File tags") score: float = Field(description="TF-IDF relevance score (or fused RRF score in semantic mode)") semantic_score: float = Field(default=0.0, description="Cosine similarity from the semantic index (0 when unavailable)") snippet: str = Field(description="Content excerpt with highlights") modified: str = Field(description="ISO 8601 modification timestamp") extension: str = Field(default="", description="File extension") class SearchFacets(BaseModel): """Faceted counts for search results.""" tags: dict[str, int] = Field(default_factory=dict) vaults: dict[str, int] = Field(default_factory=dict) class AdvancedSearchResponse(BaseModel): """Advanced search response with TF-IDF scoring, facets, and pagination.""" results: list[AdvancedSearchResultItem] = Field(description="Search results") total: int = Field(description="Total number of matching results") offset: int = Field(description="Current pagination offset") limit: int = Field(description="Page size") facets: SearchFacets = Field(description="Faceted counts by tag and vault") query_time_ms: float = Field(default=0, description="Server-side query time in milliseconds") semantic_available: bool = Field(default=False, description="True when the semantic (embedding) index is ready") class TitleSuggestion(BaseModel): """A file title suggestion for autocomplete.""" vault: str = Field(description="Vault name") path: str = Field(description="Relative file path") title: str = Field(description="File title") class SuggestResponse(BaseModel): """Autocomplete suggestions for file titles.""" query: str = Field(description="Original query string") suggestions: list[TitleSuggestion] = Field(description="Matching file suggestions") class TagSuggestion(BaseModel): """A tag suggestion for autocomplete.""" tag: str = Field(description="Tag name") count: int = Field(description="Number of files with this tag") class TagSuggestResponse(BaseModel): """Autocomplete suggestions for tags.""" query: str = Field(description="Original query string") suggestions: list[TagSuggestion] = Field(description="Matching tag suggestions") class GraphNode(BaseModel): """A single node in the graph view.""" id: str = Field(description="Unique node identifier") name: str = Field(description="Display name") type: str = Field(description="'vault', 'directory', or 'file'") path: str = Field(description="Relative path within vault") size: int = Field(default=0, description="File size in bytes") tags: list[str] = Field(default_factory=list, description="Tags from frontmatter") incoming_count: int = Field(default=0, description="Number of incoming wikilinks") outgoing_count: int = Field(default=0, description="Number of outgoing wikilinks") class GraphEdge(BaseModel): """An edge between two nodes in the graph view.""" source: str = Field(description="Source node ID") target: str = Field(description="Target node ID") relation: str = Field(description="'parent', 'wikilink', or 'backlink'") class GraphResponse(BaseModel): """Graph data for a vault or directory.""" vault: str = Field(description="Vault name") path: str = Field(description="Root path for the graph") scope: str = Field(default="directory", description="'directory' or 'full'") nodes: list[GraphNode] = Field(description="Graph nodes (files and directories)") edges: list[GraphEdge] = Field(description="Graph edges (parent and wikilink relations)") class ReloadResponse(BaseModel): """Index reload confirmation with per-vault stats.""" status: str = Field(description="Reload status ('ok' or 'error')") vaults: dict[str, Any] = Field(description="Per-vault file counts after reload") class HealthResponse(BaseModel): """Application health status.""" status: str = Field(description="Health status ('ok' or 'error')") version: str = Field(description="Application version (x.y.z — latest release tag)") vaults: int = Field(description="Number of configured vaults") total_files: int = Field(description="Total indexed files across all vaults") total_tokens: int = Field(description="Total indexed tokens (approx.) across all vaults", default=0) last_full_index_ts: str = Field(description="ISO timestamp of last full index rebuild", default="") uptime_seconds: int = Field(description="Server uptime in seconds", default=0) git_describe: str = Field(default="", description="Full git describe string (commits beyond tag), empty if no git") git_commit: str = Field(default="", description="Short HEAD commit hash, empty if no git") class DirectoryCreateRequest(BaseModel): """Request to create a new directory.""" path: str = Field(description="Relative path of the new directory") class DirectoryCreateResponse(BaseModel): """Response after creating a directory.""" success: bool = Field(description="Whether creation succeeded") path: str = Field(description="Path of the created directory") class DirectoryRenameRequest(BaseModel): """Request to rename a directory.""" path: str = Field(description="Current path of the directory") new_name: str = Field(description="New name for the directory") class DirectoryRenameResponse(BaseModel): """Response after renaming a directory.""" success: bool = Field(description="Whether rename succeeded") old_path: str = Field(description="Original directory path") new_path: str = Field(description="New directory path") class DirectoryDeleteResponse(BaseModel): """Response after deleting a directory.""" success: bool = Field(description="Whether deletion succeeded") deleted_count: int = Field(description="Number of files recursively deleted") class FileCreateRequest(BaseModel): """Request to create a new file.""" path: str = Field(description="Relative path of the new file") content: str = Field(default="", description="Initial content") class FileCreateResponse(BaseModel): """Response after creating a file.""" success: bool = Field(description="Whether creation succeeded") path: str = Field(description="Path of the created file") class BatchUploadFileItem(BaseModel): """A single file/dir entry in a batch upload request.""" path: str = Field(description="Relative path of the item within the batch") content: str | None = Field(default=None, description="Base64 encoded or text content for files") is_dir: bool = Field(default=False, description="True if entry represents an empty directory") class BatchUploadRequest(BaseModel): """Request payload for batch file/directory upload.""" target_dir: str = Field(default="", description="Base directory in vault to upload into (empty for root)") files: list[BatchUploadFileItem] = Field(description="List of files and directories to upload") overwrite: bool = Field(default=True, description="Whether to overwrite existing files (creates backups)") class BatchUploadResponse(BaseModel): """Response from batch file/directory upload.""" success: bool = Field(description="True if all files uploaded without error") vault: str = Field(description="Vault name") target_dir: str = Field(description="Target directory") uploaded: list[str] = Field(description="List of created/updated file paths") created_dirs: list[str] = Field(description="List of created directory paths") errors: list[dict[str, Any]] = Field(default_factory=list, description="List of items that failed") total_files: int = Field(description="Total uploaded files count") class FileRenameRequest(BaseModel): """Request to rename a file.""" path: str = Field(description="Current path of the file") new_name: str = Field(description="New name for the file") class FileRenameResponse(BaseModel): """Response after renaming a file.""" success: bool = Field(description="Whether rename succeeded") old_path: str new_path: str class FileMoveRequest(BaseModel): """Request to move a file or directory to a different parent directory.""" source_path: str = Field(description="Current relative path of the file/directory") destination_dir: str = Field(description="Target directory relative path (empty string for vault root)") class FileMoveResponse(BaseModel): """Response after moving a file or directory.""" success: bool = Field(description="Whether move succeeded") old_path: str = Field(description="Original path") new_path: str = Field(description="New path after move") item_type: str = Field(description="Type of item moved: 'file' or 'directory'") class BackupEntry(BaseModel): """A single backup version of a file.""" timestamp: int = Field(description="Unix timestamp of when the backup was created") datetime: str = Field(description="ISO 8601 datetime string") size: int = Field(description="File size in bytes") filename: str = Field(description="Backup filename on disk") class BackupListResponse(BaseModel): """Response listing all available backups for a file.""" vault: str = Field(description="Vault name") path: str = Field(description="Relative file path") backups: list[BackupEntry] = Field(description="Available backups, newest first") class DiffRequest(BaseModel): """Request parameters for generating a diff.""" version: int = Field(description="Timestamp of the backup version to compare") compare_with: int | None = Field(default=None, description="Timestamp of another backup version. If omitted, compares with the current file.") class DiffResponse(BaseModel): """Response containing a unified diff between two file versions.""" vault: str = Field(description="Vault name") path: str = Field(description="Relative file path") version: int = Field(description="Backup version timestamp (left/old side)") compare_with: int | None = Field(default=None, description="Other backup version or null for current file (right/new side)") diff: str = Field(description="Unified diff (empty if no changes)") class RestoreRequest(BaseModel): """Request to restore a file from a backup.""" version: int = Field(description="Timestamp of the backup version to restore") class RestoreResponse(BaseModel): """Response after restoring a file from backup.""" success: bool = Field(description="Whether restore succeeded") vault: str = Field(description="Vault name") path: str = Field(description="Relative file path") restored_from: int = Field(description="Timestamp of the backup used") current_backed_up: int | None = Field(default=None, description="Timestamp of the backup created from the current version before restore, if any") # --------------------------------------------------------------------------- # SSE Manager — Server-Sent Events for real-time notifications # --------------------------------------------------------------------------- # --------------------------------------------------------------------------- class SSEManager: """Manages SSE client connections and broadcasts events.""" def __init__(self): self._clients: list[asyncio.Queue] = [] async def connect(self) -> asyncio.Queue: """Register a new SSE client and return its message queue.""" queue: asyncio.Queue = asyncio.Queue() self._clients.append(queue) logger.debug(f"SSE client connected (total: {len(self._clients)})") return queue def disconnect(self, queue: asyncio.Queue): """Remove a disconnected SSE client.""" if queue in self._clients: self._clients.remove(queue) logger.debug(f"SSE client disconnected (total: {len(self._clients)})") async def broadcast(self, event_type: str, data: dict): """Send an event to all connected SSE clients.""" message = _json.dumps(data, ensure_ascii=False) dead: list[asyncio.Queue] = [] for q in self._clients: try: q.put_nowait({"event": event_type, "data": message}) except asyncio.QueueFull: dead.append(q) for q in dead: self.disconnect(q) @property def client_count(self) -> int: return len(self._clients) sse_manager = SSEManager() # --------------------------------------------------------------------------- # Application lifespan (replaces deprecated on_event) # --------------------------------------------------------------------------- from backend.watcher import VaultWatcher # Thread pool for offloading CPU-bound search from the event loop. # Sized to 2 workers so concurrent searches don't starve other requests. _search_executor: ThreadPoolExecutor | None = None _vault_watcher: VaultWatcher | None = None async def _on_vault_change(events: list): """Callback invoked by VaultWatcher when files change in watched vaults. Processes each event (create/modify/delete/move) and updates the index incrementally, then broadcasts SSE notifications. """ updated_vaults = set() changes = [] for event in events: vault_name = event["vault"] event_type = event["type"] src = event["src"] dest = event.get("dest") try: if event_type in ("created", "modified"): result = await update_single_file(vault_name, src) if result: changes.append({"action": "updated", "vault": vault_name, "path": result["path"]}) updated_vaults.add(vault_name) elif event_type == "deleted": result = await remove_single_file(vault_name, src) if result: changes.append({"action": "deleted", "vault": vault_name, "path": result["path"]}) updated_vaults.add(vault_name) elif event_type == "moved": result = await handle_file_move(vault_name, src, dest) if result: changes.append({"action": "moved", "vault": vault_name, "path": result["path"]}) updated_vaults.add(vault_name) except Exception as e: logger.error(f"Error processing {event_type} event for {src}: {e}") if changes: await sse_manager.broadcast("index_updated", { "vaults": list(updated_vaults), "changes": changes, "total_changes": len(changes), }) logger.info(f"Hot-reload: {len(changes)} change(s) in {list(updated_vaults)}") # --------------------------------------------------------------------------- # Authentication bootstrap # --------------------------------------------------------------------------- def bootstrap_admin(): """Create the initial admin account if no users exist. Reads OBSIGATE_ADMIN_USER and OBSIGATE_ADMIN_PASSWORD from environment. If no password is set, generates a random one and logs it ONCE. Only runs when auth is enabled and no users.json exists yet. """ from backend.auth.middleware import is_auth_enabled from backend.auth.user_store import create_user, has_users if not is_auth_enabled(): return if has_users(): return # Users already exist, skip admin_user = os.environ.get("OBSIGATE_ADMIN_USER", "admin") admin_pass = os.environ.get("OBSIGATE_ADMIN_PASSWORD", "") if not admin_pass: # Generate a random password and display it ONCE in logs admin_pass = "".join( secrets.choice(string.ascii_letters + string.digits) for _ in range(16) ) logger.warning("=" * 60) logger.warning("PREMIER DÉMARRAGE — Compte admin créé automatiquement") logger.warning(f" Utilisateur : {admin_user}") logger.warning(f" Mot de passe : {admin_pass}") logger.warning("CHANGEZ CE MOT DE PASSE dès la première connexion !") logger.warning("=" * 60) try: create_user(admin_user, admin_pass, role="admin", vaults=["*"]) logger.info(f"Admin '{admin_user}' créé avec succès") except PermissionError as e: logger.critical("=" * 60) logger.critical("DÉMARRAGE IMPOSSIBLE : Erreur de permission sur le dossier 'data'") logger.critical("L'indexation et l'authentification ne peuvent pas fonctionner.") logger.critical("FIX : Vérifiez les droits du volume /app/data sur l'hôte.") logger.critical("Exemple : sudo chown -R 1000:1000 /DOCKER_CONFIG/ObsiGate/data") logger.critical("=" * 60) raise e # --------------------------------------------------------------------------- # Security headers middleware # --------------------------------------------------------------------------- class SecurityHeadersMiddleware(BaseHTTPMiddleware): """Add security headers to all HTTP responses.""" async def dispatch(self, request, call_next): response = await call_next(request) response.headers["X-Content-Type-Options"] = "nosniff" response.headers["X-Frame-Options"] = "SAMEORIGIN" response.headers["X-XSS-Protection"] = "1; mode=block" response.headers["Referrer-Policy"] = "strict-origin-when-cross-origin" response.headers["Content-Security-Policy"] = ( "default-src 'self'; " "script-src 'self' 'unsafe-inline' blob: https://cdnjs.cloudflare.com https://unpkg.com https://esm.sh https://cdn.jsdelivr.net https://static.cloudflareinsights.com; " "style-src 'self' 'unsafe-inline' https://cdnjs.cloudflare.com https://fonts.googleapis.com https://cdn.jsdelivr.net; " "img-src 'self' data: blob:; " "connect-src 'self' blob: https://esm.sh https://unpkg.com https://cdnjs.cloudflare.com https://fonts.googleapis.com https://fonts.gstatic.com https://cdn.jsdelivr.net; " "font-src 'self' data: https://fonts.gstatic.com https://esm.sh; " "worker-src 'self' blob:; " "frame-src 'self' blob:; " "object-src 'none'; " "base-uri 'self'; " "form-action 'self'; " "frame-ancestors 'self';" ) # Static assets are NOT content-hashed, so they must revalidate: # ``immutable``/long max-age made Cloudflare and mobile browsers serve # a stale build for a year (the service worker cache compounded it). # ``no-cache`` keeps caching but forces revalidation (ETag/Last-Modified). if request.url.path.startswith("/static/"): response.headers["Cache-Control"] = "no-cache" return response def _guard_insecure_auth() -> None: """Warn or refuse to start when authentication is disabled (BUG-037). With ``OBSIGATE_AUTH_ENABLED=false`` every request is served as an anonymous admin. That is convenient for local use but dangerous when the process is reachable from a network. Binding to a non-loopback host without the explicit ``OBSIGATE_ALLOW_INSECURE=true`` opt-in is refused. """ from backend.auth.middleware import ( bind_host_from_argv, is_auth_enabled, is_insecure_mode_allowed, is_loopback_host, ) if is_auth_enabled(): return if is_insecure_mode_allowed(): logger.warning( "Authentication is DISABLED and OBSIGATE_ALLOW_INSECURE=true: every request " "is treated as an anonymous administrator. Do not expose this instance." ) return host = bind_host_from_argv() if not is_loopback_host(host): raise RuntimeError( "Refusing to start: authentication is disabled (OBSIGATE_AUTH_ENABLED=false) " f"while binding to a non-loopback address ('{host}'). This would expose an " "unauthenticated instance with admin access. Enable authentication, or set " "OBSIGATE_ALLOW_INSECURE=true if you really know what you are doing." ) logger.warning( "Authentication is DISABLED (OBSIGATE_AUTH_ENABLED=false): every request is " "treated as an anonymous administrator. This is only safe on a trusted, " "loopback-only deployment." ) @asynccontextmanager async def lifespan(app: FastAPI): """Application lifespan: build index on startup, cleanup on shutdown.""" global _search_executor, _vault_watcher _search_executor = ThreadPoolExecutor(max_workers=2, thread_name_prefix="search") # BUG-037: refuse to expose an unauthenticated instance on a public bind. _guard_insecure_auth() # Bootstrap admin account if needed bootstrap_admin() logger.info("ObsiGate starting — building index in background...") async def _progress_cb(event_type: str, data: dict): await sse_manager.broadcast("index_" + event_type, data) async def _background_startup(): logger.info("Background indexing started") await build_index(_progress_cb) # Build inverted index in a thread pool to avoid blocking the event loop. # The inverted index rebuild is CPU-bound (tokenization, indexing) and # would freeze HTTP responses if run in the async event loop. loop = asyncio.get_running_loop() await loop.run_in_executor(_search_executor, init_inverted_index) # Build the semantic (embedding) index in the same background thread pool. await loop.run_in_executor(_search_executor, init_semantic_index) # BUG-040: extract the PDF text deferred during the scan now that the # index and inverted index are queryable (keeps startup non-blocking). from backend.indexer import enrich_pdf_texts await enrich_pdf_texts() # Scan for plugins in all vaults logger.info("Scanning for plugins...") from backend.indexer import vault_config from backend.plugins import get_plugin_registry registry = get_plugin_registry() for vault_name, cfg in vault_config.items(): vault_path = cfg.get("path") if vault_path: try: plugins = registry.scan_vault(vault_name, vault_path) logger.info(f"Vault '{vault_name}': found {len(plugins)} plugin(s)") from backend.plugins import emit_vault_mounted emit_vault_mounted(vault_name, vault_path) except Exception as e: logger.warning(f"Plugin scan failed for vault '{vault_name}': {e}") # Start file watcher config = _load_config() watcher_enabled = config.get("watcher_enabled", True) if watcher_enabled: use_polling = config.get("watcher_use_polling", False) polling_interval = config.get("watcher_polling_interval", 5.0) debounce = config.get("watcher_debounce", 2.0) global _vault_watcher _vault_watcher = VaultWatcher( on_file_change=_on_vault_change, debounce_seconds=debounce, use_polling=use_polling, polling_interval=polling_interval, ) from backend.indexer import vault_config vaults_to_watch = {name: cfg["path"] for name, cfg in vault_config.items()} await _vault_watcher.start(vaults_to_watch) logger.info("File watcher started in background.") else: logger.info("File watcher disabled by configuration.") logger.info("Background startup complete.") asyncio.create_task(_background_startup()) logger.info("ObsiGate ready (listening for requests while indexing).") yield # Shutdown await collab_manager.stop() if _vault_watcher: await _vault_watcher.stop() _vault_watcher = None _search_executor.shutdown(wait=False) _search_executor = None from backend.version import get_git_commit, get_git_describe, get_version app = FastAPI( title="ObsiGate API", version=get_version(), lifespan=lifespan, description=API_DESCRIPTION.strip(), openapi_tags=TAGS_METADATA, docs_url="/docs", redoc_url="/redoc", openapi_url="/openapi.json", contact={"name": "ObsiGate", "url": "https://git.dracodev.net/Projets/ObsiGate"}, license_info={"name": "MIT"}, ) # Enrich the auto-generated OpenAPI 3.1 schema (#72): tags per category, # examples, security schemes and documented error responses. _original_openapi = app.openapi def _custom_openapi(): if app.openapi_schema: return app.openapi_schema schema = _original_openapi() app.openapi_schema = enrich_openapi_schema(schema) return app.openapi_schema app.openapi = _custom_openapi # type: ignore[method-assign] @app.exception_handler(ServiceError) async def _service_error_handler(request: Request, exc: ServiceError): """Map shared-layer domain errors to HTTP responses (``{"detail": ...}``).""" return JSONResponse(status_code=exc.status, content={"detail": exc.message}) # GZip compression — reduces bandwidth by ~70% for text responses # Custom wrapper: skip compression for SSE streams (/api/events) from fastapi.middleware.gzip import GZipMiddleware from starlette.types import Receive, Scope, Send class SSESafeGZipMiddleware(GZipMiddleware): """GZip middleware that skips SSE (Server-Sent Events) streams. GZip buffering breaks incremental streaming required by SSE. We detect SSE endpoints by path and bypass compression entirely. """ # SSE endpoints that must not be buffered by GZip. _SSE_PATHS = ( "/api/events", "/api/admin/stream", "/api/ai/bookslm/chat", "/api/ai/bookslm/agent", "/mcp", ) async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None: if scope["type"] == "http" and scope.get("path") in self._SSE_PATHS: # Bypass GZip: passthrough directly to the inner app await self.app(scope, receive, send) else: await super().__call__(scope, receive, send) app.add_middleware(SSESafeGZipMiddleware, minimum_size=1000) # Security headers on all responses app.add_middleware(SecurityHeadersMiddleware) # Auth router from backend.audit import log_file_delete, log_file_save from backend.auth.middleware import ( check_vault_access, require_admin, require_auth, ) from backend.auth.router import router as auth_router from backend.secret_redactor import redact_file_content # Lazy import: WeasyPrint PDF export (requires GTK, may not be available everywhere) try: from backend.pdf_export import build_pdf_html, generate_pdf except Exception: # pragma: no cover - WeasyPrint/GTK missing generate_pdf = None # type: ignore[assignment] build_pdf_html = None # type: ignore[assignment] logging.getLogger("obsigate").warning("PDF export unavailable (WeasyPrint/GTK not found)") # Multi-format export (HTML / MD bundle / ePub) — pure Python, no heavy deps. from backend.ai_routes import router as ai_router from backend.bookslm_routes import router as bookslm_router from backend.export import ExportError, export_epub, export_html, export_md_bundle from backend.saved_searches import delete_saved, get_saved, save_search from backend.share import ( create_share, get_share_by_token, list_shares, record_access, revoke_share, update_shares_after_rename, ) from backend.skills_routes import router as skills_router from backend.webhooks import ( create_webhook, delete_webhook, dispatch_webhooks, get_webhooks, update_webhook, ) app.include_router(auth_router) app.include_router(ai_router) app.include_router(bookslm_router) app.include_router(skills_router) # Admin Dashboard endpoints (system stats, audit logs, backups, stream) try: from backend.admin import router as admin_router app.include_router(admin_router) logger.info("Admin dashboard router mounted at /api/admin/*") except ImportError as e: logger.warning(f"Could not load admin dashboard router: {e}") # Push Notifications endpoints (Web Push API + VAPID) try: from backend.push import router as push_router app.include_router(push_router) logger.info("Push notifications router mounted at /api/push/*") except ImportError as e: logger.warning(f"Could not load push notifications router: {e}") # Plugins system endpoints try: from backend.plugins import router as plugins_router app.include_router(plugins_router) logger.info("Plugins router mounted at /api/plugins/*") except ImportError as e: logger.warning(f"Could not load plugins router: {e}") # MCP server (Streamable HTTP) for external clients (#79 phase E) try: from backend.mcp.server import McpMount, mcp_app app.router.routes.append(McpMount(mcp_app)) logger.info("MCP server mounted at /mcp") except Exception as e: # pragma: no cover - optional dependency logger.warning(f"Could not mount MCP server: {e}") # Resolve frontend path relative to this file FRONTEND_DIR = Path(__file__).resolve().parent.parent / "frontend" # --------------------------------------------------------------------------- # API documentation landing page (#72) # --------------------------------------------------------------------------- @app.get("/api", include_in_schema=False, response_class=HTMLResponse) @app.get("/api/", include_in_schema=False, response_class=HTMLResponse) async def api_docs_landing(): """Human-friendly API documentation landing page (links to /docs, /redoc).""" return HTMLResponse(render_api_landing(get_version())) # --------------------------------------------------------------------------- # Path safety helper # --------------------------------------------------------------------------- def _content_disposition(disposition: str, filename: str) -> str: """Build a header-safe Content-Disposition value. HTTP header values must be ASCII. Unicode filenames are sent per RFC 5987 via ``filename*`` (percent-encoded UTF-8) with a pure-ASCII ``filename`` fallback. This avoids a UnicodeDecodeError / HTTP 500 when the filename contains accented characters (e.g. 'Bière blonde…pdf'). """ from urllib.parse import quote ascii_name = "".join(c for c in filename if c.isascii() and (c.isalnum() or c in " _-.")).strip() or "file" ext = Path(filename).suffix if ext and not Path(ascii_name).suffix: ascii_name = ascii_name + ext return f"{disposition}; filename=\"{ascii_name}\"; filename*=UTF-8''{quote(filename)}" def _resolve_safe_path(vault_root: Path, relative_path: str | None) -> Path: """Resolve a relative path safely within the vault root. Thin wrapper around the shared :func:`backend.services.paths.resolve_safe_path` (single implementation used by both routes and tools). The raised :class:`ServiceError` is mapped to an ``HTTPException`` response by the global exception handler in this module. Args: vault_root: The vault's root directory (absolute). relative_path: The user-supplied relative path. Returns: Resolved absolute ``Path``. """ from backend.services.paths import resolve_safe_path as _service_resolve return _service_resolve(vault_root, relative_path) def _backup_file(file_path: Path, vault_name: str, relative_path: str): """Create a timestamped backup of a file before modification. Thin wrapper around :func:`backend.services.backups.create_backup` (single implementation used by both routes and tools). Backups are stored in ``{backup_root}/{vault}/{relative_path}.{timestamp}.bak``; the operation is best-effort and never blocks the caller. """ from backend.services.backups import create_backup create_backup(file_path, vault_name, relative_path) def _check_vault_writable(vault_root: Path) -> bool: """Check if a vault is writable (not mounted read-only). Args: vault_root: The vault's root directory (absolute). Returns: True if the vault is writable, False otherwise. """ return os.access(vault_root, os.W_OK) # --------------------------------------------------------------------------- # Markdown rendering helpers (singleton renderer) # --------------------------------------------------------------------------- import unicodedata def _heading_slugify(text: str) -> str: """Generate a URL-safe slug from heading text. Matches the JavaScript slugify algorithm exactly using Unicode-aware character classification: 1. Strip HTML tags (e.g. wikilink spans rendered inside headings) 2. Decode HTML entities (e.g. ``&`` → ``&``) 3. Lowercase 4. NFD normalize + strip combining marks 5. Keep only Unicode letters, numbers, spaces, hyphens 6. Replace spaces with hyphens, collapse multiple hyphens Args: text: The heading text content (may contain inline HTML). Returns: A URL-safe slug string. """ # Strip any inline HTML so it does not pollute the slug text = re.sub(r"<[^>]+>", "", text) # Decode HTML entities so & becomes & before slugification text = html_mod.unescape(text) text = text.lower() text = unicodedata.normalize("NFD", text) text = "".join(ch for ch in text if not unicodedata.combining(ch)) # Unicode-aware: keep letters (L*), numbers (N*), spaces, and hyphens cleaned = [] for ch in text: cat = unicodedata.category(ch) if cat.startswith('L') or cat.startswith('N') or ch in (' ', '-'): cleaned.append(ch) text = "".join(cleaned) text = re.sub(r"\s+", "-", text) text = re.sub(r"-+", "-", text) result = text.strip("-") return result if result else "heading" def _add_heading_ids(html: str) -> str: """Post-process rendered HTML to add IDs to heading tags. Adds an ``id`` attribute to every ``

`` through ``

`` tag using a slug generated from the heading's text content. Duplicate slugs get a ``-2``, ``-3``, etc. suffix. Args: html: Rendered HTML string. Returns: HTML with heading IDs injected. """ used_ids: dict[str, int] = {} def _replace_heading(match): tag = match.group(1) content = match.group(2) slug = _heading_slugify(content) count = used_ids.get(slug, 0) used_ids[slug] = count + 1 if count > 0: slug = f"{slug}-{count + 1}" return f'<{tag} id="{slug}">{content}' # Match h1-h6 tags with text content (no existing id attribute) return re.sub( r'<(h[1-6])>([^<]*(?:<(?!/?h[1-6])[^<]*)*)', _replace_heading, html, ) # Cached mistune renderer — avoids re-creating on every request _markdown_renderer = mistune.create_markdown( escape=False, plugins=["table", "strikethrough", "footnotes", "task_lists"], ) def _convert_wikilinks(content: str, current_vault: str) -> str: """Convert ``[[wikilinks]]`` and ``[[target|display]]`` to clickable HTML. Supports: - Internal file links: ``[[My Note]]`` / ``[[My Note|display]]`` - Same-document anchors: ``[[#Heading]]`` / ``[[#Heading|display]]`` Resolved file links get a ``data-vault`` / ``data-path`` attribute pair. Anchor links target the slugified heading ID in the current document. Unresolved links are rendered as ````. Args: content: Markdown string potentially containing wikilinks. current_vault: Active vault name for resolution priority. Returns: Markdown string with wikilinks replaced by HTML anchors. """ def _replace(match): target = match.group(1).strip() display = match.group(2).strip() if match.group(2) else target # Same-document anchor link: [[#Heading|display]] if target.startswith("#"): anchor_text = target[1:].strip() anchor_slug = _heading_slugify(anchor_text) link_display = display if display != target else anchor_text return f'{link_display}' found = find_file_in_index(target, current_vault) if found: return ( f'{display}' ) return f'{display}' pattern = r'\[\[([^\]|]+)(?:\|([^\]]+))?\]\]' return re.sub(pattern, _replace, content) def _normalize_line_breaks(text: str) -> str: """Convert single newlines to hard breaks (matching Obsidian default behavior). In standard Markdown, a single ``\\n`` is a "soft break" — it renders as a space, not a visible line break. Obsidian defaults to treating single newlines as hard breaks (equivalent to ``
``). This function pre-processes the Markdown source so that mistune renders standalone lines on separate rows, while still honouring blank lines as paragraph separators. Fenced code blocks (`` ``` ``) are left untouched so their internal newlines are preserved verbatim. """ parts = re.split(r"(```[\s\S]*?```)", text) for i, part in enumerate(parts): if part.startswith("```"): continue # Protect fenced code blocks # Single \n (not preceded or followed by another \n) → two spaces + \n parts[i] = re.sub(r"(? str: """Render a markdown string to HTML with wikilink and image support. Uses the cached singleton mistune renderer for performance. Args: raw_md: Raw markdown text (frontmatter already stripped). vault_name: Current vault for wikilink resolution context. current_file_path: Absolute path to the current markdown file. Returns: HTML string. """ # Get vault data for image resolution vault_data = get_vault_data(vault_name) vault_root = Path(vault_data["path"]) if vault_data else None attachments_path = vault_data.get("config", {}).get("attachmentsPath") if vault_data else None # Redact secrets before rendering (P0 security) raw_md = redact_file_content(raw_md, str(current_file_path) if current_file_path else "") # Preprocess images first if vault_root: raw_md = preprocess_images(raw_md, vault_name, vault_root, current_file_path, attachments_path) # Convert wikilinks converted = _convert_wikilinks(raw_md, vault_name) # Normalize line breaks to match Obsidian behavior (single \n → hard break) converted = _normalize_line_breaks(converted) rendered = _markdown_renderer(converted) # Add heading IDs for TOC navigation rendered = _add_heading_ids(rendered) # Sanitize: raw HTML in vault content must never reach the DOM (BUG-021). rendered = sanitize_html(rendered) return rendered # --------------------------------------------------------------------------- # API Endpoints # --------------------------------------------------------------------------- @app.get("/api/health", response_model=HealthResponse) async def api_health(): """Health check endpoint for Docker and monitoring. Returns: Application status, version, vault count and total file count. """ total_files = sum(len(v["files"]) for v in index.values()) total_tokens = sum(len(v.get("files", [])) * 1000 for v in index.values()) # rough approx import time from backend.indexer import _last_full_index_ts uptime = int(time.time() - _SERVER_START_TIME) if '_SERVER_START_TIME' in globals() else 0 return { "status": "ok", "version": app.version, "vaults": len(index), "total_files": total_files, "total_tokens": total_tokens, "last_full_index_ts": _last_full_index_ts, "uptime_seconds": uptime, "git_describe": get_git_describe(), "git_commit": get_git_commit(), } @app.get("/api/health/detailed", response_model=HealthResponse) async def api_health_detailed(current_user=Depends(require_admin)): """Detailed health check — admin only. Returns enriched metrics including memory, disk, SSE connections, and backup stats. """ import psutil from backend.admin import _count_active_sessions, _get_disk_stats from backend.indexer import _last_full_index_ts, index total_files = sum(len(v["files"]) for v in index.values()) total_tokens = sum(len(v.get("files", [])) * 1000 for v in index.values()) import time uptime = int(time.time() - _SERVER_START_TIME) if '_SERVER_START_TIME' in globals() else 0 # Memory vm = psutil.virtual_memory() mem_used_mb = round(vm.used / (1024 ** 2), 1) mem_total_mb = round(vm.total / (1024 ** 2), 1) mem_pct = round(vm.percent, 1) # CPU cpu_pct = psutil.cpu_percent(interval=None) # Disk disk_used_gb, disk_total_gb = _get_disk_stats() disk_free_gb = round(disk_total_gb - disk_used_gb, 2) disk_pct = round((disk_used_gb / disk_total_gb * 100) if disk_total_gb > 0 else 0, 1) # SSE connections (approximation) active_sessions = _count_active_sessions() # Backups from backend.admin import _scan_backups backup_rows = _scan_backups() total_backups = len(backup_rows) total_backup_size_mb = round(sum(r["size"] for r in backup_rows) / (1024 ** 2), 2) oldest_backup_age_days = 0.0 if backup_rows: now_ts = int(time.time()) oldest_ts = min(r["timestamp"] for r in backup_rows) oldest_backup_age_days = round((now_ts - oldest_ts) / 86400, 2) # Index details index_detail = {} for name, data in index.items(): index_detail[name] = { "file_count": len(data["files"]), "tag_count": len(data["tags"]), "token_count_approx": len(data.get("files", [])) * 1000, } return { "status": "ok", "version": app.version, "vaults": len(index), "total_files": total_files, "total_tokens": total_tokens, "last_full_index_ts": _last_full_index_ts, "uptime_seconds": uptime, "git_describe": get_git_describe(), "git_commit": get_git_commit(), # Enriched fields "memory": { "used_mb": mem_used_mb, "total_mb": mem_total_mb, "percent": mem_pct, }, "cpu": { "percent": cpu_pct, }, "disk": { "used_gb": disk_used_gb, "total_gb": disk_total_gb, "free_gb": disk_free_gb, "percent": disk_pct, }, "connections": { "active_sse": active_sessions, }, "backups": { "total_count": total_backups, "total_size_mb": total_backup_size_mb, "oldest_age_days": oldest_backup_age_days, }, "index": index_detail, } @app.get("/api/vaults", response_model=list[VaultInfo]) async def api_vaults(current_user=Depends(require_auth)): """List configured vaults the user has access to. Returns: List of vault summary objects filtered by user permissions. """ return list_accessible_vaults(current_user) @app.get("/api/recent", response_model=RecentResponse) async def api_recent(limit: int | None = Query(None), vault: str | None = Query(None), mode: str | None = Query("opened"), current_user=Depends(require_auth)): config = _load_config() actual_limit = limit if limit is not None else config.get("recent_files_limit", 20) username = current_user.get("username") user_vaults = current_user.get("_token_vaults") or current_user.get("vaults", []) return list_recent( username, user_vaults, vault=vault, limit=actual_limit, mode=mode or "opened", ) @app.get("/api/bookmarks", response_model=BookmarksResponse) async def api_bookmarks(vault: str | None = Query(None), current_user=Depends(require_auth)): username = current_user.get("username") user_vaults = current_user.get("_token_vaults") or current_user.get("vaults", []) if not username: return {"files": []} history = get_bookmarks(username, vault_filter=vault) files_resp = [] for item in history: v_name = item["vault"] if "*" not in user_vaults and v_name not in user_vaults: continue # Find in index to get metadata f_idx = find_file_in_index(item["path"], v_name) if f_idx: files_resp.append({ "path": f_idx["path"], "title": f_idx.get("title") or item["path"].split("/")[-1], "vault": v_name, "mtime": item["bookmarked_at"], "mtime_human": humanize_mtime(item["bookmarked_at"]), "size_bytes": f_idx.get("size", 0), "tags": [f"#{t}" for t in f_idx.get("tags", [])][:5], "bookmarked": True }) else: files_resp.append({ "path": item["path"], "title": item.get("title") or item["path"].split("/")[-1], "vault": v_name, "mtime": item["bookmarked_at"], "mtime_human": humanize_mtime(item["bookmarked_at"]), "tags": [], "bookmarked": True }) return { "files": files_resp, "total": len(files_resp) } class BookmarkToggleRequest(BaseModel): vault: str path: str title: str | None = None @app.post("/api/bookmarks/toggle", response_model=BookmarkToggleResponse) async def api_toggle_bookmark(req: BookmarkToggleRequest, current_user=Depends(require_auth)): username = current_user.get("username") if not username: raise HTTPException(status_code=401, detail="Not authenticated") # Check vault access if not check_vault_access(req.vault, current_user): raise HTTPException(status_code=403, detail="Access denied to vault") is_now_bookmarked = toggle_bookmark(username, req.vault, req.path, req.title or "") # Update the file's YAML frontmatter: favoris: true/false vault_data = get_vault_data(req.vault) if vault_data: file_path = _resolve_safe_path(Path(vault_data["path"]), req.path) if file_path.exists() and file_path.suffix == ".md": try: raw = file_path.read_text(encoding="utf-8", errors="replace") post = frontmatter.loads(raw) if is_now_bookmarked: post.metadata["favoris"] = True elif "favoris" in post.metadata: del post.metadata["favoris"] new_raw = frontmatter.dumps(post) _backup_file(file_path, req.vault, req.path) file_path.write_text(new_raw, encoding="utf-8") await update_single_file(req.vault, str(file_path)) except Exception as e: logger.warning(f"Failed to update favoris metadata on {req.vault}/{req.path}: {e}") return {"bookmarked": is_now_bookmarked} @app.get("/api/saved-searches", response_model=list[SavedSearch]) async def api_saved_searches(current_user=Depends(require_auth)): username = current_user.get("username") if not username: raise HTTPException(401) return get_saved(username) @app.post("/api/saved-searches", response_model=SavedSearch) async def api_save_search(body: dict = Body(...), current_user=Depends(require_auth)): username = current_user.get("username") if not username: raise HTTPException(401) return save_search(username, body) @app.delete("/api/saved-searches/{search_id}", response_model=StatusResponse) async def api_delete_saved_search(search_id: str, current_user=Depends(require_auth)): username = current_user.get("username") if not username: raise HTTPException(401) if not delete_saved(username, search_id): raise HTTPException(404, "Not found") return {"status": "deleted"} @app.get("/api/browse/{vault_name}", response_model=BrowseResponse) async def api_browse(vault_name: str, path: str = "", current_user=Depends(require_auth)): """Browse directories and files in a vault at a given path level. Returns sorted entries (directories first, then files) with metadata. Hidden files/directories (starting with ``"."`` ) are excluded. Args: vault_name: Name of the vault to browse. path: Relative directory path within the vault (empty = root). Returns: ``BrowseResponse`` with vault name, path, and item list. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") return browse_directory(vault_name, path) # Map file extensions to highlight.js language hints EXT_TO_LANG = { ".py": "python", ".js": "javascript", ".ts": "typescript", ".jsx": "jsx", ".tsx": "tsx", ".sh": "bash", ".bash": "bash", ".zsh": "bash", ".fish": "fish", ".bat": "batch", ".cmd": "batch", ".ps1": "powershell", ".json": "json", ".yaml": "yaml", ".yml": "yaml", ".toml": "toml", ".xml": "xml", ".csv": "plaintext", ".cfg": "ini", ".ini": "ini", ".conf": "ini", ".env": "bash", ".html": "html", ".css": "css", ".scss": "scss", ".less": "less", ".java": "java", ".c": "c", ".cpp": "cpp", ".h": "c", ".hpp": "cpp", ".cs": "csharp", ".go": "go", ".rs": "rust", ".rb": "ruby", ".php": "php", ".sql": "sql", ".r": "r", ".swift": "swift", ".kt": "kotlin", ".txt": "plaintext", ".log": "plaintext", ".lua": "lua", ".pl": "perl", ".pm": "perl", ".ex": "elixir", ".exs": "elixir", ".dart": "dart", ".tf": "haskell", ".gradle": "groovy", ".groovy": "groovy", ".graphql": "graphql", ".gql": "graphql", ".prisma": "sql", ".proto": "c", ".vb": "basic", ".asm": "x86asm", ".s": "armasm", ".vue": "xml", ".svelte": "xml", ".astro": "xml", ".properties": "ini", ".service": "ini", ".hosts": "ini", ".ksh": "bash", ".dockerfile": "dockerfile", ".makefile": "makefile", ".cmake": "cmake", } @app.get("/api/file/{vault_name}/raw", response_model=FileRawResponse) async def api_file_raw(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)): """Return raw file content as plain text. Args: vault_name: Name of the vault. path: Relative file path within the vault. Returns: ``FileRawResponse`` with vault, path, and raw text content. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") return read_raw_file(vault_name, path) @app.get("/api/file/{vault_name}/download", response_class=FileResponse) async def api_file_download(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)): """Download a file as an attachment. Args: vault_name: Name of the vault. path: Relative file path within the vault. Returns: ``FileResponse`` with ``application/octet-stream`` content-type. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, path) if not file_path.exists() or not file_path.is_file(): raise HTTPException(status_code=404, detail=f"File not found: {path}") # Record history record_open(current_user.get("username"), vault_name, path) return FileResponse( path=str(file_path), filename=file_path.name, media_type="application/octet-stream", ) @app.get( "/api/file/{vault_name}/pdf", response_class=Response, responses={200: {"content": {"application/pdf": {}}, "description": "PDF document"}}, ) async def api_file_pdf(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)): """Download a markdown file as PDF.""" if generate_pdf is None: raise HTTPException(501, "PDF export unavailable (WeasyPrint/GTK not available)") if not check_vault_access(vault_name, current_user): raise HTTPException(403, f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(404, f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, path) if not file_path.exists(): raise HTTPException(404, f"File not found: {path}") try: raw = file_path.read_text(encoding="utf-8", errors="replace") except Exception: raise HTTPException(500, "Cannot read file") record_open(current_user.get("username"), vault_name, path) raw = redact_file_content(raw, str(file_path)) post = parse_markdown_file(raw) html = _render_markdown(post.content, vault_name, file_path) title = post.metadata.get("title", file_path.stem) pdf_html = build_pdf_html(html, str(title)) pdf_bytes = generate_pdf(pdf_html, str(title)) safe_name = "".join(c for c in str(title) if c.isascii() and (c.isalnum() or c in " _-.")).strip() or "document" return Response(content=pdf_bytes, media_type="application/pdf", headers={"Content-Disposition": f'attachment; filename="{safe_name}.pdf"'}) # --------------------------------------------------------------------------- # Multi-format export endpoints (HTML / Markdown bundle / ePub) # --------------------------------------------------------------------------- def _resolve_export_target(vault_name: str, path: str, current_user: dict) -> tuple[Path, Path]: """Resolve a vault + relative path into (vault_root, absolute file path). Enforces auth (vault access) and path traversal protection. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) target = _resolve_safe_path(vault_root, path) return vault_root, target @app.get( "/api/export/html", response_class=Response, responses={200: {"content": {"text/html": {}}, "description": "Standalone HTML file"}}, ) async def api_export_html( vault: str = Query(..., description="Vault name"), path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth), ): """Export a markdown note as a standalone HTML file.""" try: vault_root, target = _resolve_export_target(vault, path, current_user) html_bytes = export_html(vault_root, target) except ExportError as e: raise HTTPException(status_code=400, detail=str(e)) record_open(current_user.get("username"), vault, path) safe_name = _safe_export_name(target.stem) return Response( content=html_bytes, media_type="text/html; charset=utf-8", headers={"Content-Disposition": f'attachment; filename="{safe_name}.html"'}, ) @app.get( "/api/export/md-bundle", response_class=Response, responses={200: {"content": {"application/zip": {}}, "description": "Markdown ZIP bundle"}}, ) async def api_export_md_bundle( vault: str = Query(..., description="Vault name"), path: str = Query(..., description="Relative path to directory or file"), current_user=Depends(require_auth), ): """Export a directory (or single file) of markdown as a ZIP bundle.""" try: vault_root, target = _resolve_export_target(vault, path, current_user) zip_bytes = export_md_bundle(vault_root, target) except ExportError as e: raise HTTPException(status_code=400, detail=str(e)) safe_name = _safe_export_name(target.name) return Response( content=zip_bytes, media_type="application/zip", headers={"Content-Disposition": f'attachment; filename="{safe_name}.zip"'}, ) @app.get( "/api/export/epub", response_class=Response, responses={200: {"content": {"application/epub+zip": {}}, "description": "ePub document"}}, ) async def api_export_epub( vault: str = Query(..., description="Vault name"), path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth), ): """Export a markdown note as an ePub document.""" try: vault_root, target = _resolve_export_target(vault, path, current_user) epub_bytes = export_epub(vault_root, target) except ExportError as e: raise HTTPException(status_code=400, detail=str(e)) record_open(current_user.get("username"), vault, path) safe_name = _safe_export_name(target.stem) return Response( content=epub_bytes, media_type="application/epub+zip", headers={"Content-Disposition": f'attachment; filename="{safe_name}.epub"'}, ) def _safe_export_name(name: str) -> str: """ASCII-safe, filename-safe download name (falls back to 'document').""" cleaned = "".join(c for c in name if c.isascii() and (c.isalnum() or c in " _-.")).strip() return cleaned or "document" @app.put("/api/file/{vault_name}/save", response_model=FileSaveResponse) async def api_file_save( vault_name: str, path: str = Query(..., description="Relative path to file"), body: dict = Body(...), backup: bool = Query(True, description="Create a backup before saving (default true, set false for auto-save)"), current_user=Depends(require_auth), ): """Save (overwrite) a file's content. Expects a JSON body with a ``content`` key containing the new text. The path is validated against traversal attacks before writing. Args: vault_name: Name of the vault. path: Relative file path within the vault. body: JSON body with ``content`` string. Returns: ``FileSaveResponse`` confirming the write. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") content = body.get("content", "") result = service_edit_file(vault_name, path, content, backup=backup) # Audit log client_ip = current_user.get("_request_ip", "unknown") log_file_save(current_user["username"], vault_name, path, len(content), client_ip) return {"status": "ok", "vault": result["vault"], "path": result["path"], "size": result["size"]} @app.delete("/api/file/{vault_name}", response_model=FileDeleteResponse) async def api_file_delete(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)): """Delete a file from the vault. The path is validated against traversal attacks before deletion. Args: vault_name: Name of the vault. path: Relative file path within the vault. Returns: ``FileDeleteResponse`` confirming the deletion. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") result = service_delete_file(vault_name, path) # Audit log client_ip = current_user.get("_request_ip", "unknown") log_file_delete(current_user["username"], vault_name, path, client_ip) # Update index await remove_single_file(vault_name, path) # Broadcast SSE event await sse_manager.broadcast("file_deleted", { "vault": vault_name, "path": path, }) from backend.plugins import emit_file_deleted emit_file_deleted(vault_name, path) # Remove from recent files remove_recent(current_user["username"], vault_name, path) # Dispatch webhooks await dispatch_webhooks("file_deleted", {"vault": vault_name, "path": path}) return {"status": "ok", "vault": result["vault"], "path": result["path"]} # --------------------------------------------------------------------------- # Directory management endpoints # --------------------------------------------------------------------------- @app.post("/api/directory/{vault_name}", response_model=DirectoryCreateResponse) async def api_directory_create( vault_name: str, body: DirectoryCreateRequest, current_user=Depends(require_auth), ): """Create a new directory in a vault. Args: vault_name: Name of the vault. body: Request body with directory path. Returns: DirectoryCreateResponse confirming creation. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") result = service_create_directory(vault_name, body.path) # Update path_index with the new directory from backend.indexer import _index_lock from backend.indexer import path_index as _path_idx with _index_lock: if vault_name not in _path_idx: _path_idx[vault_name] = [] existing = {p["path"] for p in _path_idx[vault_name]} # Build all parent segments parts = body.path.split("/") for i in range(1, len(parts) + 1): seg_path = "/".join(parts[:i]) if seg_path and seg_path not in existing: existing.add(seg_path) _path_idx[vault_name].append({ "path": seg_path, "name": parts[i - 1], "type": "directory", }) # Broadcast SSE event await sse_manager.broadcast("directory_created", { "vault": vault_name, "path": result["path"], }) await dispatch_webhooks("directory_created", {"vault": vault_name, "path": result["path"]}) return {"success": True, "path": result["path"]} @app.patch("/api/directory/{vault_name}", response_model=DirectoryRenameResponse) async def api_directory_rename( vault_name: str, body: DirectoryRenameRequest, current_user=Depends(require_auth), ): """Rename a directory in a vault. Args: vault_name: Name of the vault. body: Request body with current path and new name. Returns: DirectoryRenameResponse with old and new paths. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") result = service_rename_directory(vault_name, body.path, body.new_name) old_path_str = result["old_path"] new_path_str = result["new_path"] # Update index for all files in the directory from backend.indexer import reload_single_vault await reload_single_vault(vault_name) # Broadcast SSE event await sse_manager.broadcast("directory_renamed", { "vault": vault_name, "old_path": old_path_str, "new_path": new_path_str, }) await dispatch_webhooks("directory_renamed", {"vault": vault_name, "old_path": old_path_str, "new_path": new_path_str}) return {"success": True, "old_path": old_path_str, "new_path": new_path_str} @app.delete("/api/directory/{vault_name}", response_model=DirectoryDeleteResponse) async def api_directory_delete( vault_name: str, path: str = Query(..., description="Relative path to directory"), current_user=Depends(require_auth), ): """Delete a directory and all its contents from a vault. Args: vault_name: Name of the vault. path: Relative directory path within the vault. Returns: DirectoryDeleteResponse with count of deleted files. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") result = service_delete_directory(vault_name, path, recursive=True) file_count = result["deleted_count"] # Update index from backend.indexer import reload_single_vault await reload_single_vault(vault_name) # Broadcast SSE event await sse_manager.broadcast("directory_deleted", { "vault": vault_name, "path": result["path"], "deleted_count": file_count, }) await dispatch_webhooks("directory_deleted", {"vault": vault_name, "path": result["path"]}) return {"success": True, "deleted_count": file_count} # --------------------------------------------------------------------------- # File creation and rename endpoints # --------------------------------------------------------------------------- @app.post("/api/file/{vault_name}", response_model=FileCreateResponse) async def api_file_create( vault_name: str, body: FileCreateRequest, current_user=Depends(require_auth), ): """Create a new file in a vault. Args: vault_name: Name of the vault. body: Request body with file path and initial content. Returns: FileCreateResponse confirming creation. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") result = service_create_file(vault_name, body.path, body.content) # Update index await update_single_file(vault_name, result["path"]) # Broadcast SSE event await sse_manager.broadcast("file_created", { "vault": vault_name, "path": result["path"], }) await dispatch_webhooks("file_created", {"vault": vault_name, "path": result["path"]}) from backend.plugins import emit_file_created emit_file_created(vault_name, result["path"]) return {"success": True, "path": result["path"]} @app.post("/api/vault/{vault_name}/batch-upload", response_model=BatchUploadResponse) async def api_batch_upload( vault_name: str, body: BatchUploadRequest, current_user=Depends(require_auth), ): """Upload multiple files and directories (recursively) into a vault. Accepts base64 encoded or plain text files with relative directory paths. Creates missing parent folders safely. Args: vault_name: Target vault name. body: BatchUploadRequest with target_dir and files list. Returns: BatchUploadResponse with summary of uploaded files and errors. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") import base64 items: list[dict[str, Any]] = [] for f in body.files: if f.is_dir: items.append({"path": f.path, "is_dir": True}) continue raw_bytes = b"" if f.content is not None: # Check if content is base64 encoded data URI or raw base64 content_str = f.content if content_str.startswith("data:") and ";base64," in content_str: content_str = content_str.split(";base64,", 1)[1] try: raw_bytes = base64.b64decode(content_str) except Exception: # Fallback to utf-8 text encoding raw_bytes = f.content.encode("utf-8") items.append({"path": f.path, "content": raw_bytes, "is_dir": False}) result = service_batch_upload_files( vault_name, body.target_dir, items, overwrite=body.overwrite, ) # Update index and SSE notifications for uploaded files for path in result["uploaded"]: try: await update_single_file(vault_name, path) await sse_manager.broadcast("file_created", { "vault": vault_name, "path": path, }) await dispatch_webhooks("file_created", {"vault": vault_name, "path": path}) except Exception as e: logger.warning(f"Failed to post-process upload of {path}: {e}") # SSE notification for tree refresh if result["uploaded"] or result["created_dirs"]: await sse_manager.broadcast("tree_updated", { "vault": vault_name, "target_dir": result["target_dir"], }) return result @app.patch("/api/file/{vault_name}", response_model=FileRenameResponse) async def api_file_rename( vault_name: str, body: FileRenameRequest, current_user=Depends(require_auth), ): """Rename a file in a vault. Args: vault_name: Name of the vault. body: Request body with current path and new name. Returns: FileRenameResponse with old and new paths. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") result = service_rename_file(vault_name, body.path, body.new_name) old_path_str = result["old_path"] new_path_str = result["new_path"] # Update index await handle_file_move(vault_name, old_path_str, new_path_str) # Update bookmarks, history, and shares update_bookmarks_after_rename(vault_name, old_path_str, new_path_str) update_history_after_rename(vault_name, old_path_str, new_path_str) update_shares_after_rename(vault_name, old_path_str, new_path_str) # Broadcast SSE event await sse_manager.broadcast("file_renamed", { "vault": vault_name, "old_path": old_path_str, "new_path": new_path_str, }) await dispatch_webhooks("file_renamed", {"vault": vault_name, "old_path": old_path_str, "new_path": new_path_str}) return {"success": True, "old_path": old_path_str, "new_path": new_path_str} @app.post("/api/move/{vault_name}", response_model=FileMoveResponse) async def api_file_move( vault_name: str, body: FileMoveRequest, current_user=Depends(require_auth), ): """Move a file or directory to a different parent directory within the same vault. Supports both files and directories. The item keeps its original name; only the parent directory changes. Args: vault_name: Name of the vault. body: Request body with source_path and destination_dir. Returns: FileMoveResponse with old and new paths. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") result = service_move_path(vault_name, body.source_path, body.destination_dir) old_path_str = result["old_path"] new_path_str = result["new_path"] item_type = result["item_type"] # Update index if item_type == "directory": from backend.indexer import reload_single_vault await reload_single_vault(vault_name) else: await handle_file_move(vault_name, old_path_str, new_path_str) # Broadcast SSE event await sse_manager.broadcast("item_moved", { "vault": vault_name, "old_path": old_path_str, "new_path": new_path_str, "item_type": item_type, }) await dispatch_webhooks("item_moved", {"vault": vault_name, "old_path": old_path_str, "new_path": new_path_str, "item_type": item_type}) return {"success": True, "old_path": old_path_str, "new_path": new_path_str, "item_type": item_type} # --------------------------------------------------------------------------- # Backup & Diff endpoints # --------------------------------------------------------------------------- def _get_backup_dir(vault_name: str, relative_path: str) -> Path: """Return the directory where backups for a specific file are stored. Thin wrapper around :func:`backend.services.backups.get_backup_dir` (single implementation used by both routes and tools). """ return service_get_backup_dir(vault_name, relative_path) def _list_backup_files(vault_name: str, relative_path: str) -> list[dict]: """List all backup files for a given vault file, sorted newest first. Thin wrapper around :func:`backend.services.backups.list_backup_files` (single implementation used by both routes and tools). """ return service_list_backup_files(vault_name, relative_path) @app.get("/api/file/{vault_name}/backups", response_model=BackupsResponse) async def api_file_backups( vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth), ): """List all available backups for a file. Args: vault_name: Name of the vault. path: Relative path of the file within the vault. Returns: BackupListResponse with backups sorted newest first. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, path) if not file_path.exists() or not file_path.is_file(): raise HTTPException(status_code=404, detail=f"File not found: {path}") try: backups = _list_backup_files(vault_name, path) except Exception as e: logger.error(f"Error listing backups for {vault_name}/{path}: {type(e).__name__}: {e}", exc_info=True) raise HTTPException(status_code=500, detail=f"Erreur lors de la lecture des backups: {e!s}") return {"vault": vault_name, "path": path, "backups": backups} @app.get("/api/file/{vault_name}/diff", response_model=DiffResponse) async def api_file_diff( vault_name: str, path: str = Query(..., description="Relative path to file"), version: int = Query(..., description="Timestamp of the backup version (left/old side)"), compare_with: int | None = Query(default=None, description="Timestamp of another backup (right/new side). If omitted, compares with the current file."), current_user=Depends(require_auth), ): """Generate a unified diff between a backup version and another version or the current file. Args: vault_name: Name of the vault. path: Relative path of the file within the vault. version: Timestamp of the backup to use as the old/left side. compare_with: Optional timestamp of another backup as the new/right side. If omitted, the current file on disk is used. Returns: DiffResponse containing the unified diff string. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") return service_diff_backup(vault_name, path, version, compare_with) @app.post("/api/file/{vault_name}/restore", response_model=RestoreResponse) async def api_file_restore( vault_name: str, path: str = Query(..., description="Relative path to file"), body: RestoreRequest = ..., # type: ignore current_user=Depends(require_auth), ): """Restore a file from a backup version. The current file is backed up before being overwritten (so the operation is reversible). Args: vault_name: Name of the vault. path: Relative path of the file within the vault. body: RestoreRequest with the backup version timestamp. Returns: RestoreResponse confirming the restore. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") result = service_restore_backup(vault_name, path, body.version) current_backed_up = result["current_backed_up"] # Update index await update_single_file(vault_name, path) # Broadcast SSE event await sse_manager.broadcast("file_restored", { "vault": vault_name, "path": path, "restored_from": body.version, "current_backed_up": current_backed_up, }) await dispatch_webhooks("file_restored", {"vault": vault_name, "path": path, "restored_from": body.version}) return { "success": True, "vault": vault_name, "path": path, "restored_from": body.version, "current_backed_up": current_backed_up, } @app.get("/api/file/{vault_name}/backlinks", response_model=BacklinksResponse) async def api_file_backlinks( vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth), ): """Get backlinks (files linking to this file via wikilinks). Returns a list of files that contain `[[wikilinks]]` pointing to the requested file, across all accessible vaults. Args: vault_name: Name of the vault containing the target file. path: Relative path of the target file within the vault. Returns: ``{"vault": str, "path": str, "backlinks": [...]}`` """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") user_vaults = current_user.get("_token_vaults") or current_user.get("vaults", []) backlinks = get_backlinks(vault_name, path) # Filter by user-accessible vaults if "*" not in user_vaults: backlinks = [b for b in backlinks if b["vault"] in user_vaults] return { "vault": vault_name, "path": path, "backlinks": backlinks, "total": len(backlinks), } @app.get("/api/file/{vault_name}", response_model=FileContentResponse) async def api_file(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)): """Return rendered HTML and metadata for a file. Markdown files are parsed for frontmatter, rendered with wikilink support, and returned with extracted tags. Other supported file types are syntax-highlighted as code blocks. Args: vault_name: Name of the vault. path: Relative file path within the vault. Returns: ``FileContentResponse`` with HTML, metadata, and tags. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, path) if not file_path.exists() or not file_path.is_file(): raise HTTPException(status_code=404, detail=f"File not found: {path}") # Record history record_open(current_user.get("username"), vault_name, path, title=file_path.name) ext = file_path.suffix.lower() # === PDF: special handling before read_text (binary file) === if ext == ".pdf": try: from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text, extract_pdf_toc pdf_text = extract_pdf_text(file_path, max_chars=100000) pdf_meta = extract_pdf_metadata(file_path) pdf_toc = extract_pdf_toc(file_path) size = file_path.stat().st_size return { "vault": vault_name, "path": path, "title": pdf_meta.get("title") or file_path.name, "tags": [], "frontmatter": {}, "html": f"

PDF — {pdf_meta.get('pages', '?')} pages

{pdf_text[:5000]}
", "raw_length": size, "extension": ext, "is_markdown": False, "is_pdf": True, "unsupported": False, "pdf_metadata": pdf_meta, "pdf_toc": pdf_toc, "size_bytes": size, } except Exception as e: logger.error(f"PDF read error for {path}: {e}") raise HTTPException(status_code=500, detail=f"Error reading PDF: {e!s}") # === Images: return as viewable image === IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".gif", ".svg", ".webp", ".bmp", ".ico"} if ext in IMAGE_EXTENSIONS: size = file_path.stat().st_size mime_map = { ".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg", ".gif": "image/gif", ".svg": "image/svg+xml", ".webp": "image/webp", ".bmp": "image/bmp", ".ico": "image/x-icon", } mime = mime_map.get(ext, "application/octet-stream") html = ( f'
' f'' f'
' ) return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": html, "raw_length": size, "extension": ext, "is_markdown": False, "is_image": True, "image_mime": mime, "size_bytes": size, } try: raw = file_path.read_text(encoding="utf-8", errors="replace") except PermissionError as e: logger.error(f"Permission denied reading file {path}: {e}") raise HTTPException(status_code=403, detail=f"Permission denied: cannot read file {path}") except UnicodeDecodeError: # Binary / unsupported file — return structured info with download option size = file_path.stat().st_size return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": "", "raw_length": size, "extension": ext, "is_markdown": False, "unsupported": True, "size_bytes": size, } except Exception as e: logger.error(f"Unexpected error reading file {path}: {e}") raise HTTPException(status_code=500, detail=f"Error reading file: {e!s}") # === CSV: render as HTML table === if ext == ".csv": import csv import io as csv_io reader = csv.reader(csv_io.StringIO(raw)) rows = list(reader) if not rows: html = "

Fichier CSV vide

" else: headers = rows[0] data_rows = rows[1:] html = '
' for h in headers: html += f"" html += "" for row in data_rows: html += "" for cell in row: html += f"" html += "" html += "
{h}
{cell}
" return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": html, "raw_length": len(raw), "extension": ext, "is_markdown": False, "is_csv": True, } # === JSON: syntax-highlighted display === if ext == ".json": import json as json_mod try: parsed = json_mod.loads(raw) formatted = json_mod.dumps(parsed, indent=2, ensure_ascii=False) except json_mod.JSONDecodeError: formatted = raw html = f"
{html_mod.escape(formatted)}
" return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": html, "raw_length": len(raw), "extension": ext, "is_markdown": False, "is_json": True, } # === Excalidraw .excalidraw.md (Obsidian plugin format) === if path.lower().endswith(".excalidraw.md"): import re as re_mod raw_lower = file_path.read_text(encoding="utf-8", errors="replace") # Check for excalidraw-plugin in frontmatter or body if "excalidraw-plugin:" in raw_lower: # Extract compressed JSON block match = re_mod.search(r'```compressed-json\n(.*?)\n```', raw_lower, re_mod.DOTALL) if match: compressed = match.group(1).strip() return { "vault": vault_name, "path": path, "title": file_path.name.replace(".excalidraw.md", ""), "tags": [], "frontmatter": {}, "html": "", "raw_length": len(raw_lower), "extension": ".excalidraw.md", "is_markdown": False, "is_excalidraw": True, "excalidraw_data_compressed": compressed, } # Fallback: treat as regular markdown raw = raw_lower if ext == ".excalidraw": import json as json_mod try: parsed = json_mod.loads(raw) except json_mod.JSONDecodeError: parsed = None if parsed and parsed.get("type") == "excalidraw": return { "vault": vault_name, "path": path, "title": parsed.get("appState", {}).get("name") or file_path.name, "tags": [], "frontmatter": {}, "html": "", "raw_length": len(raw), "extension": ext, "is_markdown": False, "is_excalidraw": True, "excalidraw_data": { "elements": parsed.get("elements", []), "appState": parsed.get("appState", {}), "files": parsed.get("files", {}), }, } else: # Not a valid Excalidraw file — fall through to text viewer pass # === Plain text / other readable files === TEXT_EXTENSIONS = {".txt", ".log", ".yml", ".yaml", ".toml", ".ini", ".cfg", ".sh", ".bash", ".py", ".js", ".ts", ".html", ".css", ".xml", ".rst", ".tex", ".sql", ".conf", ".env"} if ext in TEXT_EXTENSIONS or ext == ".md": pass # handled below or by markdown section if ext == ".md": post = parse_markdown_file(raw) # Extract metadata using shared indexer logic tags = _extract_tags(post) title = post.metadata.get("title", file_path.stem.replace("-", " ").replace("_", " ")) html_content = _render_markdown(post.content, vault_name, file_path) return { "vault": vault_name, "path": path, "title": str(title), "tags": tags, "frontmatter": dict(post.metadata) if post.metadata else {}, "html": html_content, "raw_length": len(raw), "extension": ext, "is_markdown": True, } else: # Non-markdown: wrap in syntax-highlighted code block lang = EXT_TO_LANG.get(ext, "") if not lang: # Fichiers sans extension usuels (Dockerfile, Makefile, etc.) NAME_TO_LANG = { "dockerfile": "dockerfile", "makefile": "makefile", "cmakelists.txt": "cmake", "jenkinsfile": "groovy", "vagrantfile": "ruby", "rakefile": "ruby", "gemfile": "ruby", "procfile": "plaintext", "bashrc": "bash", "bash_profile": "bash", "zshrc": "bash", "profile": "bash", "gitignore": "plaintext", } lang = NAME_TO_LANG.get(file_path.name.lower(), "plaintext") escaped = html_mod.escape(raw) html_content = f'
{escaped}
' return { "vault": vault_name, "path": path, "title": file_path.name, "tags": [], "frontmatter": {}, "html": html_content, "raw_length": len(raw), "extension": ext, "is_markdown": False, } @app.get("/api/file/{vault_name}/pdf/stream", response_class=FileResponse) async def api_pdf_stream( request: Request, vault_name: str, path: str = Query(...), current_user=Depends(require_auth), ): """Stream a PDF file with Content-Type: application/pdf for inline browser viewing. Supports HTTP Range requests (206 Partial Content) so browsers can progressively render large PDFs in the native viewer. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, path) if not file_path.exists() or not file_path.is_file(): raise HTTPException(status_code=404, detail=f"File not found: {path}") if file_path.suffix.lower() != ".pdf": raise HTTPException(status_code=400, detail="Not a PDF file") file_size = file_path.stat().st_size range_header = request.headers.get("range") if range_header: # Parse "bytes=start-end" (single range only; multi-range is not used by viewers) m = re.match(r"bytes=(\d*)-(\d*)", range_header) if not m: raise HTTPException(status_code=416, headers={"Content-Range": f"bytes */{file_size}"}) start_s, end_s = m.group(1), m.group(2) if start_s == "" and end_s == "": raise HTTPException(status_code=416, headers={"Content-Range": f"bytes */{file_size}"}) if start_s == "": # suffix range: last N bytes length = min(int(end_s), file_size) start = file_size - length end = file_size - 1 else: start = int(start_s) end = int(end_s) if end_s else file_size - 1 end = min(end, file_size - 1) if start > end or start >= file_size: raise HTTPException(status_code=416, headers={"Content-Range": f"bytes */{file_size}"}) chunk_size = end - start + 1 async def _partial(): # Open + reads offloaded to threads (avoid blocking the event loop — ASYNC230) f = await asyncio.to_thread(open, str(file_path), "rb") try: await asyncio.to_thread(f.seek, start) remaining = chunk_size while remaining > 0: data = await asyncio.to_thread(f.read, min(64 * 1024, remaining)) if not data: break remaining -= len(data) yield data finally: await asyncio.to_thread(f.close) return StreamingResponse( _partial(), status_code=206, media_type="application/pdf", headers={ "Content-Range": f"bytes {start}-{end}/{file_size}", "Accept-Ranges": "bytes", "Content-Length": str(chunk_size), "Content-Disposition": _content_disposition("inline", file_path.name), }, ) return FileResponse(str(file_path), media_type="application/pdf", headers={ "Accept-Ranges": "bytes", "Content-Disposition": _content_disposition("inline", file_path.name)}) @app.get("/api/file/{vault_name}/pdf/info", response_model=PdfInfoResponse) async def api_pdf_info( vault_name: str, path: str = Query(..., description="Relative path to PDF file"), current_user=Depends(require_auth), ): """Return PDF metadata (pages, title, author, size) without the document content. Lets the UI display file info before loading a heavy PDF into the viewer. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, path) if not file_path.exists() or not file_path.is_file(): raise HTTPException(status_code=404, detail=f"File not found: {path}") if file_path.suffix.lower() != ".pdf": raise HTTPException(status_code=400, detail="Not a PDF file") from backend.pdf_reader import extract_pdf_metadata meta = extract_pdf_metadata(file_path) stat = file_path.stat() return { "vault": vault_name, "path": path, "pages": meta.get("pages", 0), "title": meta.get("title") or file_path.name, "author": meta.get("author", ""), "size_bytes": stat.st_size, } @app.get("/api/search", response_model=SearchResponse) async def api_search( q: str = Query("", description="Search query"), vault: str = Query("all", description="Vault filter"), tag: str | None = Query(None, description="Tag filter"), limit: int = Query(50, ge=1, le=200, description="Results per page"), offset: int = Query(0, ge=0, description="Pagination offset"), current_user=Depends(require_auth), ): """Full-text search across vaults with relevance scoring. Supports combining free-text queries with tag filters. Results are ranked by a multi-factor scoring algorithm. Pagination via ``limit`` and ``offset`` (defaults preserve backward compat). Args: q: Free-text search string. vault: Vault name or ``"all"`` to search everywhere. tag: Comma-separated tag names to require. limit: Max results per page (1–200). offset: Pagination offset. Returns: ``SearchResponse`` with ranked results and snippets. """ loop = asyncio.get_event_loop() # Fetch the full result set (capped at DEFAULT_SEARCH_LIMIT internally) and # paginate in the shared service so routes and tools share the same logic. return await loop.run_in_executor( _search_executor, partial(search_vaults, q, vault, tag, limit, offset), ) @app.get("/api/tags", response_model=TagsResponse) async def api_tags(vault: str | None = Query(None, description="Vault filter"), current_user=Depends(require_auth)): """Return all unique tags with occurrence counts. Args: vault: Optional vault name to restrict tag aggregation. Returns: ``TagsResponse`` with tags sorted by descending count. """ return {"vault_filter": vault, "tags": service_list_tags(vault)} @app.get("/api/tree-search", response_model=TreeSearchResponse) async def api_tree_search( q: str = Query("", description="Search query"), vault: str = Query("all", description="Vault filter"), current_user=Depends(require_auth), ): """Search for files and directories in the tree structure using pre-built index. Uses the in-memory path index for instant filtering without filesystem access. Args: q: Search string to match against file/directory paths. vault: Vault name or "all" to search everywhere. Returns: ``TreeSearchResponse`` with matching paths. """ return search_paths(q, vault) @app.get("/api/vault/{vault_name}/paths", response_model=VaultPathsResponse) async def api_vault_paths( vault_name: str, limit: int = Query(5000, ge=1, le=20000, description="Maximum number of indexed paths to return"), current_user=Depends(require_auth), ): """Return a flat list of every indexed file and directory in a vault. Used by the AI assistant ``@`` mention menu to filter paths instantly on the client (one request instead of one per keystroke). Args: vault_name: Name of the vault. limit: Maximum number of entries returned. Returns: ``VaultPathsResponse`` with the vault's indexed paths. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") return list_paths(vault_name, limit=limit) @app.get("/api/search/advanced", response_model=AdvancedSearchResponse) async def api_advanced_search( q: str = Query("", description="Advanced search query (supports tag:, vault:, title:, path:, ext: operators)"), vault: str = Query("all", description="Vault filter"), tag: str | None = Query(None, description="Comma-separated tag filter"), limit: int = Query(50, ge=1, le=200, description="Results per page"), offset: int = Query(0, ge=0, description="Pagination offset"), sort: str = Query("relevance", description="Sort by 'relevance' or 'modified'"), case_sensitive: bool = Query(False, description="Match case"), whole_word: bool = Query(False, description="Match whole words only"), regex: bool = Query(False, description="Treat query as regex"), include_paths: str | None = Query(None, description="Comma-separated glob patterns to include"), exclude_paths: str | None = Query(None, description="Comma-separated glob patterns to exclude"), created: str | None = Query(None, description="Created date filter (>date, `` or ``#`` — filter by tag - ``vault:`` — filter by vault - ``title:`` — filter by title substring - ``path:`` — filter by path substring - ``ext:`` — filter by file extension - ``created:>2024-01-01`` — filter by creation date - ``modified:<7d`` or ``modified:2024-01-01..2024-06-01`` — filter by modification date - ``size:>1MB`` or ``size:100KB..1MB`` — filter by file size - Remaining text is scored using TF-IDF with accent normalization. - Toggles: case_sensitive, whole_word, regex - Path filters: include_paths, exclude_paths (glob patterns) - ``semantic=true`` — fuse the TF-IDF ranking with the semantic (embedding) ranking via Reciprocal Rank Fusion and expose ``semantic_score`` per result. Results include ````-highlighted snippets and faceted tag/vault counts. """ loop = asyncio.get_event_loop() search_fn = partial(advanced_search_vaults, q, vault=vault, tag=tag, limit=limit, offset=offset, sort=sort, case_sensitive=case_sensitive, whole_word=whole_word, regex=regex, include_paths=include_paths, exclude_paths=exclude_paths, created=created, modified=modified, size=size, semantic=semantic) try: return await loop.run_in_executor(_search_executor, search_fn) except ValueError as e: raise HTTPException(400, str(e)) from e @app.post("/api/search/replace", response_model=ReplaceResponse) async def api_search_replace( body: dict = Body(...), current_user=Depends(require_auth), ): """Find and replace across vault files.""" query = body.get("query", "") replacement = body.get("replacement", "") vault_filter = body.get("vault", "all") case_sensitive = body.get("case_sensitive", False) whole_word = body.get("whole_word", False) regex_mode = body.get("regex", False) include_paths = body.get("include_paths") exclude_paths = body.get("exclude_paths") replace_all = body.get("replace_all", False) dry_run = body.get("dry_run", not replace_all) if not query: raise HTTPException(400, "Query is required") result = service_replace_in_files( query, replacement, vault=vault_filter, case_sensitive=case_sensitive, whole_word=whole_word, regex=regex_mode, include_paths=include_paths, exclude_paths=exclude_paths, replace_all=replace_all, dry_run=dry_run, is_vault_allowed=lambda v: check_vault_access(v, current_user), ) if dry_run: return result # Side effects for applied replacements (audit + incremental index). for match in result.get("replaced", []): log_file_save(current_user["username"], match["vault"], match["path"], match.get("size", 0)) vault_data = get_vault_data(match["vault"]) if vault_data: abs_path = str(Path(vault_data["path"]) / match["path"]) await update_single_file(match["vault"], abs_path) return result @app.get("/api/suggest", response_model=SuggestResponse) async def api_suggest( q: str = Query("", description="Prefix to search for in file titles"), vault: str = Query("all", description="Vault filter"), limit: int = Query(10, ge=1, le=50, description="Max suggestions"), current_user=Depends(require_auth), ): """Suggest file titles matching a prefix (accent-insensitive). Used for autocomplete in the search input. Args: q: User-typed prefix (minimum 2 characters). vault: Vault name or ``"all"``. limit: Max number of suggestions. Returns: ``SuggestResponse`` with matching file title suggestions. """ suggestions = suggest_titles(q, vault_filter=vault, limit=limit) return {"query": q, "suggestions": suggestions} @app.get("/api/tags/suggest", response_model=TagSuggestResponse) async def api_tags_suggest( q: str = Query("", description="Prefix to search for in tags"), vault: str = Query("all", description="Vault filter"), limit: int = Query(10, ge=1, le=50, description="Max suggestions"), current_user=Depends(require_auth), ): """Suggest tags matching a prefix (accent-insensitive). Used for autocomplete when typing ``tag:`` or ``#`` in the search input. Args: q: User-typed prefix (with or without ``#``, minimum 2 characters). vault: Vault name or ``"all"``. limit: Max number of suggestions. Returns: ``TagSuggestResponse`` with matching tag suggestions and counts. """ suggestions = suggest_tags(q, vault_filter=vault, limit=limit) return {"query": q, "suggestions": suggestions} @app.get("/api/index/reload", response_model=ReloadResponse) async def api_reload(current_user=Depends(require_admin)): """Force a full re-index of all configured vaults. Returns: ``ReloadResponse`` with per-vault file and tag counts. """ stats = await reload_index() await sse_manager.broadcast("index_reloaded", { "vaults": list(stats.keys()), "stats": stats, }) return {"status": "ok", "vaults": stats} @app.get("/api/graph/{vault_name}", response_model=GraphResponse) async def api_graph( vault_name: str, path: str = Query("", description="Relative path to focus on"), depth: int = Query(1, ge=0, le=3, description="How many levels deep to expand"), scope: str = Query("directory", description="'directory' (default) or 'full' for entire vault"), tag: str = Query("", description="Filter: only show files with this tag"), current_user=Depends(require_auth), ): """Return graph data (nodes and edges) for a vault or directory. Nodes represent files and directories. Edges represent parent-child relationships and wikilinks between markdown files. Args: vault_name: Name of the vault. path: Relative directory path to focus on (empty = root). depth: Expansion depth (0 = only direct children, 1-3 = deeper). scope: 'directory' for subtree, 'full' for entire vault. tag: Optional tag filter (only files with this tag appear). Returns: ``GraphResponse`` with nodes and edges. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") return service_get_graph(vault_name, path=path, depth=depth, scope=scope, tag=tag) @app.get("/api/index/reload/{vault_name}", response_model=VaultStatsResponse) async def api_reload_vault(vault_name: str, current_user=Depends(require_admin)): """Force a re-index of a single vault. Args: vault_name: Name of the vault to reindex. Returns: Dict with vault statistics. """ try: from backend.indexer import reload_single_vault stats = await reload_single_vault(vault_name) await sse_manager.broadcast("vault_reloaded", { "vault": vault_name, "stats": stats, }) return {"status": "ok", "vault": vault_name, "stats": stats} except ValueError as e: raise HTTPException(status_code=404, detail=str(e)) # --------------------------------------------------------------------------- # SSE endpoint — Server-Sent Events stream # --------------------------------------------------------------------------- @app.get( "/api/events", response_class=StreamingResponse, responses={200: {"content": {"text/event-stream": {}}, "description": "Server-Sent Events stream"}}, ) async def api_events(current_user=Depends(require_auth)): """SSE stream for real-time index update notifications. Sends keepalive comments every 30s. Events: - ``index_updated``: partial index change (file create/modify/delete/move) - ``index_reloaded``: full re-index completed - ``vault_added``: new vault added dynamically - ``vault_removed``: vault removed dynamically """ queue = await sse_manager.connect() async def event_generator(): try: # Send initial connection event yield f"event: connected\ndata: {_json.dumps({'sse_clients': sse_manager.client_count})}\n\n" while True: try: msg = await asyncio.wait_for(queue.get(), timeout=30.0) yield f"event: {msg['event']}\ndata: {msg['data']}\n\n" except asyncio.TimeoutError: # Keepalive comment yield ": keepalive\n\n" except asyncio.CancelledError: break finally: sse_manager.disconnect(queue) return StreamingResponse( event_generator(), media_type="text/event-stream", headers={ "Cache-Control": "no-cache", "Connection": "keep-alive", "X-Accel-Buffering": "no", }, ) # --------------------------------------------------------------------------- # Dynamic vault management endpoints # --------------------------------------------------------------------------- @app.post("/api/vaults/add", response_model=VaultStatsResponse) async def api_add_vault(body: dict = Body(...), current_user=Depends(require_admin)): """Add a new vault dynamically without restarting. Body: name: Display name for the vault. path: Absolute filesystem path to the vault directory. """ name = body.get("name", "").strip() vault_path = body.get("path", "").strip() if not name or not vault_path: raise HTTPException(status_code=400, detail="Both 'name' and 'path' are required") if name in index: raise HTTPException(status_code=409, detail=f"Vault '{name}' already exists") if not Path(vault_path).exists(): raise HTTPException(status_code=400, detail=f"Path does not exist: {vault_path}") stats = await add_vault_to_index(name, vault_path) # Start watching the new vault if _vault_watcher: await _vault_watcher.add_vault(name, vault_path) await sse_manager.broadcast("vault_added", {"vault": name, "stats": stats}) return {"status": "ok", "vault": name, "stats": stats} @app.delete("/api/vaults/{vault_name}", response_model=VaultActionResponse) async def api_remove_vault(vault_name: str, current_user=Depends(require_admin)): """Remove a vault from the index and stop watching it. Args: vault_name: Name of the vault to remove. """ if vault_name not in index: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") # Stop watching if _vault_watcher: await _vault_watcher.remove_vault(vault_name) await remove_vault_from_index(vault_name) await sse_manager.broadcast("vault_removed", {"vault": vault_name}) return {"status": "ok", "vault": vault_name} @app.get("/api/vaults/status", response_model=VaultsStatusResponse) async def api_vaults_status(current_user=Depends(require_auth)): """Detailed status of all vaults including watcher state. Returns per-vault: file count, tag count, watching status, vault path. """ statuses = {} for vname, vdata in index.items(): watching = _vault_watcher is not None and vname in _vault_watcher.observers statuses[vname] = { "file_count": len(vdata.get("files", [])), "tag_count": len(vdata.get("tags", {})), "path": vdata.get("path", ""), "watching": watching, } return { "vaults": statuses, "watcher_active": _vault_watcher is not None, "sse_clients": sse_manager.client_count, } @app.get( "/api/image/{vault_name}", response_class=Response, responses={200: {"content": {"application/octet-stream": {}}, "description": "Image bytes"}}, ) async def api_image(vault_name: str, path: str = Query(..., description="Relative path to image"), current_user=Depends(require_auth)): """Serve an image file with proper MIME type. Args: vault_name: Name of the vault. path: Relative file path within the vault. Returns: Image file with appropriate content-type header. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, path) if not file_path.exists() or not file_path.is_file(): raise HTTPException(status_code=404, detail=f"Image not found: {path}") # Determine MIME type mime_type, _ = mimetypes.guess_type(str(file_path)) if not mime_type: # Default to octet-stream if unknown mime_type = "application/octet-stream" try: # Read and return the image file content = file_path.read_bytes() return Response(content=content, media_type=mime_type) except PermissionError: raise HTTPException(status_code=403, detail="Permission denied") except Exception as e: logger.error(f"Error serving image {vault_name}/{path}: {e}") raise HTTPException(status_code=500, detail=f"Error serving image: {e!s}") @app.post("/api/attachments/rescan/{vault_name}", response_model=AttachmentRescanResponse) async def api_rescan_attachments(vault_name: str, current_user=Depends(require_admin)): """Rescan attachments for a specific vault. Args: vault_name: Name of the vault to rescan. Returns: Dict with status and attachment count. """ vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") vault_path = vault_data["path"] count = await rescan_vault_attachments(vault_name, vault_path) logger.info(f"Rescanned attachments for vault '{vault_name}': {count} attachments") return {"status": "ok", "vault": vault_name, "attachment_count": count} @app.get("/api/attachments/stats", response_model=AttachmentStatsResponse) async def api_attachment_stats(vault: str | None = Query(None, description="Vault filter"), current_user=Depends(require_auth)): """Get attachment statistics for vaults. Args: vault: Optional vault name to filter stats. Returns: Dict with vault names as keys and attachment counts as values. """ stats = get_attachment_stats(vault) return {"vaults": stats} # --------------------------------------------------------------------------- # Vault Settings API — Display preferences # --------------------------------------------------------------------------- @app.get("/api/vaults/{vault_name}/settings", response_model=VaultSettingsResponse) async def api_get_vault_settings(vault_name: str, current_user=Depends(require_auth)): """Get UI display settings for a specific vault. Args: vault_name: Name of the vault. Returns: Dict with vault settings including hideHiddenFiles. """ if vault_name not in index: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") # Get persisted settings persisted = get_vault_setting(vault_name) or {} # Default settings settings = { "hideHiddenFiles": False, } settings.update(persisted) return settings @app.post("/api/vaults/{vault_name}/settings", response_model=VaultSettingsResponse) async def api_update_vault_settings(vault_name: str, body: dict = Body(...), current_user=Depends(require_admin)): """Update UI display settings for a specific vault. Args: vault_name: Name of the vault. body: Dict with settings to update (hideHiddenFiles). Returns: Updated settings dict. """ if vault_name not in index: raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found") # Validate settings settings_to_update = {} if "hideHiddenFiles" in body: if not isinstance(body["hideHiddenFiles"], bool): raise HTTPException(status_code=400, detail="hideHiddenFiles must be a boolean") settings_to_update["hideHiddenFiles"] = body["hideHiddenFiles"] # Update persisted settings try: updated = update_vault_setting(vault_name, settings_to_update) except PermissionError as e: logger.error(f"Permission error saving settings for vault '{vault_name}': {e}") raise HTTPException( status_code=500, detail="Permission denied: Cannot write to settings file. Check /app/data permissions." ) except Exception as e: logger.error(f"Error saving settings for vault '{vault_name}': {e}") raise HTTPException( status_code=500, detail=f"Failed to save settings: {e!s}" ) logger.info(f"Updated settings for vault '{vault_name}': {settings_to_update}") return updated @app.get("/api/vault/{vault_name}/files", response_model=VaultFilesResponse) async def api_vault_recent_files( vault_name: str, dir: str = Query("", description="Directory path within the vault (empty = root)"), limit: int = Query(200, description="Maximum number of files to return"), recursive: bool = Query(True, description="If true, list files recursively from directory and all subdirectories"), current_user=Depends(require_auth), ): """List files in a vault directory sorted by modification time (newest first). Returns file metadata suitable for a vault home page display. Unlike /api/browse, this endpoint sorts by mtime and returns additional metadata (size, modified time, extension). When recursive=True (default), lists files from the directory AND all its subdirectories, with a ``rel_dir`` field indicating the subdirectory path relative to the requested directory. Args: vault_name: Name of the vault. dir: Relative directory path within the vault (empty for root). limit: Maximum files to return (default 200). recursive: If true, recursively list files in subdirectories (default true). Returns: JSON with vault, directory, count, recursive flag, and list of file entries. """ if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'") return list_all_files(vault_name, dir=dir, limit=limit, recursive=recursive) @app.get("/api/vaults/settings/all", response_model=AllVaultSettingsResponse) async def api_get_all_vault_settings(current_user=Depends(require_auth)): """Get UI display settings for all vaults. Returns: Dict mapping vault names to their settings. """ all_settings = {} for vault_name in index: persisted = get_vault_setting(vault_name) or {} settings = { "hideHiddenFiles": False, } settings.update(persisted) all_settings[vault_name] = settings return all_settings # --------------------------------------------------------------------------- # Backup Management API # --------------------------------------------------------------------------- @app.get("/api/backups", response_model=BackupsListResponse) async def api_backups_list( vault: str | None = Query(None, description="Filter by vault name"), current_user=Depends(require_auth), ): """List all backups across vaults, grouped by file.""" result: list[dict[str, Any]] = [] try: for vault_name in index: if vault and vault_name != vault: continue if not check_vault_access(vault_name, current_user): continue vd = get_vault_data(vault_name) if not vd: continue vault_root = Path(vd["path"]) backup_root = Path(os.environ.get("OBSIGATE_BACKUP_DIR", ".obsigate-backup")) if not backup_root.is_absolute(): backup_root = vault_root / backup_root vault_backup_dir = backup_root / vault_name if not vault_backup_dir.exists(): continue for fpath in vault_backup_dir.rglob("*.bak"): if not fpath.is_file(): continue st = fpath.stat() fsize = st.st_size ts_part = fpath.name.rsplit(".", 2) if len(ts_part) < 3 or not ts_part[-2].isdigit(): continue ts = int(ts_part[-2]) rel_dir = str(fpath.parent.relative_to(vault_backup_dir)).replace("\\", "/") rel_file = rel_dir + "/" + ts_part[0] if rel_dir != "." else ts_part[0] result.append({ "vault": vault_name, "file": rel_file, "backup_file": fpath.name, "timestamp": ts, "datetime": datetime.fromtimestamp(ts, tz=timezone.utc).isoformat(), "size": fsize, "full_path": str(fpath), }) result.sort(key=lambda x: x["timestamp"], reverse=True) total_size = sum(r["size"] for r in result) return {"backups": result, "total": len(result), "total_size_bytes": total_size} except Exception as e: logger.error(f"Error listing backups: {type(e).__name__}: {e}", exc_info=True) raise HTTPException(status_code=500, detail=f"Erreur listing backups: {e!s}") @app.post("/api/backups/delete", response_model=BackupsDeletedResponse) async def api_backups_delete( body: dict = Body(...), current_user=Depends(require_auth), ): """Delete one or more backup files.""" paths = body.get("paths", []) if not paths: raise HTTPException(status_code=400, detail="No backup paths provided") deleted = 0 for p in paths: try: fpath = Path(p) # Security: ensure path is within a backup directory if ".obsigate-backup" not in str(fpath): continue if fpath.exists() and fpath.is_file(): fpath.unlink() deleted += 1 except Exception as e: logger.warning(f"Failed to delete backup {p}: {e}") return {"deleted": deleted} @app.post("/api/backups/purge", response_model=BackupsDeletedResponse) async def api_backups_purge( body: dict = Body(...), current_user=Depends(require_auth), ): """Purge all backups for a specific file or entire vault.""" vault_name = body.get("vault") file_path = body.get("file") # optional if not vault_name: raise HTTPException(status_code=400, detail="Vault name required") if not check_vault_access(vault_name, current_user): raise HTTPException(status_code=403, detail="Access denied") vd = get_vault_data(vault_name) if not vd: raise HTTPException(status_code=404, detail="Vault not found") vault_root = Path(vd["path"]) backup_root = Path(os.environ.get("OBSIGATE_BACKUP_DIR", ".obsigate-backup")) if not backup_root.is_absolute(): backup_root = vault_root / backup_root if file_path: # Delete backups for specific file backup_dir = backup_root / vault_name / Path(file_path).parent if backup_dir.exists(): fname = Path(file_path).name deleted = 0 for f in backup_dir.iterdir(): if f.is_file() and f.name.startswith(fname + ".") and f.name.endswith(".bak"): f.unlink() deleted += 1 return {"deleted": deleted} return {"deleted": 0} else: # Delete all backups for vault vault_backup_dir = backup_root / vault_name if vault_backup_dir.exists(): deleted = 0 for f in vault_backup_dir.rglob("*.bak"): if f.is_file(): f.unlink() deleted += 1 return {"deleted": deleted} return {"deleted": 0} @app.get("/api/backups/content", response_model=BackupContentResponse) async def api_backups_content( path: str = Query(..., description="Full path to backup file"), current_user=Depends(require_auth), ): """Return the content of a specific backup file.""" try: fpath = Path(path) if ".obsigate-backup" not in str(fpath): raise HTTPException(status_code=403, detail="Access denied") if not fpath.exists() or not fpath.is_file(): raise HTTPException(status_code=404, detail="Backup not found") content = fpath.read_text(encoding="utf-8", errors="replace") # Truncate large files to 100KB if len(content) > 102400: content = content[:102400] + "\n\n... (tronque a 100 Ko)" return {"content": content, "name": fpath.name, "size": len(content)} except HTTPException: raise except Exception as e: raise HTTPException(status_code=500, detail=str(e)) @app.post("/api/backups/compress", response_model=BackupsCompressResponse) async def api_backups_compress( body: dict = Body(...), current_user=Depends(require_auth), ): """Compress backups older than N days. Body: {older_than_days: 30, dry_run: false}""" import gzip as gz_mod older_than = body.get("older_than_days", 30) dry_run = body.get("dry_run", False) cutoff = time.time() - (older_than * 86400) compressed = 0 saved_bytes = 0 for vault_name in index: if not check_vault_access(vault_name, current_user): continue vd = get_vault_data(vault_name) if not vd: continue vault_root = Path(vd["path"]) backup_root = Path(os.environ.get("OBSIGATE_BACKUP_DIR", ".obsigate-backup")) if not backup_root.is_absolute(): backup_root = vault_root / backup_root vault_dir = backup_root / vault_name if not vault_dir.exists(): continue for fpath in vault_dir.rglob("*.bak"): if not fpath.is_file(): continue if fpath.name.endswith(".bak.gz"): continue mtime = fpath.stat().st_mtime if mtime > cutoff: continue if not dry_run: try: gz_path = fpath.with_suffix(fpath.suffix + ".gz") data = fpath.read_bytes() with gz_mod.open(str(gz_path), "wb", compresslevel=6) as gzf: gzf.write(data) orig_size = len(data) gz_size = gz_path.stat().st_size if gz_size < orig_size: fpath.unlink() saved_bytes += (orig_size - gz_size) else: gz_path.unlink() # compression didn't help compressed += 1 except Exception as e: logger.warning(f"Failed to compress {fpath}: {e}") else: compressed += 1 return {"compressed": compressed, "saved_bytes": saved_bytes, "dry_run": dry_run} @app.post("/api/backups/auto", response_model=BackupsAutoResponse) async def api_backups_auto( body: dict = Body(...), current_user=Depends(require_auth), ): """Create backups for files modified since a given time. Body: {since_hours: 24}""" since_hours = body.get("since_hours", 24) cutoff = time.time() - (since_hours * 3600) backed_up = 0 for vault_name in index: if not check_vault_access(vault_name, current_user): continue vd = get_vault_data(vault_name) if not vd: continue vault_root = Path(vd["path"]) for fpath in vault_root.rglob("*"): if not fpath.is_file(): continue if fpath.name.startswith('.'): continue if any(p.startswith('.') or p in {'.obsidian', '.trash', '.git', '.obsigate-backup', '__pycache__', 'node_modules'} for p in fpath.relative_to(vault_root).parts): continue mtime = fpath.stat().st_mtime if mtime < cutoff: continue try: rel = str(fpath.relative_to(vault_root)).replace("\\", "/") _backup_file(fpath, vault_name, rel) backed_up += 1 except Exception as e: logger.warning(f"Auto-backup failed for {rel}: {e}") return {"backed_up": backed_up, "since_hours": since_hours} # --------------------------------------------------------------------------- # Configuration API # --------------------------------------------------------------------------- _BASE_DIR = Path(__file__).resolve().parent.parent _CONFIG_PATH = _BASE_DIR / "data" / "config.json" _DEFAULT_CONFIG = { "search_workers": 2, "debounce_ms": 300, "results_per_page": 50, "min_query_length": 2, "search_timeout_ms": 30000, "max_content_size": 100000, "snippet_context_chars": 120, "max_snippet_highlights": 5, "title_boost": 3.0, "path_boost": 1.5, "watcher_enabled": True, "watcher_use_polling": False, "watcher_polling_interval": 5.0, "watcher_debounce": 2.0, "tag_boost": 2.0, "prefix_max_expansions": 50, "recent_files_limit": 20, "max_backups_per_file": 10, "ai_default_provider": "deepseek", "ai_default_models": {}, } def _load_config() -> dict: """Load config from disk, merging with defaults.""" config = dict(_DEFAULT_CONFIG) if _CONFIG_PATH.exists(): try: stored = _json.loads(_CONFIG_PATH.read_text(encoding="utf-8")) config.update(stored) except Exception as e: logger.warning(f"Failed to read config.json: {e}") return config def _save_config(config: dict) -> None: """Persist config to disk.""" try: _CONFIG_PATH.write_text( _json.dumps(config, indent=2, ensure_ascii=False), encoding="utf-8", ) except Exception as e: logger.error(f"Failed to write config.json: {e}") raise HTTPException(status_code=500, detail=f"Failed to save config: {e}") @app.get("/api/config", response_model=AppConfigResponse) async def api_get_config(current_user=Depends(require_auth)): """Return current configuration with defaults for missing keys.""" return _load_config() @app.post("/api/config", response_model=AppConfigResponse) async def api_set_config(body: dict = Body(...), current_user=Depends(require_admin)): """Update configuration. Only known keys are accepted. Keys matching ``_DEFAULT_CONFIG`` are validated and persisted. Unknown keys are silently ignored. Returns the full merged config after update. """ current = _load_config() updated_keys = [] for key, value in body.items(): if key in _DEFAULT_CONFIG: expected_type = type(_DEFAULT_CONFIG[key]) if isinstance(value, expected_type) or (expected_type is float and isinstance(value, (int, float))): current[key] = value updated_keys.append(key) else: raise HTTPException( status_code=400, detail=f"Invalid type for '{key}': expected {expected_type.__name__}, got {type(value).__name__}", ) _save_config(current) if any(k.startswith("ai_") for k in updated_keys): try: from backend.ai import reload_ai_config reload_ai_config() except Exception as e: logger.warning(f"Failed to reload AI config: {e}") logger.info(f"Config updated: {updated_keys}") return current # --------------------------------------------------------------------------- # AI API Keys — stored in data/api_keys.json, fallback to .env # --------------------------------------------------------------------------- from backend.ai import PROVIDERS, _read_ai_keys, get_ai_key AI_KEYS_FILE = Path("data/api_keys.json") def _write_ai_keys(data: dict): AI_KEYS_FILE.parent.mkdir(parents=True, exist_ok=True) tmp = AI_KEYS_FILE.with_suffix(".tmp") tmp.write_text(_json.dumps(data, indent=2), encoding="utf-8") tmp.replace(AI_KEYS_FILE) @app.get("/api/config/ai-keys", response_model=AIKeysResponse) async def api_get_ai_keys(current_user=Depends(require_admin)): """Return stored AI keys (values masked).""" keys = _read_ai_keys() masked = {} for k in ["DEEPSEEK_API_KEY", "OPENROUTER_API_KEY", "GEMINI_API_KEY", "NVIDIA_API_KEY", "QWENCLOUD_API_KEY", "XIAOMI_API_KEY", "MISTRAL_API_KEY"]: val = keys.get(k, "") or os.environ.get(k, "") if val: masked[k] = val[:4] + "..." + val[-4:] if len(val) > 8 else "***" else: masked[k] = "" return masked @app.post("/api/config/ai-keys", response_model=StatusResponse) async def api_set_ai_keys(body: dict = Body(...), current_user=Depends(require_admin)): """Save AI keys. Pass {"DEEPSEEK_API_KEY":"sk-...","OPENROUTER_API_KEY":"...","GEMINI_API_KEY":"..."}""" keys = _read_ai_keys() for k in ["DEEPSEEK_API_KEY", "OPENROUTER_API_KEY", "GEMINI_API_KEY", "NVIDIA_API_KEY", "QWENCLOUD_API_KEY", "XIAOMI_API_KEY", "MISTRAL_API_KEY"]: if body.get(k): keys[k] = body[k] _write_ai_keys(keys) logger.info("AI keys updated") return {"status": "ok"} @app.delete("/api/config/ai-keys/{provider_env}", response_model=AIKeyDeleteResponse) async def api_delete_ai_key(provider_env: str, current_user=Depends(require_admin)): """Delete a specific AI provider key from storage.""" allowed = {"DEEPSEEK_API_KEY", "OPENROUTER_API_KEY", "GEMINI_API_KEY", "NVIDIA_API_KEY", "QWENCLOUD_API_KEY", "XIAOMI_API_KEY", "MISTRAL_API_KEY"} key_name = provider_env.upper() if key_name not in allowed: raise HTTPException(status_code=400, detail=f"Clé inconnue: {provider_env}") keys = _read_ai_keys() if key_name in keys: del keys[key_name] _write_ai_keys(keys) # Also clear from env at runtime so get_ai_key() no longer finds it os.environ.pop(key_name, None) logger.info(f"AI key deleted: {key_name}") return {"status": "deleted", "key": key_name} # --------------------------------------------------------------------------- # Tool & connected-source keys (#103) — same store as the AI provider keys # --------------------------------------------------------------------------- from backend.tools.secrets import ( TOOL_KEY_NAMES as _TOOL_KEY_NAMES, ) from backend.tools.secrets import ( delete_tool_key as _delete_tool_key, ) from backend.tools.secrets import ( get_tool_key as _get_tool_key, ) from backend.tools.secrets import ( mask_value as _mask_tool_value, ) from backend.tools.secrets import ( set_tool_key as _set_tool_key, ) @app.get("/api/config/tool-keys", response_model=AIKeysResponse) async def api_get_tool_keys(current_user=Depends(require_admin)): """Return tool/connected-source configuration (tokens masked, URLs clear).""" masked = {} for name in _TOOL_KEY_NAMES: masked[name] = _mask_tool_value(name, _get_tool_key(name)) return masked @app.post("/api/config/tool-keys", response_model=StatusResponse) async def api_set_tool_keys(body: dict = Body(...), current_user=Depends(require_admin)): """Save tool/connected-source keys. Only whitelisted names (``backend.tools.secrets.TOOL_KEY_NAMES``) are accepted: Tavily/Brave/SerpAPI/Exa API keys, Gitea URL + token, GitHub token. Empty values delete the stored entry. """ updated = [] for name, value in body.items(): if name not in _TOOL_KEY_NAMES: raise HTTPException(status_code=400, detail=f"Clé inconnue: {name}") if value is not None and not isinstance(value, str): raise HTTPException(status_code=400, detail=f"Type invalide pour {name}") _set_tool_key(name, value or "") updated.append(name) logger.info(f"Tool keys updated: {updated}") return {"status": "ok"} @app.delete("/api/config/tool-keys/{name}", response_model=AIKeyDeleteResponse) async def api_delete_tool_key(name: str, current_user=Depends(require_admin)): """Delete a stored tool key (the environment fallback still applies).""" key_name = name.upper() try: existed = _delete_tool_key(key_name) except ValueError as e: raise HTTPException(status_code=400, detail=str(e)) logger.info(f"Tool key deleted: {key_name} (existed={existed})") return {"status": "deleted", "key": key_name} @app.post("/api/config/ai-keys/test", response_model=AITestResponse) async def api_test_ai_keys(current_user=Depends(require_admin)): """Test which AI providers are configured. Each provider has a dedicated (URL, header-name) test pair. - Most OpenAI-compatible APIs use `Authorization: Bearer KEY` - Xiaomi MiMo uses `api-key: KEY` - Gemini uses a query-string key """ results = {} for key_name, label, test_url_tmpl, header_name in [ # OpenAI-compatible — Authorization: Bearer ("DEEPSEEK_API_KEY", "deepseek", "https://api.deepseek.com/v1/models", "Authorization"), ("OPENROUTER_API_KEY","openrouter", "https://openrouter.ai/api/v1/models", "Authorization"), ("NVIDIA_API_KEY", "nvidia", "https://integrate.api.nvidia.com/v1/models", "Authorization"), ("QWENCLOUD_API_KEY", "qwencloud", "https://dashscope.aliyuncs.com/compatible-mode/v1/models", "Authorization"), ("MISTRAL_API_KEY", "mistral", "https://api.mistral.ai/v1/models", "Authorization"), # Xiaomi MiMo — dedicated api-key header (NOT Authorization: Bearer) ("XIAOMI_API_KEY", "xiaomi", "https://api.xiaomimimo.com/v1/models", "api-key"), # Gemini — key in query string ("GEMINI_API_KEY", "gemini", "https://generativelanguage.googleapis.com/v1beta/models?key={key}", None), ]: key = get_ai_key(key_name) if not key: results[label] = "non configuré" continue try: import urllib.request url = test_url_tmpl.replace("{key}", key) if "{key}" in test_url_tmpl else test_url_tmpl if header_name: req = urllib.request.Request(url, headers={header_name: key}) else: req = urllib.request.Request(url) urllib.request.urlopen(req, timeout=5) results[label] = "ok" except Exception as e: # Truncate the error to keep the response small. results[label] = "erreur: " + str(e)[:80] return results # --------------------------------------------------------------------------- # AI Models — list available models per provider # --------------------------------------------------------------------------- @app.get("/api/config/ai-models", response_model=AIModelsResponse) async def api_list_ai_models(provider: str = Query(...), current_user=Depends(require_admin)): """List available models for a given AI provider. Strategy: 1. Try the provider's public models endpoint (OpenAI-compatible /v1/models or Gemini). 2. If the network call fails (timeout, 4xx, 5xx, DNS, etc.), fall back to a curated static list of known-good models for that provider. 3. Always return a non-empty list when the provider is known, so the UI dropdown is never empty. """ provider = provider.lower() from backend.model_capabilities import get_capabilities_for_models from backend.provider_capabilities import remember_declared_capabilities all_providers = ("deepseek", "openrouter", "gemini", "nvidia", "qwencloud", "xiaomi", "mistral") if provider not in all_providers: return {"models": [], "error": f"Unknown provider: {provider}", "source": "validation"} key_name = f"{provider.upper()}_API_KEY" key = get_ai_key(key_name) if not key: # No key configured — return curated fallback list so the UI can # still show what WOULD be available once a key is set. fallback = _FALLBACK_MODELS.get(provider, []) return {"models": fallback, "source": "fallback", "capabilities": get_capabilities_for_models(provider, fallback), "note": "API key not configured — showing default model list"} # Build URL if provider == "gemini": url = f"https://generativelanguage.googleapis.com/v1beta/models?key={key}" elif provider == "deepseek": url = "https://api.deepseek.com/v1/models" elif provider == "openrouter": url = "https://openrouter.ai/api/v1/models" elif provider == "nvidia": url = "https://integrate.api.nvidia.com/v1/models" elif provider == "qwencloud": url = "https://dashscope.aliyuncs.com/compatible-mode/v1/models" elif provider == "xiaomi": # Xiaomi MiMo — dedicated api-key header (NOT Authorization: Bearer). # Endpoint: https://api.xiaomimimo.com/v1/models url = "https://api.xiaomimimo.com/v1/models" models = [] # parsed below with the custom header elif provider == "mistral": url = "https://api.mistral.ai/v1/models" try: if provider == "gemini": req = urllib.request.Request(url) elif provider == "xiaomi": # Xiaomi MiMo uses a dedicated api-key header. req = urllib.request.Request(url, headers={"api-key": key}) else: req = urllib.request.Request(url, headers={"Authorization": "Bearer " + key}) with urllib.request.urlopen(req, timeout=10) as resp: data = _json.loads(resp.read().decode()) if provider == "gemini": models = [m.get("name", "") for m in data.get("models", []) if m.get("name")] # Gemini returns names like "models/gemini-1.5-flash" — strip prefix models = [m.replace("models/", "") for m in models] else: models = [m.get("id", "") for m in data.get("data", []) if m.get("id")] # Cache the capabilities the provider declares for these models # (BUG-044) — get_capabilities_for_models() below then returns the # provider's own truth for the flags it declares, the curated table # for the rest. Providers that declare nothing are left untouched. remember_declared_capabilities(provider, data) if models: # Prepend the configured default if not already present default = PROVIDERS.get(provider, {}).get("model") if default and default not in models: models = [default] + models return {"models": models, "source": "live", "count": len(models), "capabilities": get_capabilities_for_models(provider, models)} # Empty list from API — fall through to fallback raise ValueError("empty model list from provider API") except Exception as e: # Network error, auth error, parsing error — use curated fallback fallback = _FALLBACK_MODELS.get(provider, []) return {"models": fallback, "source": "fallback", "error": str(e)[:200], "capabilities": get_capabilities_for_models(provider, fallback), "note": "Could not reach provider API — showing default model list"} # ── Curated fallback model lists ────────────────────────────────────────── # Used when the provider API is unreachable or returns empty. # Keep these short and focused on models known to work with the # OpenAI-compatible chat completions interface (or Gemini's generateContent). _FALLBACK_MODELS: dict[str, list[str]] = { "deepseek": [ "deepseek-chat", "deepseek-reasoner", ], "openrouter": [ "openai/gpt-4o-mini", "openai/gpt-4o", "anthropic/claude-3.5-sonnet", "anthropic/claude-3-haiku", "google/gemini-2.0-flash-exp:free", "meta-llama/llama-3.1-70b-instruct", "meta-llama/llama-3.1-8b-instruct:free", "mistralai/mistral-large-latest", ], "gemini": [ "gemini-2.0-flash", "gemini-2.0-flash-exp", "gemini-1.5-pro", "gemini-1.5-flash", "gemini-1.5-flash-8b", ], "nvidia": [ "meta/llama-3.1-405b-instruct", "meta/llama-3.1-70b-instruct", "meta/llama-3.1-8b-instruct", "mistralai/mistral-large", "google/gemma-2-27b-it", "nvidia/llama-3.1-nemotron-70b-instruct", ], "qwencloud": [ "qwen-max", "qwen-plus", "qwen-turbo", "qwen-long", "qwen-vl-max", "qwen-vl-plus", ], "xiaomi": [ # Xiaomi MiMo models — the public /v1/models endpoint requires the # `api-key` custom header (NOT Authorization: Bearer), so the live # call often fails with 401 even with the right key. We ship a # known-good list as fallback. See https://mimo.mi.com/docs/ "mimo-v2.5-pro", "mimo-v2.5", "mimo-v2.5-asr", "mimo-v2.5-tts", "mimo-v2.5-tts-voiceclone", "mimo-v2.5-tts-voicedesign", ], "mistral": [ "mistral-large-latest", "mistral-medium-latest", "mistral-small-latest", "open-mistral-7b", "open-mixtral-8x7b", "codestral-latest", ], } # --------------------------------------------------------------------------- # Diagnostics API # --------------------------------------------------------------------------- @app.get("/api/diagnostics", response_model=DiagnosticsResponse) async def api_diagnostics(current_user=Depends(require_admin)): """Return index statistics and system diagnostics. Includes document counts, token counts, memory estimates, and inverted index status. """ import sys from backend.search import get_inverted_index inv = get_inverted_index() # Per-vault stats vault_stats = {} total_files = 0 total_tags = 0 for vname, vdata in index.items(): file_count = len(vdata.get("files", [])) tag_count = len(vdata.get("tags", {})) vault_stats[vname] = {"file_count": file_count, "tag_count": tag_count} total_files += file_count total_tags += tag_count # Memory estimate for inverted index word_index_entries = sum(len(docs) for docs in inv.word_index.values()) mem_estimate_mb = round( (sys.getsizeof(inv.word_index) + word_index_entries * 80 + len(inv.doc_info) * 200 + len(inv._sorted_tokens) * 60) / (1024 * 1024), 2 ) return { "index": { "total_files": total_files, "total_tags": total_tags, "vaults": vault_stats, }, "inverted_index": { "unique_tokens": len(inv.word_index), "total_postings": word_index_entries, "documents": inv.doc_count, "sorted_tokens": len(inv._sorted_tokens), "is_stale": inv.is_stale(), "memory_estimate_mb": mem_estimate_mb, }, "config": _load_config(), "search_executor": { "active": _search_executor is not None, "max_workers": _search_executor._max_workers if _search_executor else 0, }, } # --------------------------------------------------------------------------- # Dashboard endpoint (aggregated stats) # --------------------------------------------------------------------------- @app.get("/api/dashboard", response_model=DashboardResponse) async def api_dashboard(current_user=Depends(require_auth)): """Aggregated dashboard statistics across all accessible vaults.""" user_vaults = current_user.get("_token_vaults") or current_user.get("vaults", []) vault_stats = [] total_files = 0 total_tags = set() total_size = 0 for vname, vdata in index.items(): if "*" not in user_vaults and vname not in user_vaults: continue files = vdata.get("files", []) fc = len(files) total_files += fc vtags = set() vsize = 0 for f in files: vtags.update(f.get("tags", [])) vsize += f.get("size", 0) total_tags.update(vtags) total_size += vsize vault_stats.append({"name": vname, "file_count": fc, "tag_count": len(vtags), "total_size_bytes": vsize}) return {"vaults": vault_stats, "total_files": total_files, "total_tags": len(total_tags), "total_size_bytes": total_size} # --------------------------------------------------------------------------- # Webhook CRUD endpoints # --------------------------------------------------------------------------- @app.get("/api/webhooks", response_model=list[WebhookModel]) async def api_webhooks_list(current_user=Depends(require_admin)): return get_webhooks() @app.post("/api/webhooks", response_model=WebhookModel) async def api_webhooks_create(body: dict = Body(...), current_user=Depends(require_admin)): name = body.get("name", "Unnamed") url = body.get("url", "") events = body.get("events", []) secret = body.get("secret") if not url: raise HTTPException(400, "URL is required") return create_webhook(name, url, events, secret) @app.patch("/api/webhooks/{webhook_id}", response_model=WebhookModel) async def api_webhooks_update(webhook_id: str, body: dict = Body(...), current_user=Depends(require_admin)): result = update_webhook(webhook_id, body) if not result: raise HTTPException(404, "Webhook not found") return result @app.delete("/api/webhooks/{webhook_id}", response_model=StatusResponse) async def api_webhooks_delete(webhook_id: str, current_user=Depends(require_admin)): if not delete_webhook(webhook_id): raise HTTPException(404, "Webhook not found") return {"status": "deleted"} # --------------------------------------------------------------------------- # Share (public document) endpoints # --------------------------------------------------------------------------- @app.post("/api/share/{vault_name}", response_model=ShareModel) async def api_share_create( vault_name: str, body: dict = Body(...), current_user=Depends(require_auth), ): """Create a public share link for a document. Also sets ``publish: true`` in the file's YAML frontmatter so the frontend can visually indicate the file is publicly shared. """ if not check_vault_access(vault_name, current_user): raise HTTPException(403, f"Accès refusé à la vault '{vault_name}'") path = body.get("path", "") expires = body.get("expires_in_hours") share = create_share(vault_name, path, current_user["username"], expires) share["url"] = f"/s/{share['token']}" # Set publish: true in the file's frontmatter vault_data = get_vault_data(vault_name) if vault_data: file_path = _resolve_safe_path(Path(vault_data["path"]), path) if file_path.exists() and file_path.suffix == ".md": try: raw = file_path.read_text(encoding="utf-8", errors="replace") post = frontmatter.loads(raw) if not post.metadata.get("publish"): post.metadata["publish"] = True new_raw = frontmatter.dumps(post) _backup_file(file_path, vault_name, path) file_path.write_text(new_raw, encoding="utf-8") await update_single_file(vault_name, str(file_path)) logger.info(f"Set publish:true on {vault_name}/{path}") except Exception as e: logger.warning(f"Failed to set publish metadata on {vault_name}/{path}: {e}") return share @app.get("/api/shares", response_model=list[ShareModel]) async def api_shares_list(vault: str | None = Query(None), current_user=Depends(require_auth)): """List all shares (optionally filtered by vault).""" shares = list_shares(vault) for s in shares: s["url"] = f"/s/{s['token']}" return shares @app.delete("/api/share/{share_id}", response_model=StatusResponse) async def api_share_revoke(share_id: str, current_user=Depends(require_auth)): if not revoke_share(share_id): raise HTTPException(404, "Share not found") return {"status": "revoked"} @app.get( "/s/{token}/pdf", response_class=Response, responses={200: {"content": {"application/pdf": {}}, "description": "Shared document as PDF"}}, ) async def public_share_pdf_download(token: str): """Download shared document as real PDF via WeasyPrint.""" if generate_pdf is None: raise HTTPException(501, "PDF export unavailable (WeasyPrint/GTK not available)") share = get_share_by_token(token) if not share: raise HTTPException(404, "Share not found or expired") vault_data = get_vault_data(share["vault"]) if not vault_data: raise HTTPException(404, "Vault not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, share["path"]) if not file_path.exists(): raise HTTPException(404, "File not found") try: raw = file_path.read_text(encoding="utf-8", errors="replace") except Exception: raise HTTPException(500, "Cannot read file") record_access(token) raw = redact_file_content(raw, str(file_path)) post = parse_markdown_file(raw) ext = file_path.suffix.lower() if ext == ".md": html = _render_markdown(post.content, share["vault"], file_path) else: html = f'
{html_mod.escape(raw)}
' title = post.metadata.get("title", file_path.stem) pdf_html = build_pdf_html(html, str(title)) pdf_bytes = generate_pdf(pdf_html, str(title)) safe_name = "".join(c for c in str(title) if c.isascii() and (c.isalnum() or c in " _-.")).strip() or "document" return Response(content=pdf_bytes, media_type="application/pdf", headers={"Content-Disposition": f'attachment; filename="{safe_name}.pdf"'}) @app.get("/s/{token}/raw", response_class=FileResponse) async def public_share_raw(token: str): """Download the raw (original) shared document.""" share = get_share_by_token(token) if not share: raise HTTPException(404, "Share not found or expired") vault_data = get_vault_data(share["vault"]) if not vault_data: raise HTTPException(404, "Vault not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, share["path"]) if not file_path.exists(): raise HTTPException(404, "File not found") record_access(token) return FileResponse(path=str(file_path), filename=file_path.name, media_type="application/octet-stream") @app.get("/s/{token}", response_class=HTMLResponse) async def public_share_view(token: str): """Public share view — no authentication required.""" share = get_share_by_token(token) if not share: raise HTTPException(404, "Share not found or expired") vault_data = get_vault_data(share["vault"]) if not vault_data: raise HTTPException(404, "Vault not found") vault_root = Path(vault_data["path"]) file_path = _resolve_safe_path(vault_root, share["path"]) if not file_path.exists(): raise HTTPException(404, "File not found") try: raw = file_path.read_text(encoding="utf-8", errors="replace") except Exception: raise HTTPException(500, "Cannot read file") record_access(token) raw = redact_file_content(raw, str(file_path)) post = parse_markdown_file(raw) ext = file_path.suffix.lower() if ext == ".md": html = _render_markdown(post.content, share["vault"], file_path) else: escaped = html_mod.escape(raw) html = f'
{escaped}
' title = post.metadata.get("title", file_path.stem) # Escape everything user-controlled before embedding in HTML/JS (BUG-022). title_esc = html_mod.escape(str(title)) # Neutralise ```` in the JS string literal too. title_download_js = ( _json.dumps(f"{title}.md") .replace("<", "\\u003c") .replace(">", "\\u003e") .replace("&", "\\u0026") ) # JSON-escape raw content for embedding in HTML, and neutralise ````. raw_json = ( _json.dumps(raw) .replace("<", "\\u003c") .replace(">", "\\u003e") .replace("&", "\\u0026") ) fm_html = "" if post.metadata: fm_items = [] skip_keys = {"title", "titre"} for k, v in post.metadata.items(): if k in skip_keys: continue if isinstance(v, list): v = ", ".join(str(x) for x in v) elif isinstance(v, bool): v = "✓" if v else "✗" elif v is None: v = "—" fm_items.append( f'
{html_mod.escape(str(k))}' f'{html_mod.escape(str(v))}
' ) if fm_items: fm_html = f'
Frontmatter
{"".join(fm_items)}
' return HTMLResponse(f""" {title_esc} — ObsiGate Share
{title_esc}
{fm_html}{html}
""") # --------------------------------------------------------------------------- # Syncthing conflict endpoints # --------------------------------------------------------------------------- @app.get("/api/conflicts", response_model=ConflictsResponse) async def api_conflicts(current_user=Depends(require_auth)): """List sync-conflict files across accessible vaults.""" user_vaults = current_user.get("_token_vaults") or current_user.get("vaults", []) all_conflicts = get_conflicts() if "*" not in user_vaults: all_conflicts = [c for c in all_conflicts if c["vault"] in user_vaults] return {"conflicts": all_conflicts, "total": len(all_conflicts)} @app.post("/api/conflicts/resolve", response_model=ConflictResolveResponse) async def api_conflict_resolve(body: dict = Body(...), current_user=Depends(require_auth)): """Resolve a conflict: keep_local (delete conflict file) or keep_conflict (replace original).""" vault_name = body.get("vault") conflict_path = body.get("conflict_path") original_path = body.get("original_path") action = body.get("action") # "keep_local" or "keep_conflict" # mypy: narrow down from dict values assert isinstance(vault_name, str), "'vault' is required and must be a string" assert isinstance(conflict_path, str), "'conflict_path' is required and must be a string" assert isinstance(original_path, str), "'original_path' is required and must be a string" if not check_vault_access(vault_name, current_user): raise HTTPException(403, f"Accès refusé à la vault '{vault_name}'") vault_data = get_vault_data(vault_name) if not vault_data: raise HTTPException(404, "Vault not found") vault_root = Path(vault_data["path"]) conf_file = _resolve_safe_path(vault_root, conflict_path) orig_file = _resolve_safe_path(vault_root, original_path) if not conf_file.exists(): raise HTTPException(404, "Conflict file not found") try: if action == "keep_conflict": _backup_file(orig_file, vault_name, original_path) shutil.copy2(conf_file, orig_file) logger.info(f"Conflict resolved (keep_conflict): {conflict_path} → {original_path}") conf_file.unlink() await remove_single_file(vault_name, conflict_path) log_file_delete(current_user["username"], vault_name, conflict_path) await sse_manager.broadcast("file_deleted", {"vault": vault_name, "path": conflict_path}) return {"status": "resolved", "action": action} except Exception as e: raise HTTPException(500, f"Error resolving conflict: {e!s}") # --------------------------------------------------------------------------- # Real-time collaboration — WebSocket endpoint (ROADMAP #62) # --------------------------------------------------------------------------- @app.websocket("/ws/collab/{vault_name}/{path:path}") async def collab_websocket(websocket: WebSocket, vault_name: str, path: str): """Real-time collaborative editing over WebSocket (ROADMAP #62). One *room* is created per ``vault::path``; all clients editing the same file share Yjs/CRDT updates, awareness (cursors/selection) and a debounced server-side persistence of the markdown content. Authentication is performed manually (FastAPI ``Depends`` do not run for WebSocket routes) and vault access is enforced per connection. """ from backend.services.errors import ServiceError from backend.services.vaults import get_vault_root user = authenticate_websocket(websocket) if user is None: await websocket.close(code=4401) return if not check_vault_access(vault_name, user): await websocket.close(code=4403) return try: vault_root = get_vault_root(vault_name) file_path = _resolve_safe_path(vault_root, path) except ServiceError: await websocket.close(code=4404) return if not file_path.exists() or not file_path.is_file(): await websocket.close(code=4404) return await websocket.accept() await collab_manager.connect(websocket, vault_name, path, file_path, user) # --------------------------------------------------------------------------- # Static files & SPA fallback # --------------------------------------------------------------------------- if FRONTEND_DIR.exists(): # ``Cache-Control`` for /static is set by SecurityHeadersMiddleware (no-cache). app.mount("/static", StaticFiles(directory=str(FRONTEND_DIR)), name="static") @app.get("/sw.js") async def serve_service_worker(): """Serve the service worker for PWA support.""" sw_file = FRONTEND_DIR / "sw.js" if sw_file.exists(): return FileResponse( sw_file, media_type="application/javascript", headers={ "Cache-Control": "no-cache, no-store, must-revalidate", "Service-Worker-Allowed": "/" } ) raise HTTPException(status_code=404, detail="Service worker not found") @app.get("/manifest.json") async def serve_manifest(): """Serve the PWA manifest.""" manifest_file = FRONTEND_DIR / "manifest.json" if manifest_file.exists(): return FileResponse( manifest_file, media_type="application/manifest+json", headers={"Cache-Control": "no-cache"} ) raise HTTPException(status_code=404, detail="Manifest not found") @app.get("/popout/{vault_name}/{path:path}") async def serve_popout(vault_name: str, path: str): """Serve the minimalist popout page for a specific file.""" popout_file = FRONTEND_DIR / "popout.html" if popout_file.exists(): return HTMLResponse(content=popout_file.read_text(encoding="utf-8"), headers={"Cache-Control": "no-cache"}) raise HTTPException(status_code=404, detail="Popout template not found") @app.get("/editor-poc") async def serve_editor_poc(): """Serve the standalone Editor POC page (multi-zone toolbar demo).""" poc_file = FRONTEND_DIR / "editor-poc.html" if poc_file.exists(): return HTMLResponse(content=poc_file.read_text(encoding="utf-8"), headers={"Cache-Control": "no-cache"}) raise HTTPException(status_code=404, detail="Editor POC not found") @app.get("/admin.html", response_class=HTMLResponse) async def serve_admin_page(_current_user=Depends(require_admin)): """Serve the admin dashboard page (ROADMAP #71) — admin-gated. Must be declared BEFORE the SPA catch-all ``/{full_path:path}`` or the admin page would be shadowed by ``index.html`` (the reported bug: the Admin menu kept returning to the main page). """ admin_file = FRONTEND_DIR / "admin.html" if admin_file.exists(): return HTMLResponse(content=admin_file.read_text(encoding="utf-8"), headers={"Cache-Control": "no-cache"}) raise HTTPException(status_code=404, detail="Admin page not found") @app.get("/{full_path:path}") async def serve_spa(full_path: str): """Serve the SPA index.html for all non-API routes.""" index_file = FRONTEND_DIR / "index.html" if index_file.exists(): return HTMLResponse(content=index_file.read_text(encoding="utf-8"), headers={"Cache-Control": "no-cache"}) raise HTTPException(status_code=404, detail="Frontend not found")