A17 — nouveau endpoint GET /api/file/{vault}/xlsx/dashboard (read_workbook_
dashboard : plages nommees avec portee depuis defined_names read_only,
comptage graphiques/TCD par parts OPC, stats par feuille bornées 500x40 :
cellules/lignes/colonnes/formules/numerique + 8 premieres valeurs en cartes
KPI) et panneau frontend toggled depuis la toolbar (table des plages,
cartes KPI par feuille, hint actions IA). Bouton absent pour .csv et
formats en lecture seule ; le menu structure est saute quand le bouton
n'existe pas. i18n FR/EN (xlsx.dashboard_*), 8 tests backend + 2 tests
JSDOM + contre-preuve (5 echecs sur neutralisation), ruff/mypy 0.
🤖 Generated with Codebuff
Co-Authored-By: Codebuff <[email protected]>
679 lines
27 KiB
Python
679 lines
27 KiB
Python
"""File browsing & reading endpoints (ROADMAP #85, tranche 6a).
|
|
|
|
Handlers déplacés depuis :mod:`backend.main` sans changement de
|
|
comportement : mêmes chemins (``/api/browse/*``, ``/api/file/*`` en
|
|
lecture), mêmes modèles de réponse (déménagés dans
|
|
:mod:`backend.schemas`), mêmes dépendances d'authentification.
|
|
|
|
Adaptations strictement équivalentes :
|
|
- ``_resolve_safe_path`` → :mod:`backend.services.paths` (pass-through).
|
|
- ``_render_markdown`` vient de :mod:`backend.render` (#85 T9, sans cycle
|
|
d'import).
|
|
- ``_content_disposition`` / ``_media_max_inline_bytes`` / ``EXT_TO_LANG``
|
|
ont déménagé : helpers partagés dans :mod:`backend.routers.helpers`
|
|
(``EXT_TO_LANG`` n'était utilisé que par la vue fichier).
|
|
"""
|
|
|
|
import html as html_mod
|
|
import logging
|
|
from pathlib import Path
|
|
from urllib.parse import quote
|
|
|
|
from fastapi import APIRouter, Depends, HTTPException, Query
|
|
from fastapi.responses import FileResponse
|
|
|
|
from backend.auth.middleware import check_vault_access, require_auth
|
|
from backend.history import record_open
|
|
from backend.indexer import (
|
|
_extract_tags,
|
|
get_backlinks,
|
|
get_vault_data,
|
|
parse_markdown_file,
|
|
)
|
|
from backend.media_types import is_audio, is_image, is_video, media_mime_type
|
|
from backend.render import _render_markdown
|
|
from backend.routers.helpers import media_max_inline_bytes
|
|
from backend.schemas import (
|
|
BacklinksResponse,
|
|
BrowseResponse,
|
|
FileContentResponse,
|
|
FileRawResponse,
|
|
XlsxDashboardResponse,
|
|
XlsxSheetWindowResponse,
|
|
)
|
|
from backend.services.files import read_raw_file
|
|
from backend.services.paths import resolve_safe_path
|
|
from backend.services.vaults import browse_directory, get_vault_root
|
|
|
|
logger = logging.getLogger("obsigate")
|
|
|
|
# Map file extensions to highlight.js language hints
|
|
EXT_TO_LANG = {
|
|
".py": "python", ".js": "javascript", ".ts": "typescript",
|
|
".jsx": "jsx", ".tsx": "tsx", ".sh": "bash", ".bash": "bash",
|
|
".zsh": "bash", ".fish": "fish", ".bat": "batch", ".cmd": "batch",
|
|
".ps1": "powershell", ".json": "json", ".yaml": "yaml", ".yml": "yaml",
|
|
".toml": "toml", ".xml": "xml", ".csv": "plaintext",
|
|
".cfg": "ini", ".ini": "ini", ".conf": "ini", ".env": "bash",
|
|
".html": "html", ".css": "css", ".scss": "scss", ".less": "less",
|
|
".java": "java", ".c": "c", ".cpp": "cpp", ".h": "c", ".hpp": "cpp",
|
|
".cs": "csharp", ".go": "go", ".rs": "rust", ".rb": "ruby",
|
|
".php": "php", ".sql": "sql", ".r": "r", ".swift": "swift",
|
|
".kt": "kotlin", ".txt": "plaintext", ".log": "plaintext",
|
|
".lua": "lua", ".pl": "perl", ".pm": "perl", ".ex": "elixir", ".exs": "elixir",
|
|
".dart": "dart", ".tf": "haskell", ".gradle": "groovy", ".groovy": "groovy",
|
|
".graphql": "graphql", ".gql": "graphql", ".prisma": "sql", ".proto": "c",
|
|
".vb": "basic", ".asm": "x86asm", ".s": "armasm",
|
|
".vue": "xml", ".svelte": "xml", ".astro": "xml",
|
|
".properties": "ini", ".service": "ini", ".hosts": "ini",
|
|
".ksh": "bash", ".dockerfile": "dockerfile",
|
|
".makefile": "makefile", ".cmake": "cmake",
|
|
}
|
|
|
|
router = APIRouter(tags=["files"])
|
|
|
|
|
|
@router.get("/api/browse/{vault_name}", response_model=BrowseResponse)
|
|
async def api_browse(vault_name: str, path: str = "", current_user=Depends(require_auth)):
|
|
"""Browse directories and files in a vault at a given path level.
|
|
|
|
Returns sorted entries (directories first, then files) with metadata.
|
|
Hidden files/directories (starting with ``"."`` ) are excluded.
|
|
|
|
Args:
|
|
vault_name: Name of the vault to browse.
|
|
path: Relative directory path within the vault (empty = root).
|
|
|
|
Returns:
|
|
``BrowseResponse`` with vault name, path, and item list.
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
return browse_directory(vault_name, path)
|
|
|
|
|
|
@router.get("/api/file/{vault_name}/raw", response_model=FileRawResponse)
|
|
async def api_file_raw(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
|
|
"""Return raw file content as plain text.
|
|
|
|
Args:
|
|
vault_name: Name of the vault.
|
|
path: Relative file path within the vault.
|
|
|
|
Returns:
|
|
``FileRawResponse`` with vault, path, and raw text content.
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
return read_raw_file(vault_name, path)
|
|
|
|
|
|
@router.get("/api/file/{vault_name}/download", response_class=FileResponse)
|
|
async def api_file_download(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
|
|
"""Download a file as an attachment.
|
|
|
|
Args:
|
|
vault_name: Name of the vault.
|
|
path: Relative file path within the vault.
|
|
|
|
Returns:
|
|
``FileResponse`` with ``application/octet-stream`` content-type.
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
vault_data = get_vault_data(vault_name)
|
|
if not vault_data:
|
|
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
|
|
|
vault_root = Path(vault_data["path"])
|
|
file_path = resolve_safe_path(vault_root, path)
|
|
|
|
if not file_path.exists() or not file_path.is_file():
|
|
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
|
|
|
# Record history
|
|
record_open(current_user.get("username"), vault_name, path)
|
|
|
|
return FileResponse(
|
|
path=str(file_path),
|
|
filename=file_path.name,
|
|
media_type="application/octet-stream",
|
|
)
|
|
|
|
|
|
@router.get("/api/file/{vault_name}/backlinks", response_model=BacklinksResponse)
|
|
async def api_file_backlinks(
|
|
vault_name: str,
|
|
path: str = Query(..., description="Relative path to file"),
|
|
current_user=Depends(require_auth),
|
|
):
|
|
"""Get backlinks (files linking to this file via wikilinks).
|
|
|
|
Returns a list of files that contain `[[wikilinks]]` pointing
|
|
to the requested file, across all accessible vaults.
|
|
|
|
Args:
|
|
vault_name: Name of the vault containing the target file.
|
|
path: Relative path of the target file within the vault.
|
|
|
|
Returns:
|
|
``{"vault": str, "path": str, "backlinks": [...]}``
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
vault_data = get_vault_data(vault_name)
|
|
if not vault_data:
|
|
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
|
|
|
user_vaults = current_user.get("_token_vaults") or current_user.get("vaults", [])
|
|
backlinks = get_backlinks(vault_name, path)
|
|
|
|
# Filter by user-accessible vaults
|
|
if "*" not in user_vaults:
|
|
backlinks = [b for b in backlinks if b["vault"] in user_vaults]
|
|
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"backlinks": backlinks,
|
|
"total": len(backlinks),
|
|
}
|
|
|
|
|
|
@router.get(
|
|
"/api/file/{vault_name}/xlsx/dashboard", response_model=XlsxDashboardResponse
|
|
)
|
|
def api_file_xlsx_dashboard(
|
|
vault_name: str,
|
|
path: str = Query(..., description="Relative path to the .xlsx workbook"),
|
|
current_user=Depends(require_auth),
|
|
):
|
|
"""Return the dashboard metadata of an .xlsx workbook (#153 A17).
|
|
|
|
Named ranges (workbook- or sheet-scoped), chart/pivot object counts and
|
|
per-sheet KPI stats (non-empty cells, rows/cols coverage, formulas,
|
|
numeric cells, first numeric values as KPI cards). Read-only, bounded by
|
|
the 500x40 render caps; never raises for an unreadable workbook — an
|
|
empty payload comes back and the viewer hides the panel.
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
_vault_root = get_vault_root(vault_name)
|
|
file_path = resolve_safe_path(_vault_root, path)
|
|
if not file_path.is_file():
|
|
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
|
if file_path.suffix.lower() not in (".xlsx", ".xlsm"):
|
|
raise HTTPException(
|
|
status_code=415, detail="Le fichier n'est pas un classeur .xlsx/.xlsm"
|
|
)
|
|
|
|
from backend.xlsx_reader import read_workbook_dashboard
|
|
|
|
try:
|
|
dashboard = read_workbook_dashboard(file_path)
|
|
except Exception as e:
|
|
logger.error(f"XLSX dashboard read error for {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
**dashboard,
|
|
}
|
|
|
|
|
|
@router.get(
|
|
"/api/file/{vault_name}/xlsx/sheet", response_model=XlsxSheetWindowResponse
|
|
)
|
|
def api_file_xlsx_sheet(
|
|
vault_name: str,
|
|
path: str = Query(..., description="Relative path to the .xlsx file"),
|
|
sheet: str = Query(..., description="Sheet name (as shown in the viewer tab)"),
|
|
offset: int = Query(0, ge=0, description="0-based index of the first row to return"),
|
|
limit: int = Query(
|
|
200, ge=1, le=1000, description="Rows to return (server-capped)"
|
|
),
|
|
current_user=Depends(require_auth),
|
|
):
|
|
"""Return a window of rows of one sheet of an .xlsx workbook (#153 A9).
|
|
|
|
Backs the viewer's lazy loading: instead of every sheet in a single JSON
|
|
payload, the client asks for the block it is about to display. The row
|
|
numbers and the ``data-cell`` references are the real A1 coordinates of the
|
|
sheet, so a window behaves like the full render (editing a cell in it
|
|
targets the right cell).
|
|
|
|
The response also carries ``total_rows``/``total_cols`` and the ``truncated``
|
|
flag, so the client can say what is hidden behind the 500x40 render caps
|
|
instead of silently hiding it.
|
|
|
|
Args:
|
|
vault_name: Name of the vault.
|
|
path: Relative path of the .xlsx file within the vault.
|
|
sheet: Sheet name; **404** if the workbook has no such sheet.
|
|
offset: 0-based index of the first row to return.
|
|
limit: Rows to return, capped server-side at 1000.
|
|
|
|
Returns:
|
|
``XlsxSheetWindowResponse`` with the rendered ``html`` of the window.
|
|
|
|
Raises:
|
|
HTTPException: 403 (vault access), 404 (vault, file or sheet unknown),
|
|
415 (not an .xlsx file), 500 (unreadable workbook).
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
vault_data = get_vault_data(vault_name)
|
|
if not vault_data:
|
|
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
|
|
|
file_path = resolve_safe_path(Path(vault_data["path"]), path)
|
|
if not file_path.is_file():
|
|
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
|
if file_path.suffix.lower() != ".xlsx":
|
|
raise HTTPException(status_code=415, detail="Le fichier n'est pas un classeur .xlsx")
|
|
|
|
# Import tardif : openpyxl n'est chargé que si un .xlsx est réellement demandé.
|
|
from backend.xlsx_reader import read_sheet_window
|
|
|
|
try:
|
|
window = read_sheet_window(file_path, sheet, offset=offset, limit=limit)
|
|
except Exception as e:
|
|
logger.error(f"XLSX sheet read error for {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
|
|
if window is None:
|
|
raise HTTPException(status_code=404, detail=f"Feuille introuvable: {sheet}")
|
|
|
|
return {"vault": vault_name, "path": path, **window}
|
|
|
|
|
|
@router.get("/api/file/{vault_name}", response_model=FileContentResponse)
|
|
async def api_file(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
|
|
"""Return rendered HTML and metadata for a file.
|
|
|
|
Markdown files are parsed for frontmatter, rendered with wikilink
|
|
support, and returned with extracted tags. Other supported file
|
|
types are syntax-highlighted as code blocks.
|
|
|
|
Args:
|
|
vault_name: Name of the vault.
|
|
path: Relative file path within the vault.
|
|
|
|
Returns:
|
|
``FileContentResponse`` with HTML, metadata, and tags.
|
|
"""
|
|
if not check_vault_access(vault_name, current_user):
|
|
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
|
vault_data = get_vault_data(vault_name)
|
|
if not vault_data:
|
|
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
|
|
|
vault_root = Path(vault_data["path"])
|
|
file_path = resolve_safe_path(vault_root, path)
|
|
|
|
if not file_path.exists() or not file_path.is_file():
|
|
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
|
|
|
# Record history
|
|
record_open(current_user.get("username"), vault_name, path, title=file_path.name)
|
|
|
|
ext = file_path.suffix.lower()
|
|
|
|
# === PDF: special handling before read_text (binary file) ===
|
|
if ext == ".pdf":
|
|
try:
|
|
from backend.pdf_reader import extract_pdf_metadata, extract_pdf_text, extract_pdf_toc
|
|
pdf_text = extract_pdf_text(file_path, max_chars=100000)
|
|
pdf_meta = extract_pdf_metadata(file_path)
|
|
pdf_toc = extract_pdf_toc(file_path)
|
|
size = file_path.stat().st_size
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": pdf_meta.get("title") or file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": f"<div class='pdf-viewer'><p>PDF — {pdf_meta.get('pages', '?')} pages</p><pre>{pdf_text[:5000]}</pre></div>",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_pdf": True,
|
|
"unsupported": False,
|
|
"pdf_metadata": pdf_meta,
|
|
"pdf_toc": pdf_toc,
|
|
"size_bytes": size,
|
|
}
|
|
except Exception as e:
|
|
logger.error(f"PDF read error for {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading PDF: {e!s}")
|
|
|
|
# === Excel .xlsx: render sheets as HTML tables (binary, before read_text) ===
|
|
if ext == ".xlsx":
|
|
try:
|
|
from backend.xlsx_reader import inspect_workbook, render_sheets
|
|
|
|
# #153 A15 — every sheet dict already carries its styles, aligns,
|
|
# merges and freeze anchor (read_workbook_meta, one normal-mode
|
|
# load inside render_sheets).
|
|
sheets = render_sheets(file_path)
|
|
size = file_path.stat().st_size
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": sheets[0]["html"] if sheets else "",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_xlsx": True,
|
|
"xlsx_sheets": sheets,
|
|
# #153 A1 — parts a save would drop; the viewer warns and asks
|
|
# for an explicit confirmation before forcing the write.
|
|
"xlsx_lossy_features": inspect_workbook(file_path),
|
|
"unsupported": False,
|
|
"size_bytes": size,
|
|
}
|
|
except Exception as e:
|
|
logger.error(f"XLSX read error for {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
|
|
|
|
# === Images: return as viewable image ===
|
|
if is_image(ext):
|
|
size = file_path.stat().st_size
|
|
mime = media_mime_type(str(file_path))
|
|
# #108-B1 — the raw endpoint returns JSON (FileRawResponse), so the
|
|
# standalone <img> must point to /api/image, which serves the bytes
|
|
# with the right MIME type. Paths are URL-encoded (accents, spaces).
|
|
img_url = f"/api/image/{quote(vault_name, safe='')}?path={quote(path, safe='')}"
|
|
html = (
|
|
f'<div class="image-viewer">'
|
|
f'<img src="{img_url}" '
|
|
f'alt="{html_mod.escape(file_path.name, quote=True)}" '
|
|
f'style="max-width:100%;max-height:80vh;object-fit:contain" />'
|
|
f'</div>'
|
|
)
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": html,
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_image": True,
|
|
"image_mime": mime,
|
|
"size_bytes": size,
|
|
}
|
|
|
|
# === Audio / Video: HTML5 players streamed from /api/media (roadmap #109) ===
|
|
if is_audio(ext) or is_video(ext):
|
|
size = file_path.stat().st_size
|
|
mime = media_mime_type(str(file_path))
|
|
media_kind = "audio" if is_audio(ext) else "video"
|
|
|
|
# #109-A3 — beyond the inline limit the viewer falls back to download
|
|
# (a single uvicorn worker must not be pinned by multi-GB media).
|
|
if size > media_max_inline_bytes():
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": "",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"unsupported": True,
|
|
"media_too_large": True,
|
|
"size_bytes": size,
|
|
}
|
|
|
|
# #109-A2 — byte-range endpoint: enables scrub and is required by Safari.
|
|
stream_url = f"/api/media/{quote(vault_name, safe='')}?path={quote(path, safe='')}"
|
|
if media_kind == "audio":
|
|
html = (
|
|
f'<div class="audio-viewer">'
|
|
f'<audio controls preload="metadata" src="{stream_url}"></audio>'
|
|
f'</div>'
|
|
)
|
|
else:
|
|
html = (
|
|
f'<div class="video-viewer">'
|
|
f'<video controls playsinline preload="metadata" src="{stream_url}"></video>'
|
|
f'</div>'
|
|
)
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": html,
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_audio": media_kind == "audio",
|
|
"is_video": media_kind == "video",
|
|
"media_mime": mime,
|
|
"stream_url": stream_url,
|
|
"size_bytes": size,
|
|
}
|
|
|
|
try:
|
|
raw = file_path.read_text(encoding="utf-8", errors="replace")
|
|
except PermissionError as e:
|
|
logger.error(f"Permission denied reading file {path}: {e}")
|
|
raise HTTPException(status_code=403, detail=f"Permission denied: cannot read file {path}")
|
|
except UnicodeDecodeError:
|
|
# Binary / unsupported file — return structured info with download option
|
|
size = file_path.stat().st_size
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": "",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"unsupported": True,
|
|
"size_bytes": size,
|
|
}
|
|
except Exception as e:
|
|
logger.error(f"Unexpected error reading file {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading file: {e!s}")
|
|
|
|
# === Excel .xlsm: same editable viewer as .xlsx, macros preserved on save ===
|
|
if ext == ".xlsm":
|
|
try:
|
|
from backend.xlsx_reader import inspect_workbook, render_sheets
|
|
|
|
sheets = render_sheets(file_path)
|
|
size = file_path.stat().st_size
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": sheets[0]["html"] if sheets else "",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_xlsx": True,
|
|
"xlsx_sheets": sheets,
|
|
# Macros are NOT lossy for .xlsm: keep_vba re-serializes them
|
|
# (an empty LOSSY probe is what makes the save gate pass).
|
|
"xlsx_lossy_features": [],
|
|
"unsupported": False,
|
|
"size_bytes": size,
|
|
}
|
|
except Exception as e:
|
|
logger.error(f"XLSX read error for {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
|
|
|
|
# === Legacy/ODF spreadsheets (.xls, .ods): read-only table view ===
|
|
if ext in (".xls", ".ods"):
|
|
try:
|
|
from backend.xlsx_reader import render_legacy_workbook
|
|
|
|
sheets = render_legacy_workbook(file_path, ext)
|
|
size = file_path.stat().st_size
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": sheets[0]["html"] if sheets else "",
|
|
"raw_length": size,
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_xlsx": True,
|
|
"xlsx_readonly": True,
|
|
"xlsx_sheets": sheets,
|
|
"unsupported": False,
|
|
"size_bytes": size,
|
|
}
|
|
except Exception as e:
|
|
logger.error(f"Spreadsheet read error for {path}: {e}")
|
|
raise HTTPException(status_code=500, detail=f"Error reading spreadsheet: {e!s}")
|
|
|
|
# === CSV: spreadsheet-style table (same shape as the xlsx viewer) ===
|
|
if ext == ".csv":
|
|
from backend.xlsx_reader import render_csv_table
|
|
|
|
html = render_csv_table(raw)
|
|
return {
|
|
"vault": vault_name, "path": path,
|
|
"title": file_path.name, "tags": [], "frontmatter": {},
|
|
"html": html, "raw_length": len(raw), "extension": ext,
|
|
"is_markdown": False, "is_csv": True,
|
|
}
|
|
|
|
# === JSON: syntax-highlighted display ===
|
|
if ext == ".json":
|
|
import json as json_mod
|
|
try:
|
|
parsed = json_mod.loads(raw)
|
|
formatted = json_mod.dumps(parsed, indent=2, ensure_ascii=False)
|
|
except json_mod.JSONDecodeError:
|
|
formatted = raw
|
|
html = f"<pre class='json-viewer'><code>{html_mod.escape(formatted)}</code></pre>"
|
|
return {
|
|
"vault": vault_name, "path": path,
|
|
"title": file_path.name, "tags": [], "frontmatter": {},
|
|
"html": html, "raw_length": len(raw), "extension": ext,
|
|
"is_markdown": False, "is_json": True,
|
|
}
|
|
|
|
# === Excalidraw .excalidraw.md (Obsidian plugin format) ===
|
|
if path.lower().endswith(".excalidraw.md"):
|
|
import re as re_mod
|
|
raw_lower = file_path.read_text(encoding="utf-8", errors="replace")
|
|
# Check for excalidraw-plugin in frontmatter or body
|
|
if "excalidraw-plugin:" in raw_lower:
|
|
# Extract compressed JSON block
|
|
match = re_mod.search(r'```compressed-json\n(.*?)\n```', raw_lower, re_mod.DOTALL)
|
|
if match:
|
|
compressed = match.group(1).strip()
|
|
return {
|
|
"vault": vault_name, "path": path,
|
|
"title": file_path.name.replace(".excalidraw.md", ""),
|
|
"tags": [], "frontmatter": {},
|
|
"html": "", "raw_length": len(raw_lower),
|
|
"extension": ".excalidraw.md",
|
|
"is_markdown": False,
|
|
"is_excalidraw": True,
|
|
"excalidraw_data_compressed": compressed,
|
|
}
|
|
# Fallback: treat as regular markdown
|
|
raw = raw_lower
|
|
if ext == ".excalidraw":
|
|
import json as json_mod
|
|
try:
|
|
parsed = json_mod.loads(raw)
|
|
except json_mod.JSONDecodeError:
|
|
parsed = None
|
|
if parsed and parsed.get("type") == "excalidraw":
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": parsed.get("appState", {}).get("name") or file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": "",
|
|
"raw_length": len(raw),
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
"is_excalidraw": True,
|
|
"excalidraw_data": {
|
|
"elements": parsed.get("elements", []),
|
|
"appState": parsed.get("appState", {}),
|
|
"files": parsed.get("files", {}),
|
|
},
|
|
}
|
|
else:
|
|
# Not a valid Excalidraw file — fall through to text viewer
|
|
pass
|
|
|
|
# === Plain text / other readable files ===
|
|
TEXT_EXTENSIONS = {".txt", ".log", ".yml", ".yaml", ".toml", ".ini", ".cfg",
|
|
".sh", ".bash", ".py", ".js", ".ts", ".html", ".css",
|
|
".xml", ".rst", ".tex", ".sql", ".conf", ".env"}
|
|
if ext in TEXT_EXTENSIONS or ext == ".md":
|
|
pass # handled below or by markdown section
|
|
|
|
if ext == ".md":
|
|
post = parse_markdown_file(raw)
|
|
|
|
# Extract metadata using shared indexer logic
|
|
tags = _extract_tags(post)
|
|
|
|
title = post.metadata.get("title", file_path.stem.replace("-", " ").replace("_", " "))
|
|
html_content = _render_markdown(post.content, vault_name, file_path)
|
|
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": str(title),
|
|
"tags": tags,
|
|
"frontmatter": dict(post.metadata) if post.metadata else {},
|
|
"html": html_content,
|
|
"raw_length": len(raw),
|
|
"extension": ext,
|
|
"is_markdown": True,
|
|
}
|
|
else:
|
|
# Non-markdown: wrap in syntax-highlighted code block
|
|
lang = EXT_TO_LANG.get(ext, "")
|
|
if not lang:
|
|
# Fichiers sans extension usuels (Dockerfile, Makefile, etc.)
|
|
NAME_TO_LANG = {
|
|
"dockerfile": "dockerfile", "makefile": "makefile",
|
|
"cmakelists.txt": "cmake", "jenkinsfile": "groovy",
|
|
"vagrantfile": "ruby", "rakefile": "ruby", "gemfile": "ruby",
|
|
"procfile": "plaintext", "bashrc": "bash", "bash_profile": "bash",
|
|
"zshrc": "bash", "profile": "bash", "gitignore": "plaintext",
|
|
}
|
|
lang = NAME_TO_LANG.get(file_path.name.lower(), "plaintext")
|
|
escaped = html_mod.escape(raw)
|
|
html_content = f'<pre><code class="language-{lang}">{escaped}</code></pre>'
|
|
|
|
return {
|
|
"vault": vault_name,
|
|
"path": path,
|
|
"title": file_path.name,
|
|
"tags": [],
|
|
"frontmatter": {},
|
|
"html": html_content,
|
|
"raw_length": len(raw),
|
|
"extension": ext,
|
|
"is_markdown": False,
|
|
}
|