feat: garde-fous d'écriture des classeurs Excel #153 (P0)

L'édition d'un .xlsx pouvait détruire une partie du classeur, le
concurrencer en silence, ou diffuser une injection de formule.

- BUG-085 : inspect_workbook() détecte ce qu'un round-trip openpyxl perd
  (valeurs calculées en cache, slicers, contrôles, connexions, custom
  XML, signature, commentaires enrichis, macros) → xlsx_lossy_features
  exposé en lecture, bandeau FR/EN, et 409 xlsx_lossy_content sans
  `force` (confirmation explicite puis reprise). Périmètre réel
  revalidé : graphiques, images et TCD survivent au round-trip.
- BUG-086 : écriture atomique (fichier .tmp + os.replace) : un plantage
  ne peut plus tronquer le classeur, le backup reste intact.
- BUG-087 : verrou par fichier autour du read-modify-write (timeout 15 s,
  409 conflict) ; endpoint xlsx/save devenu synchrone pour que
  l'attente s'exécute dans le threadpool.
- BUG-088 : une saisie en '=' ou '@' est stockée en texte, sauf opt-in
  `allow_formula` ou le bouton f(x) de la visionneuse. Le handler
  ServiceError expose désormais code + details, que api() propage.
- BUG-084 : la suppression d'une vault purge enfin l'index inversé
  (documents fantômes qui continuaient de matcher) et is_stale() devient
  is_ready(), le nom étant trompeur (la staleness n'existe plus).

Tests : 1390 pytest, 10 JSDOM (xlsx-viewer.test.mjs, branché au CI),
3 E2E Playwright, suite E2E complète verte, ruff/mypy 0.

🤖 Generated with Codebuff
Co-Authored-By: Codebuff <[email protected]>
This commit is contained in:
2026-09-27 20:39:12 -04:00
parent 4c4b1222d5
commit 31d4616baf
40 changed files with 1628 additions and 87 deletions
+6
View File
@@ -1226,6 +1226,12 @@ async def remove_vault_from_index(vault_name: str):
if not _file_lookup[key]:
_file_lookup.pop(key, None)
# Notify the inverted index, otherwise every document of the vault
# stays in it as a ghost (postings, doc_info, doc_vault, vault_docs)
# and keeps matching searches for a vault that no longer exists.
if _on_index_change:
_on_index_change('remove', vault_name, rel_path, f) # type: ignore[misc]
# Clean path_index
path_index.pop(vault_name, None)
+14 -2
View File
@@ -389,8 +389,20 @@ app.openapi = _custom_openapi # type: ignore[method-assign]
@app.exception_handler(ServiceError)
async def _service_error_handler(request: Request, exc: ServiceError):
"""Map shared-layer domain errors to HTTP responses (``{"detail": ...}``)."""
return JSONResponse(status_code=exc.status, content={"detail": exc.message})
"""Map shared-layer domain errors to HTTP responses (``{"detail": ...}``).
``code`` and ``details`` travel with the message so the client can react to
a specific case instead of parsing prose (#153 A1 : ``xlsx_lossy_content``
asks the viewer to confirm before forcing a lossy write).
"""
return JSONResponse(
status_code=exc.status,
content={
"detail": exc.message,
"code": exc.code,
"details": exc.details,
},
)
# GZip compression — reduces bandwidth by ~70% for text responses
# Custom wrapper: skip compression for SSE streams (/api/events)
+1 -1
View File
@@ -182,7 +182,7 @@ _ENDPOINT_EXAMPLES: dict[tuple[str, str], dict[str, Any]] = {
"response": {"status": "ok", "vault": "TestVault", "path": "notes/Accueil.md", "size": 26},
},
("put", "/api/file/{vault_name}/xlsx/save"): {
"request": {"sheet": "Budget", "cells": {"B1": "250"}},
"request": {"sheet": "Budget", "cells": {"B1": "250"}, "allow_formula": False, "force": False},
"response": {"status": "ok", "vault": "TestVault", "path": "data/budget.xlsx", "size": 1},
},
("post", "/api/search/replace"): {
+1 -1
View File
@@ -481,7 +481,7 @@ async def api_diagnostics(current_user=Depends(require_admin)):
"total_postings": word_index_entries,
"documents": inv.doc_count,
"sorted_tokens": len(inv._sorted_tokens),
"is_stale": inv.is_stale(),
"is_ready": inv.is_ready(),
"memory_estimate_mb": mem_estimate_mb,
},
"config": _load_config(),
+4 -1
View File
@@ -241,7 +241,7 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
# === Excel .xlsx: render sheets as HTML tables (binary, before read_text) ===
if ext == ".xlsx":
try:
from backend.xlsx_reader import render_sheets
from backend.xlsx_reader import inspect_workbook, render_sheets
sheets = render_sheets(file_path)
size = file_path.stat().st_size
@@ -257,6 +257,9 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
"is_markdown": False,
"is_xlsx": True,
"xlsx_sheets": sheets,
# #153 A1 — parts a save would drop; the viewer warns and asks
# for an explicit confirmation before forcing the write.
"xlsx_lossy_features": inspect_workbook(file_path),
"unsupported": False,
"size_bytes": size,
}
+31 -5
View File
@@ -114,17 +114,35 @@ async def api_file_save(
@router.put("/api/file/{vault_name}/xlsx/save", response_model=FileSaveResponse)
async def api_file_xlsx_save(
def api_file_xlsx_save(
vault_name: str,
path: str = Query(..., description="Relative path to the .xlsx file"),
body: dict = Body(..., description='{"sheet": str, "cells": {"A1": value}}'),
body: dict = Body(
...,
description=(
'{"sheet": str, "cells": {"A1": value}, '
'"allow_formula": false, "force": false}'
),
),
current_user=Depends(require_auth),
):
"""Apply cell edits to an .xlsx workbook.
Expects a JSON body with ``sheet`` and ``cells`` (A1 references to new
scalar values, max 500 per request). A backup is created before the
workbook is rewritten.
scalar values, max 500 per request) plus two optional boolean flags:
* ``allow_formula`` — keep values starting with ``=``/``@`` as real
formulas. Off by default (#153 A4): such a value is stored as text so a
later Excel session cannot execute it (DDE).
* ``force`` — write a workbook carrying features openpyxl cannot re-serialize
(slicers, form controls, connections, custom XML, signature, cached formula
results). Without it the call fails **409** ``xlsx_lossy_content`` and the
client asks the user to confirm (#153 A1).
A backup is created before the workbook is rewritten, and the new archive
swaps in atomically. Declared as a sync endpoint on purpose: the openpyxl
round-trip and the per-file lock wait (#153 A3) then run in the threadpool
instead of blocking the event loop.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
@@ -138,8 +156,16 @@ async def api_file_xlsx_save(
for ref, value in cells.items():
if not isinstance(ref, str) or not isinstance(value, (str, int, float, bool, type(None))):
raise HTTPException(status_code=400, detail=f"Cellule invalide: {ref!r}")
flags: dict[str, bool] = {}
for name in ("allow_formula", "force"):
raw = body.get(name, False)
if not isinstance(raw, bool):
raise HTTPException(status_code=400, detail=f"Flag invalide: {name}")
flags[name] = raw
result = service_edit_xlsx_cells(vault_name, path, sheet, cells)
result = service_edit_xlsx_cells(
vault_name, path, sheet, cells, **flags
)
log_file_save(
current_user["username"], vault_name, path,
sum(len(str(v)) for v in cells.values()),
+8
View File
@@ -288,6 +288,14 @@ class FileContentResponse(BaseModel):
xlsx_sheets: list[dict[str, Any]] | None = Field(
default=None, description="Rendered xlsx sheets [{name, html}]"
)
xlsx_lossy_features: list[str] | None = Field(
default=None,
description=(
"Workbook parts an openpyxl save would drop (#153 A1) — e.g. "
"cached_values, slicers, form_controls, connections, custom_xml, "
"signature, rich_comments, macros. Empty/absent = nothing at risk."
),
)
is_json: bool | None = Field(default=None, description="True for JSON files")
is_excalidraw: bool | None = Field(default=None, description="True for Excalidraw diagram files")
excalidraw_data: dict[str, Any] | None = Field(default=None, description="Excalidraw diagram data (elements, appState, files)")
+14 -4
View File
@@ -371,9 +371,15 @@ class InvertedIndex:
self._sorted_tokens: SortedList = SortedList()
self._ready: bool = False # True after initial build
def is_stale(self) -> bool:
"""Return True if the index has not been built yet."""
return not self._ready
def is_ready(self) -> bool:
"""Return True once the initial build has completed.
The index is then kept current incrementally by ``add_document()`` /
``remove_document()``, so it never goes stale: there is no generation
counter, no cooldown and no lazy rebuild. Searches simply fall back to
a full scan while this is False (see ``search()``).
"""
return self._ready
def rebuild(self) -> None:
"""Rebuild inverted index from the global ``index`` dict.
@@ -537,6 +543,10 @@ class InvertedIndex:
self.doc_vault.pop(doc_key, None)
if vault_name in self.vault_docs:
self.vault_docs[vault_name].discard(doc_key)
# Drop the empty entry so a fully removed vault leaves no trace
# (it is a defaultdict: a bare lookup would recreate the key).
if not self.vault_docs[vault_name]:
del self.vault_docs[vault_name]
# Tags (per-document, NOT the global tag_norm_map)
for tag in file_info.get("tags", []):
td = self.tag_docs.get(tag.lower())
@@ -739,7 +749,7 @@ def search(
results: list[dict[str, Any]] = []
inv = get_inverted_index()
use_index = (not inv.is_stale()) and inv.doc_count > 0
use_index = inv.is_ready() and inv.doc_count > 0
if use_index:
# BUG-033: retrieve candidates from the inverted index instead of
-4
View File
@@ -457,10 +457,6 @@ class SemanticIndex:
"""Return True once a full rebuild has completed."""
return self._ready
def is_stale(self) -> bool:
"""Alias used by callers that check index freshness."""
return not self._ready
def _ensure_provider(self) -> EmbeddingProvider:
if self.provider is None:
self.provider = get_embedding_provider()
+113 -27
View File
@@ -16,7 +16,9 @@ import logging
import os
import re
import shutil
from collections.abc import Callable
import threading
from collections.abc import Callable, Iterator
from contextlib import contextmanager
from pathlib import Path
from typing import Any
@@ -230,6 +232,41 @@ _XLSX_CELL_RE = re.compile(r"^[A-Z]{1,3}[1-9][0-9]{0,7}$")
# number; dates/booleans stay text (upgrade path: parse locale dates too).
_XLSX_INT_RE = re.compile(r"^[+-]?\d+$")
_XLSX_FLOAT_RE = re.compile(r"^[+-]?(?:\d+\.\d*|\.\d+)$")
# #153 A4 — openpyxl turns any string starting with "=" into a formula, which
# Excel then evaluates on open (DDE / =cmd|… / =HYPERLINK exfiltration). "@" is
# the legacy Lotus-style trigger. "+"/"-" are left alone: they are numbers here.
_XLSX_FORMULA_RE = re.compile(r"^[=@]")
# #153 A3 — per-file write lock. Two concurrent saves (two tabs, the AI agent
# and the viewer, a watcher restore) would otherwise read-modify-write on the
# same archive and the last writer silently wins. Kept deliberately small: the
# lock only covers the load → edit → atomic-replace window.
_XLSX_LOCK_TIMEOUT = 15.0
_xlsx_locks: dict[str, threading.Lock] = {}
_xlsx_locks_guard = threading.Lock()
@contextmanager
def _xlsx_write_lock(key: str) -> Iterator[None]:
"""Serialize the read-modify-write of one workbook path.
Raises:
ServiceError: ``conflict`` (409) when the lock is still held after
:data:`_XLSX_LOCK_TIMEOUT` seconds.
"""
with _xlsx_locks_guard:
lock = _xlsx_locks.setdefault(key, threading.Lock())
if not lock.acquire(timeout=_XLSX_LOCK_TIMEOUT):
raise ServiceError(
"Workbook is being modified by another operation, retry shortly",
code="conflict",
status=409,
details={"path": key, "timeout_seconds": _XLSX_LOCK_TIMEOUT},
)
try:
yield
finally:
lock.release()
def _coerce_xlsx_value(value: Any) -> Any:
@@ -246,6 +283,19 @@ def _coerce_xlsx_value(value: Any) -> Any:
return value
def _write_cell(ws: Any, ref: str, value: Any, *, allow_formula: bool) -> None:
"""Assign one cell, forcing text when it looks like a formula.
``cell.data_type = "s"`` is what stops openpyxl from emitting ``<f>``: the
text is then stored as an inline/shared string and Excel shows it verbatim.
"""
cell = ws[ref]
coerced = _coerce_xlsx_value(value)
cell.value = coerced
if not allow_formula and isinstance(coerced, str) and _XLSX_FORMULA_RE.match(coerced):
cell.data_type = "s"
def edit_xlsx_cells(
vault_name: str,
path: str,
@@ -253,15 +303,29 @@ def edit_xlsx_cells(
cells: dict[str, Any],
*,
backup: bool = True,
allow_formula: bool = False,
force: bool = False,
) -> dict[str, Any]:
"""Apply a batch of cell edits to an ``.xlsx`` workbook.
Raises:
ServiceError: ``not_found`` (404), ``read_only`` (403) or
``invalid`` (400) for a bad sheet, cell reference or value.
Args:
vault_name: Name of the vault the workbook belongs to.
path: Vault-relative path of the ``.xlsx`` file.
sheet: Worksheet title to edit.
cells: Mapping of A1 references to new scalar values.
backup: Create a timestamped ``.bak`` before rewriting the archive.
allow_formula: Keep values starting with ``=``/``@`` as real formulas.
Off by default (#153 A4): a typed ``=cmd|…`` is a DDE payload when
the file is later opened in Excel.
force: Write even when the workbook carries features openpyxl drops
(slicers, form controls, connections, custom XML, signature, cached
formula results — see :data:`backend.xlsx_reader.LOSSY_PARTS`).
ponytail: openpyxl round-trips values/formulas/styles but drops charts,
images and pivot tables; use the SheetJS path if a workbook needs those.
Raises:
ServiceError: ``not_found`` (404), ``read_only`` (403), ``conflict``
(409, concurrent write), ``xlsx_lossy_content`` (409, a lossy write was
attempted without ``force``) or ``invalid`` (400) for a bad sheet, cell
reference or value.
"""
root = get_vault_root(vault_name)
_ensure_writable(root)
@@ -286,30 +350,52 @@ def edit_xlsx_cells(
f"Invalid cell reference: {ref!r}", code="invalid", status=400
)
from openpyxl import load_workbook
if not force:
from backend.xlsx_reader import inspect_workbook
try:
wb = load_workbook(file_path)
except Exception as exc:
raise ServiceError(
f"Cannot open workbook: {exc}", code="invalid", status=400
) from exc
if sheet not in wb.sheetnames:
raise ServiceError(
f"Unknown sheet: {sheet}",
code="invalid",
status=400,
details={"sheets": wb.sheetnames},
)
lossy = inspect_workbook(file_path)
if lossy:
raise ServiceError(
"Saving this workbook would drop features ObsiGate cannot "
"preserve; retry with force=true after confirmation",
code="xlsx_lossy_content",
status=409,
details={"path": path, "features": lossy},
)
rel_path = _rel(root, file_path)
if backup:
create_backup(file_path, vault_name, rel_path)
with _xlsx_write_lock(str(file_path)):
from openpyxl import load_workbook
ws = wb[sheet]
for ref, value in cells.items():
ws[ref].value = _coerce_xlsx_value(value)
wb.save(file_path)
try:
wb = load_workbook(file_path)
except Exception as exc:
raise ServiceError(
f"Cannot open workbook: {exc}", code="invalid", status=400
) from exc
if sheet not in wb.sheetnames:
raise ServiceError(
f"Unknown sheet: {sheet}",
code="invalid",
status=400,
details={"sheets": wb.sheetnames},
)
rel_path = _rel(root, file_path)
if backup:
create_backup(file_path, vault_name, rel_path)
ws = wb[sheet]
for ref, value in cells.items():
_write_cell(ws, ref, value, allow_formula=allow_formula)
# #153 A2 — write beside the target then swap: a crash mid-save leaves
# the original workbook intact instead of a truncated archive.
tmp_path = file_path.with_name(f"{file_path.name}.{os.getpid()}.tmp")
try:
wb.save(tmp_path)
os.replace(tmp_path, file_path)
except Exception:
tmp_path.unlink(missing_ok=True)
raise
logger.info(f"XLSX cells saved: {vault_name}/{rel_path} [{sheet}] +{len(cells)}")
return {
+73
View File
@@ -3,11 +3,16 @@
Read-only: formulas are shown as their text (``data_only=False``) so a
round-trip through the viewer never depends on Excel's cached values.
Write-side lives in ``backend.services.mutations.edit_xlsx_cells``.
:func:`inspect_workbook` lists the workbook features that an openpyxl
round-trip would drop (#153 A1) so the UI can warn before saving.
"""
from __future__ import annotations
import html
import re
import zipfile
from datetime import date, datetime
from pathlib import Path
from typing import Any
@@ -20,6 +25,29 @@ from openpyxl.utils import get_column_letter
MAX_ROWS = 500
MAX_COLS = 40
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
# DO survive the round-trip, so they are deliberately absent from this map.
LOSSY_PARTS: dict[str, tuple[str, ...]] = {
"slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"),
"form_controls": ("xl/ctrlProps/", "xl/activeX/"),
"connections": ("xl/queryTables/", "xl/connections.xml"),
"custom_xml": ("customXml/",),
"signature": ("_xmlsignatures/",),
"rich_comments": ("xl/threadedComments/", "xl/persons/"),
"macros": ("xl/vbaProject.bin",),
}
# A formula cell carrying its last computed result: ``<f>…</f><v>…</v>``.
# openpyxl writes an EMPTY ``<v></v>`` itself, hence the ``[^<]`` guard: only a
# non-empty value counts. openpyxl keeps the formula but drops the cached result,
# so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until
# Excel recalculates.
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
# Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES).
_MAX_PROBE_BYTES = 8_000_000
def _fmt(value: Any) -> str:
if value is None:
@@ -68,6 +96,51 @@ def _table(grid: list[list[str]]) -> str:
return "".join(out)
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
"""True when at least one formula cell still carries its computed value."""
budget = _MAX_PROBE_BYTES
for name in zf.namelist():
if not name.startswith("xl/worksheets/sheet") or not name.endswith(".xml"):
continue
try:
with zf.open(name) as fh:
while budget > 0:
chunk = fh.read(65536)
if not chunk:
break
budget -= len(chunk)
if _CACHED_FORMULA_RE.search(chunk):
return True
except (KeyError, OSError, zipfile.BadZipFile):
continue
return False
def inspect_workbook(file_path: Path) -> list[str]:
"""Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*.
Read-only inspection of the OPC package (central directory + a bounded scan
of the sheet XML). Never raises: an unreadable or encrypted workbook simply
yields ``[]`` and the save path keeps its current behaviour.
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
the cached result, so the workbook stays correct once Excel recalculates it.
"""
try:
with zipfile.ZipFile(file_path) as zf:
names = set(zf.namelist())
found = {
key
for key, prefixes in LOSSY_PARTS.items()
if any(name.startswith(prefix) for name in names for prefix in prefixes)
}
if _has_cached_formulas(zf):
found.add("cached_values")
return sorted(found)
except (OSError, zipfile.BadZipFile):
return []
def render_sheets(file_path: Path) -> list[dict[str, str]]:
"""Return ``[{"name": sheet_title, "html": table_html}, ...]``."""
wb = load_workbook(str(file_path), read_only=True, data_only=False)