feat: completude de l'editeur Excel - mise en forme, sortie, concurrence, cache, outils IA #156
CI / lint (push) Successful in 2m36s
CI / security (push) Successful in 1m49s
CI / test (push) Successful in 4m33s
CI / build (push) Successful in 2m39s
CI / e2e (push) Successful in 15m45s

L'editeur lisait les styles mais ne les ecrivait pas, exportait la feuille
entiere et laissait le dernier-ecrivain gagner entre processus.

- A8 : bouton Mise en forme (gras/italique/souligne, alignements, couleurs via
  selecteur natif, formats de nombre, fusion, volets figes, largeur/hauteur) et
  nouvelle route PUT .../xlsx/style (verrou, backup, swap atomique, garde de
  perte, If-Match) ; A9 : decision "pas de moteur de formule" annoncee dans
  l'UI ; A10 : undo/redo unifie, piles par fichier conservees au re-rendu
- A11 : export de la selection + Markdown/HTML/impression et recherche sur
  toutes les feuilles ; A12 : concurrence optimiste (If-Match -> 409 reparable,
  retry qui relit) ; A13 : cache des metadonnees par (chemin, mtime, taille)
- A14 : outils IA .xlsm/.csv + search_workbook, analyze_range,
  edit_xlsx_structure
- tests : test_xlsx_styles.py (17), test_spreadsheet_tools.py (34),
  TestOptimisticConcurrency/TestMetaCache, JSDOM xlsx-viewer 107/107
This commit is contained in:
2026-09-30 07:04:47 -04:00
parent 1bacfd69d9
commit c5c225a68e
32 changed files with 4615 additions and 302 deletions
+11
View File
@@ -185,6 +185,17 @@ _ENDPOINT_EXAMPLES: dict[tuple[str, str], dict[str, Any]] = {
"request": {"sheet": "Budget", "cells": {"B1": "250"}, "allow_formula": False, "force": False},
"response": {"status": "ok", "vault": "TestVault", "path": "data/budget.xlsx", "size": 1},
},
("put", "/api/file/{vault_name}/xlsx/style"): {
"request": {
"ops": [
{"op": "cell", "sheet": "Budget", "range": "A1:B1", "style": {"bold": True, "fill_color": "#ffe08a"}},
{"op": "col_width", "sheet": "Budget", "col": "A", "width": 24},
],
"force": False,
"if_match": "18f2c0ab-1f4",
},
"response": {"status": "ok", "vault": "TestVault", "path": "data/budget.xlsx", "size": 2, "revision": "18f2c0ab-1f6"},
},
# GET : pas d'exemple de requête (un requestBody sur un GET serait un OpenAPI
# invalide) — les paramètres sont documentés par leurs Query().
("get", "/api/file/{vault_name}/xlsx/sheet"): {
+17 -5
View File
@@ -42,6 +42,7 @@ from backend.schemas import (
XlsxSheetWindowResponse,
)
from backend.services.files import read_raw_file
from backend.services.mutations import file_revision
from backend.services.paths import resolve_safe_path
from backend.services.vaults import browse_directory, get_vault_root
@@ -226,7 +227,7 @@ def api_file_xlsx_dashboard(
)
def api_file_xlsx_sheet(
vault_name: str,
path: str = Query(..., description="Relative path to the .xlsx file"),
path: str = Query(..., description="Relative path to the .xlsx/.xlsm file"),
sheet: str = Query(..., description="Sheet name (as shown in the viewer tab)"),
offset: int = Query(0, ge=0, description="0-based index of the first row to return"),
limit: int = Query(
@@ -234,7 +235,7 @@ def api_file_xlsx_sheet(
),
current_user=Depends(require_auth),
):
"""Return a window of rows of one sheet of an .xlsx workbook (#153 A9).
"""Return a window of rows of one sheet of an .xlsx/.xlsm workbook (#153 A9).
Backs the viewer's lazy loading: instead of every sheet in a single JSON
payload, the client asks for the block it is about to display. The row
@@ -258,7 +259,7 @@ def api_file_xlsx_sheet(
Raises:
HTTPException: 403 (vault access), 404 (vault, file or sheet unknown),
415 (not an .xlsx file), 500 (unreadable workbook).
415 (not an .xlsx/.xlsm file), 500 (unreadable workbook).
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
@@ -269,8 +270,12 @@ def api_file_xlsx_sheet(
file_path = resolve_safe_path(Path(vault_data["path"]), path)
if not file_path.is_file():
raise HTTPException(status_code=404, detail=f"File not found: {path}")
if file_path.suffix.lower() != ".xlsx":
raise HTTPException(status_code=415, detail="Le fichier n'est pas un classeur .xlsx")
# BUG-097 — a .xlsm rides the same editable viewer (and its lazy loading),
# so its row windows must be servable too; other formats stay refused.
if file_path.suffix.lower() not in (".xlsx", ".xlsm"):
raise HTTPException(
status_code=415, detail="Le fichier n'est pas un classeur .xlsx/.xlsm"
)
# Import tardif : openpyxl n'est chargé que si un .xlsx est réellement demandé.
from backend.xlsx_reader import read_sheet_window
@@ -368,6 +373,10 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
"is_markdown": False,
"is_xlsx": True,
"xlsx_sheets": sheets,
# #156-A12 — optimistic-concurrency token: the viewer sends it
# back as `if_match` so another writer cannot be overwritten in
# silence (409 `conflict` instead).
"xlsx_revision": file_revision(file_path),
# #153 A1 — parts a save would drop; the viewer warns and asks
# for an explicit confirmation before forcing the write.
"xlsx_lossy_features": inspect_workbook(file_path),
@@ -507,6 +516,7 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
"is_markdown": False,
"is_xlsx": True,
"xlsx_sheets": sheets,
"xlsx_revision": file_revision(file_path),
# Macros are NOT lossy for .xlsm: keep_vba re-serializes them
# (an empty LOSSY probe is what makes the save gate pass).
"xlsx_lossy_features": [],
@@ -554,6 +564,8 @@ async def api_file(vault_name: str, path: str = Query(..., description="Relative
"title": file_path.name, "tags": [], "frontmatter": {},
"html": html, "raw_length": len(raw), "extension": ext,
"is_markdown": False, "is_csv": True,
# #156-A12 — same stale-write guard as the workbooks.
"xlsx_revision": file_revision(file_path),
}
# === JSON: syntax-highlighted display ===
+109 -10
View File
@@ -67,6 +67,9 @@ from backend.services.mutations import (
from backend.services.mutations import (
mutate_xlsx_structure as service_mutate_xlsx_structure,
)
from backend.services.mutations import (
mutate_xlsx_style as service_mutate_xlsx_style,
)
from backend.services.mutations import (
rename_directory as service_rename_directory,
)
@@ -127,7 +130,7 @@ def api_file_xlsx_save(
...,
description=(
'{"sheet": str, "cells": {"A1": value}, '
'"allow_formula": false, "force": false}'
'"allow_formula": false, "force": false, "if_match": str}'
),
),
current_user=Depends(require_auth),
@@ -144,6 +147,10 @@ def api_file_xlsx_save(
(slicers, form controls, connections, custom XML, signature, cached formula
results). Without it the call fails **409** ``xlsx_lossy_content`` and the
client asks the user to confirm (#153 A1).
* ``if_match`` — revision token returned by the read (#156-A12). When it no
longer matches the file on disk the write is refused with **409**
``conflict`` (``reason=stale_revision``) instead of overwriting a change
made by another writer. Omitted: last writer wins (curl, AI tools).
A backup is created before the workbook is rewritten, and the new archive
swaps in atomically. Declared as a sync endpoint on purpose: the openpyxl
@@ -168,16 +175,22 @@ def api_file_xlsx_save(
if not isinstance(raw, bool):
raise HTTPException(status_code=400, detail=f"Flag invalide: {name}")
flags[name] = raw
if_match = body.get("if_match")
if if_match is not None and not isinstance(if_match, str):
raise HTTPException(status_code=400, detail="Jeton invalide: if_match")
result = service_edit_xlsx_cells(
vault_name, path, sheet, cells, **flags
vault_name, path, sheet, cells, expected_revision=if_match, **flags
)
log_file_save(
current_user["username"], vault_name, path,
sum(len(str(v)) for v in cells.values()),
current_user.get("_request_ip", "unknown"),
)
return {"status": "ok", "vault": result["vault"], "path": result["path"], "size": result["size"]}
return {
"status": "ok", "vault": result["vault"], "path": result["path"],
"size": result["size"], "revision": result.get("revision"),
}
@router.put("/api/file/{vault_name}/csv/save", response_model=FileSaveResponse)
@@ -186,7 +199,11 @@ def api_file_csv_save(
path: str = Query(..., description="Relative path to the .csv file"),
body: dict = Body(
...,
description='{"cells": {"A1": value}} — A1-addressed text edits (#153 A16)',
description=(
'{"cells": {"A1": value}, "if_match": str} — A1-addressed text '
'edits (#153 A16). `if_match` is the revision token of the read '
'(#156-A12): a stale token fails with 409 `conflict`.'
),
),
current_user=Depends(require_auth),
):
@@ -195,6 +212,9 @@ def api_file_csv_save(
The grid is re-parsed with :mod:`csv`, patched and re-serialized
(RFC 4180 quoting). References beyond the extent grow the grid. Values
are stored verbatim as text — a CSV has no formula engine.
``if_match`` (optional, #156-A12) is the revision the client read: when the
file changed in the meantime the write is refused with **409** ``conflict``.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
@@ -206,13 +226,20 @@ def api_file_csv_save(
if not isinstance(ref, str) or not isinstance(value, (str, int, float, bool, type(None))):
raise HTTPException(status_code=400, detail=f"Cellule invalide: {ref!r}")
result = service_save_csv_cells(vault_name, path, cells)
if_match = body.get("if_match")
if if_match is not None and not isinstance(if_match, str):
raise HTTPException(status_code=400, detail="Jeton invalide: if_match")
result = service_save_csv_cells(vault_name, path, cells, expected_revision=if_match)
log_file_save(
current_user["username"], vault_name, path,
sum(len(str(v)) for v in cells.values()),
current_user.get("_request_ip", "unknown"),
)
return {"status": "ok", "vault": result["vault"], "path": result["path"], "size": result["size"]}
return {
"status": "ok", "vault": result["vault"], "path": result["path"],
"size": result["size"], "revision": result.get("revision"),
}
@router.put("/api/file/{vault_name}/xlsx/structure", response_model=FileSaveResponse)
@@ -224,7 +251,7 @@ def api_file_xlsx_structure(
description=(
'{"actions": [{"op": "sheet_add", "name": "X"}, '
'{"op": "row_insert", "sheet": "X", "at": 2, "count": 1}], '
'"force": false}'
'"force": false, "if_match": str}'
),
),
current_user=Depends(require_auth),
@@ -239,7 +266,9 @@ def api_file_xlsx_structure(
Without ``force`` the call fails **409** ``xlsx_lossy_content`` when the
workbook carries features openpyxl cannot rewrite (same gate as the cell
edits). A backup is created before the archive is replaced.
edits). A backup is created before the archive is replaced. The optional
``if_match`` revision (#156-A12) refuses a structural rewrite on a file that
changed since it was read (**409** ``conflict``).
Args:
vault_name: Name of the vault.
@@ -258,16 +287,86 @@ def api_file_xlsx_structure(
raw_force = body.get("force", False)
if not isinstance(raw_force, bool):
raise HTTPException(status_code=400, detail="Flag invalide: force")
if_match = body.get("if_match")
if if_match is not None and not isinstance(if_match, str):
raise HTTPException(status_code=400, detail="Jeton invalide: if_match")
result = service_mutate_xlsx_structure(
vault_name, path, actions, force=raw_force
vault_name, path, actions, force=raw_force, expected_revision=if_match
)
log_file_save(
current_user["username"], vault_name, path,
len(actions),
current_user.get("_request_ip", "unknown"),
)
return {"status": "ok", "vault": result["vault"], "path": result["path"], "size": len(result["applied"])}
return {
"status": "ok", "vault": result["vault"], "path": result["path"],
"size": len(result["applied"]), "revision": result.get("revision"),
}
@router.put("/api/file/{vault_name}/xlsx/style", response_model=FileSaveResponse)
def api_file_xlsx_style(
vault_name: str,
path: str = Query(..., description="Relative path to the .xlsx/.xlsm file"),
body: dict = Body(
...,
description=(
'{"ops": [{"op": "cell", "sheet": "X", "range": "A1:B2", '
'"style": {"bold": true, "fill_color": "#ffe08a"}}], '
'"force": false, "if_match": str}'
),
),
current_user=Depends(require_auth),
):
"""Write formatting on an .xlsx/.xlsm workbook (#156-A8).
``ops`` is an ordered list applied in one locked, atomic rewrite:
``cell`` (``sheet``, ``range``/``cell``, ``style`` with ``bold``,
``italic``, ``underline``, ``font_color``/``fill_color`` as ``#rrggbb``,
``align`` and ``number_format``), ``merge``/``unmerge`` (``range``),
``col_width`` (``col``, ``width``), ``row_height`` (``row``, ``height``)
and ``freeze`` (``cell``, empty to release).
Without ``force`` the call fails **409** ``xlsx_lossy_content`` when the
workbook carries features openpyxl cannot rewrite (same gate as the cell
edits). A backup is created before the archive is replaced; the optional
``if_match`` revision (#156-A12) refuses a rewrite on a file that changed
since it was read (**409** ``conflict``).
Args:
vault_name: Name of the vault.
path: Relative path to the ``.xlsx``/``.xlsm`` file.
body: JSON body with ``ops`` (1 to 50) and optional ``force``.
Returns:
``FileSaveResponse`` confirming the write.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
ops = body.get("ops")
if not isinstance(ops, list) or not ops or len(ops) > 50:
raise HTTPException(status_code=400, detail="Ops invalides (1 à 50 par requête)")
raw_force = body.get("force", False)
if not isinstance(raw_force, bool):
raise HTTPException(status_code=400, detail="Flag invalide: force")
if_match = body.get("if_match")
if if_match is not None and not isinstance(if_match, str):
raise HTTPException(status_code=400, detail="Jeton invalide: if_match")
result = service_mutate_xlsx_style(
vault_name, path, ops, force=raw_force, expected_revision=if_match
)
log_file_save(
current_user["username"], vault_name, path,
len(ops),
current_user.get("_request_ip", "unknown"),
)
return {
"status": "ok", "vault": result["vault"], "path": result["path"],
"size": len(result["applied"]), "revision": result.get("revision"),
}
@router.delete("/api/file/{vault_name}", response_model=FileDeleteResponse)
+14
View File
@@ -301,6 +301,14 @@ class FileContentResponse(BaseModel):
"when the sheet exceeds the 500x40 render caps (#153 A8)"
),
)
xlsx_revision: str | None = Field(
default=None,
description=(
"Optimistic-concurrency token of the spreadsheet (#156-A12): the "
"client sends it back as the `if_match` of a write so a change made "
"elsewhere is refused (409 `conflict`) instead of overwritten"
),
)
xlsx_lossy_features: list[str] | None = Field(
default=None,
description=(
@@ -401,6 +409,12 @@ class FileSaveResponse(BaseModel):
vault: str = Field(description="Vault name")
path: str = Field(description="Relative file path within the vault")
size: int = Field(description="Size of saved content in characters")
# #156-A12 — optimistic-concurrency token of the file AFTER the write, so a
# client can chain writes without re-reading (absent on non-spreadsheets).
revision: str | None = Field(
default=None,
description="Opaque revision of the saved spreadsheet (send it back as `if_match`)",
)
class FileDeleteResponse(BaseModel):
+390 -6
View File
@@ -278,6 +278,67 @@ def _xlsx_write_lock(key: str) -> Iterator[None]:
lock.release()
def _invalidate_meta(file_path: Path) -> None:
"""Drop the cached workbook metadata after a write (#156-A13).
Imported lazily so the module stays importable when the spreadsheet reader
is not needed (it pulls openpyxl in).
"""
try:
from backend.xlsx_reader import invalidate_workbook_meta
invalidate_workbook_meta(file_path)
except Exception: # pragma: no cover - cache invalidation is best-effort
logger.debug("workbook meta invalidation skipped", exc_info=True)
def file_revision(file_path: Path) -> str:
"""Opaque revision token of a file — ``mtime_ns:size`` in hex (#156-A12).
Cheap by design (one ``stat``) and enough for optimistic concurrency: any
writer that replaces the file changes at least one of the two fields. The
viewer reads it with the file and sends it back as the ``if_match`` of a
write, so an external editor (Excel, the watcher, another worker) can no
longer be silently overwritten.
Returns:
The token, or ``""`` when the file cannot be stat'ed.
"""
try:
st = file_path.stat()
except OSError:
return ""
return f"{st.st_mtime_ns:x}-{st.st_size:x}"
def _check_revision(file_path: Path, expected: str | None) -> None:
"""Refuse a write on a file that changed since it was read (#156-A12).
``expected`` comes from the read payload (``if_match``): when it is absent
the write keeps its pre-A12 behaviour (last writer wins), so curl, the AI
tools and the batch uploader are unaffected.
Raises:
ServiceError: ``conflict`` (409, ``reason=stale_revision``) when the
file on disk is not the revision the caller read.
"""
if not expected:
return
current = file_revision(file_path)
if current and current != expected:
raise ServiceError(
"The file changed on disk since it was read; reload before saving",
code="conflict",
status=409,
details={
"reason": "stale_revision",
"path": str(file_path),
"expected_revision": expected,
"current_revision": current,
},
)
def _coerce_xlsx_value(value: Any) -> Any:
"""Turn the string sent by the cell editor back into a scalar (#153 A10).
@@ -354,6 +415,7 @@ def edit_xlsx_cells(
backup: bool = True,
allow_formula: bool = False,
force: bool = False,
expected_revision: str | None = None,
) -> dict[str, Any]:
"""Apply a batch of cell edits to an ``.xlsx`` workbook.
@@ -416,6 +478,9 @@ def edit_xlsx_cells(
)
with _xlsx_write_lock(str(file_path)):
# #156-A12 — inside the lock: no writer can slip in between the check
# and the load.
_check_revision(file_path, expected_revision)
from openpyxl import load_workbook
# #153 A16 — .xlsm round-trips with keep_vba=True so the macro
@@ -451,6 +516,9 @@ def edit_xlsx_cells(
except Exception:
tmp_path.unlink(missing_ok=True)
raise
# #156-A13 — the metadata cache is keyed on (mtime, size); drop it too
# so a rewrite that lands on the same tick can never serve stale maps.
_invalidate_meta(file_path)
logger.info(f"XLSX cells saved: {vault_name}/{rel_path} [{sheet}] +{len(cells)}")
return {
@@ -458,6 +526,7 @@ def edit_xlsx_cells(
"vault": vault_name,
"path": rel_path,
"size": len(cells),
"revision": file_revision(file_path),
}
@@ -468,6 +537,7 @@ def mutate_xlsx_structure(
*,
backup: bool = True,
force: bool = False,
expected_revision: str | None = None,
) -> dict[str, Any]:
"""Apply structural changes to an ``.xlsx`` workbook (#153 A14).
@@ -521,6 +591,8 @@ def mutate_xlsx_structure(
)
with _xlsx_write_lock(str(file_path)):
# #156-A12 — stale-write guard (see edit_xlsx_cells).
_check_revision(file_path, expected_revision)
from openpyxl import load_workbook
from openpyxl.worksheet.copier import WorksheetCopy
@@ -636,6 +708,7 @@ def mutate_xlsx_structure(
wb.close()
raise
wb.close()
_invalidate_meta(file_path)
logger.info(
f"XLSX structure: {vault_name}/{rel_path} {applied}"
@@ -645,6 +718,306 @@ def mutate_xlsx_structure(
"vault": vault_name,
"path": rel_path,
"applied": applied,
"revision": file_revision(file_path),
}
# #156-A8 — write-side formatting. Every value is data-driven (the colours
# come from the caller, never from a hardcoded palette) and the whole batch
# rides the same lock, backup and atomic swap as the cell edits.
_STYLE_KEYS = {
"bold",
"italic",
"underline",
"font_color",
"fill_color",
"align",
"number_format",
}
_STYLE_ALIGNS = {"left", "center", "right"}
MAX_STYLE_CELLS = 10_000
_HEX_COLOR_RE = re.compile(r"^#?[0-9a-fA-F]{6}$")
def _style_color(value: Any) -> str | None:
"""Normalise ``#rrggbb`` to openpyxl's ARGB, or ``None`` to clear."""
if value is None or value == "":
return None
if not isinstance(value, str) or not _HEX_COLOR_RE.match(value.strip()):
raise ServiceError(
f"Couleur invalide: {value!r}", code="invalid", status=400
)
return "FF" + value.strip().lstrip("#").upper()
def _apply_cell_style(cell: Any, style: dict[str, Any]) -> None:
"""Apply the data-driven style fragment of one ``cell`` operation.
Only the keys listed in :data:`_STYLE_KEYS` are accepted — an unknown one
is a client bug, not something to ignore silently. Font attributes are
copied before mutation so the shared style of the other cells is left
untouched.
"""
import copy as copy_mod
from openpyxl.styles import Alignment, Color, PatternFill
unknown = set(style) - _STYLE_KEYS
if unknown:
raise ServiceError(
f"Style inconnu: {sorted(unknown)}", code="invalid", status=400
)
if {"bold", "italic", "underline", "font_color"} & set(style):
font = copy_mod.copy(cell.font)
if "bold" in style:
font.bold = bool(style["bold"])
if "italic" in style:
font.italic = bool(style["italic"])
if "underline" in style:
font.underline = "single" if style["underline"] else None
if "font_color" in style:
rgb = _style_color(style["font_color"])
font.color = Color(rgb=rgb) if rgb else None
cell.font = font
if "fill_color" in style:
rgb = _style_color(style["fill_color"])
cell.fill = (
PatternFill(start_color=rgb, end_color=rgb, fill_type="solid")
if rgb
else PatternFill(fill_type=None)
)
if "align" in style:
align = style["align"]
if align not in _STYLE_ALIGNS:
raise ServiceError(
f"Alignement invalide: {align!r}", code="invalid", status=400
)
current = cell.alignment
cell.alignment = Alignment(
horizontal=align,
vertical=getattr(current, "vertical", None),
wrap_text=getattr(current, "wrap_text", None),
)
if "number_format" in style:
fmt = style["number_format"] or "General"
if not isinstance(fmt, str) or len(fmt) > 120:
raise ServiceError(
"Format de nombre invalide", code="invalid", status=400
)
cell.number_format = fmt
def mutate_xlsx_style(
vault_name: str,
path: str,
ops: list[dict[str, Any]],
*,
backup: bool = True,
force: bool = False,
expected_revision: str | None = None,
) -> dict[str, Any]:
"""Write formatting on an ``.xlsx``/``.xlsm`` workbook (#156-A8).
``ops`` is an ordered list (1 to 50) applied in one locked, atomic rewrite:
* ``{"op": "cell", "sheet": "X", "range": "A1:B2", "style": {…}}`` with
``bold``, ``italic``, ``underline``, ``font_color`` / ``fill_color``
(``#rrggbb``, ``""`` clears), ``align`` (left/center/right) and
``number_format`` ;
* ``{"op": "merge"|"unmerge", "sheet": "X", "range": "A1:B2"}`` ;
* ``{"op": "col_width", "sheet": "X", "col": "A", "width": 24}`` ;
* ``{"op": "row_height", "sheet": "X", "row": 3, "height": 30}`` ;
* ``{"op": "freeze", "sheet": "X", "cell": "B2"}`` (``""`` releases).
Same guards as the cell edits: per-file lock, ``.bak`` backup,
``.tmp`` + ``os.replace`` atomic swap, lossy-write 409 and the optional
``expected_revision`` concurrency check (#156-A12).
"""
from openpyxl.utils import get_column_letter
from openpyxl.utils.cell import range_boundaries
root = get_vault_root(vault_name)
_ensure_writable(root)
file_path = resolve_safe_path(root, path)
if not file_path.exists() or not file_path.is_file():
raise ServiceError(
f"File not found: {path}",
code="not_found",
status=404,
details={"vault": vault_name, "path": path},
)
if file_path.suffix.lower() not in (".xlsx", ".xlsm"):
raise ServiceError(
f"Not an .xlsx/.xlsm file: {path}", code="invalid", status=400
)
if not ops or len(ops) > 50:
raise ServiceError(
"Invalid ops (1 to 50 per request)", code="invalid", status=400
)
if not force and file_path.suffix.lower() != ".xlsm":
from backend.xlsx_reader import inspect_workbook
lossy = inspect_workbook(file_path)
if lossy:
raise ServiceError(
"Restyling this workbook would drop features ObsiGate cannot "
"preserve; retry with force=true after confirmation",
code="xlsx_lossy_content",
status=409,
details={"path": path, "features": lossy},
)
def _range(ref: str) -> tuple[int, int, int, int]:
try:
min_col, min_row, max_col, max_row = range_boundaries(str(ref).upper())
except Exception as exc:
raise ServiceError(
f"Plage invalide: {ref!r}", code="invalid", status=400
) from exc
if (max_row - min_row + 1) * (max_col - min_col + 1) > MAX_STYLE_CELLS:
raise ServiceError(
f"Plage trop grande (max {MAX_STYLE_CELLS} cellules)",
code="invalid",
status=400,
)
return min_col, min_row, max_col, max_row
def _sheet(wb: Any, action: dict[str, Any]) -> Any:
name = str(action.get("sheet", ""))
if name not in wb.sheetnames:
raise ServiceError(
f"Feuille introuvable: {name}",
code="invalid",
status=400,
details={"sheets": wb.sheetnames},
)
return wb[name]
applied: list[str] = []
with _xlsx_write_lock(str(file_path)):
_check_revision(file_path, expected_revision)
from openpyxl import load_workbook
try:
wb = load_workbook(
file_path, keep_vba=file_path.suffix.lower() == ".xlsm"
)
except Exception as exc:
raise ServiceError(
f"Cannot open workbook: {exc}", code="invalid", status=400
) from exc
rel_path = _rel(root, file_path)
try:
for i, action in enumerate(ops):
if not isinstance(action, dict):
raise ServiceError(
f"Action {i + 1} invalide", code="invalid", status=400
)
op = action.get("op")
try:
if op == "cell":
ws = _sheet(wb, action)
ref = str(action.get("range") or action.get("cell") or "")
min_col, min_row, max_col, max_row = _range(ref)
style = action.get("style") or {}
if not isinstance(style, dict) or not style:
raise ServiceError(
"Style vide", code="invalid", status=400
)
for row in ws.iter_rows(
min_row=min_row,
max_row=max_row,
min_col=min_col,
max_col=max_col,
):
for cell in row:
_apply_cell_style(cell, style)
applied.append(
f"cell:{ws.title}!{ref}:{','.join(sorted(style))}"
)
elif op in ("merge", "unmerge"):
ws = _sheet(wb, action)
ref = str(action.get("range", ""))
_range(ref)
if op == "merge":
ws.merge_cells(ref)
else:
ws.unmerge_cells(ref)
applied.append(f"{op}:{ws.title}!{ref}")
elif op == "col_width":
ws = _sheet(wb, action)
col = action.get("col")
width = action.get("width")
if isinstance(col, int):
col = get_column_letter(col)
if not isinstance(col, str) or not col.strip():
raise ServiceError(
"Colonne invalide", code="invalid", status=400
)
if not isinstance(width, (int, float)) or not 0 <= float(width) <= 255:
raise ServiceError(
"Largeur invalide (0 à 255)", code="invalid", status=400
)
ws.column_dimensions[col.strip().upper()].width = float(width)
applied.append(f"col_width:{ws.title}!{col}={width}")
elif op == "row_height":
ws = _sheet(wb, action)
row = action.get("row")
height = action.get("height")
if not isinstance(row, int) or row < 1:
raise ServiceError("Ligne invalide", code="invalid", status=400)
if not isinstance(height, (int, float)) or not 0 <= float(height) <= 409:
raise ServiceError(
"Hauteur invalide (0 à 409)", code="invalid", status=400
)
ws.row_dimensions[row].height = float(height)
applied.append(f"row_height:{ws.title}!{row}={height}")
elif op == "freeze":
ws = _sheet(wb, action)
cell_ref = str(action.get("cell", ""))
if cell_ref:
_range(cell_ref)
ws.freeze_panes = cell_ref.upper() or None
applied.append(f"freeze:{ws.title}!{cell_ref or '-'}")
else:
raise ServiceError(
f"Action inconnue: {op!r}", code="invalid", status=400
)
except ServiceError:
raise
except Exception as exc:
raise ServiceError(
f"Action {i + 1} ({op}) a échoué: {exc}",
code="invalid",
status=400,
) from exc
except ServiceError:
wb.close()
raise
if backup:
create_backup(file_path, vault_name, rel_path)
tmp_path = file_path.with_name(f"{file_path.name}.{os.getpid()}.tmp")
try:
wb.save(tmp_path)
os.replace(tmp_path, file_path)
except Exception:
tmp_path.unlink(missing_ok=True)
wb.close()
raise
wb.close()
_invalidate_meta(file_path)
logger.info(f"XLSX style: {vault_name}/{rel_path} {applied}")
return {
"success": True,
"vault": vault_name,
"path": rel_path,
"applied": applied,
"revision": file_revision(file_path),
}
@@ -654,14 +1027,18 @@ def save_csv_cells(
cells: dict[str, Any],
*,
backup: bool = True,
expected_revision: str | None = None,
) -> dict[str, Any]:
"""Apply A1-addressed cell edits to a ``.csv`` file (#153 A16).
The file is re-parsed, patched and re-serialized with :mod:`csv` so
quoting follows RFC 4180. References beyond the current extent grow the
grid (missing rows/cells are filled with empty strings). Values are
stored as text: a CSV has no formula engine, so any string — including
ones starting with ``=`` — is written verbatim (the render escapes it).
quoting follows RFC 4180. BUG-098 — the delimiter is **sniffed** and the
same one is reused on write-back: a `;`-separated French CSV used to be
parsed as a single column and rewritten with `,`. References beyond the
current extent grow the grid (missing rows/cells are filled with empty
strings). Values are stored as text: a CSV has no formula engine, so any
string — including ones starting with ``=`` — is written verbatim (the
render escapes it).
Raises:
ServiceError: ``not_found`` (404), ``read_only`` (403), ``conflict``
@@ -670,6 +1047,8 @@ def save_csv_cells(
import csv as csv_mod
import io as io_mod
from backend.xlsx_reader import sniff_csv_delimiter
root = get_vault_root(vault_name)
_ensure_writable(root)
file_path = resolve_safe_path(root, path)
@@ -691,9 +1070,13 @@ def save_csv_cells(
f"Invalid cell reference: {ref!r}", code="invalid", status=400
)
# #156-A12 — same stale-write guard as the workbook path.
_check_revision(file_path, expected_revision)
raw = file_path.read_text(encoding="utf-8", errors="replace")
# BUG-098 — parse AND rewrite with the file's own delimiter.
delimiter = sniff_csv_delimiter(raw)
try:
rows = list(csv_mod.reader(io_mod.StringIO(raw)))
rows = list(csv_mod.reader(io_mod.StringIO(raw), delimiter=delimiter))
except csv_mod.Error:
rows = [[line] for line in raw.splitlines()]
@@ -721,7 +1104,7 @@ def save_csv_cells(
create_backup(file_path, vault_name, rel_path)
buf = io_mod.StringIO()
csv_mod.writer(buf, lineterminator="\n").writerows(rows)
csv_mod.writer(buf, delimiter=delimiter, lineterminator="\n").writerows(rows)
tmp_path = file_path.with_name(f"{file_path.name}.{os.getpid()}.tmp")
try:
tmp_path.write_text(buf.getvalue(), encoding="utf-8")
@@ -736,6 +1119,7 @@ def save_csv_cells(
"vault": vault_name,
"path": rel_path,
"size": len(cells),
"revision": file_revision(file_path),
}
+3
View File
@@ -56,6 +56,9 @@ _STEP_LABELS: dict[str, tuple[str, str | None]] = {
"xlsx_to_markdown": ("xlsx_read", "path"),
"update_xlsx_cells": ("xlsx_update", "path"),
"append_xlsx_rows": ("xlsx_append", "path"),
"search_workbook": ("xlsx_search", "query"),
"analyze_range": ("xlsx_analyze", "path"),
"edit_xlsx_structure": ("xlsx_structure", "path"),
"create_docx": ("docx_create", "path"),
"create_csv": ("csv_create", "path"),
"create_pdf": ("pdf_create", "path"),
+50 -1
View File
@@ -332,12 +332,42 @@ class XlsxToMarkdownInput(BaseModel):
)
class SearchWorkbookInput(BaseModel):
"""Find a text across the sheets of a spreadsheet (#156 A14)."""
vault: str = Field(..., description="Vault name")
path: str = Field(
..., description="Vault-relative path of the file (.xlsx, .xlsm or .csv)"
)
query: str = Field(..., description="Text to look for")
sheet: str = Field("", description="Restrict to one sheet (empty = all sheets)")
case_sensitive: bool = Field(False, description="Match case")
class AnalyzeRangeInput(BaseModel):
"""Aggregate the values of an A1 range (#156 A14)."""
vault: str = Field(..., description="Vault name")
path: str = Field(
..., description="Vault-relative path of the file (.xlsx, .xlsm or .csv)"
)
sheet: str = Field(
"", description="Sheet name (empty = the first/active sheet)"
)
range: str = Field(
"",
description="A1 range to analyse (e.g. 'B2:B50'); empty = the whole sheet",
)
class UpdateXlsxCellsInput(BaseModel):
"""Batch-edit cells of an existing .xlsx workbook (#153 A6)."""
vault: str = Field(..., description="Vault name")
path: str = Field(..., description="Vault-relative path of the .xlsx file")
sheet: str = Field(..., description="Worksheet title to edit")
sheet: str = Field(
"", description="Worksheet title to edit (ignored for a .csv)"
)
cells: dict[str, str | int | float | bool | None] = Field(
..., description="A1 reference -> new value (max 500 per call)"
)
@@ -370,6 +400,25 @@ class AppendXlsxRowsInput(BaseModel):
)
class EditXlsxStructureInput(BaseModel):
"""Structural CRUD on an existing workbook (#156 A14)."""
vault: str = Field(..., description="Vault name")
path: str = Field(..., description="Vault-relative path of the .xlsx/.xlsm file")
actions: list[dict[str, Any]] = Field(
...,
description=(
"Ordered structural actions (1-50): sheet_add/sheet_rename/"
"sheet_duplicate/sheet_delete, row_insert/row_delete/"
"col_insert/col_delete"
),
)
force: bool = Field(
False,
description="Write even when features openpyxl cannot rewrite would be dropped",
)
class CsvInput(BaseModel):
"""Create a .csv file in a vault from rows of cells."""
+392 -105
View File
@@ -1,21 +1,30 @@
"""Spreadsheet tools (#153 A6) — read and mutate existing ``.xlsx`` workbooks.
"""Spreadsheet tools (#153 A6, #156 A14) — read and mutate existing workbooks.
Complements :mod:`backend.tools.documents` (``create_xlsx`` creates a *new*
file; here the assistant can read and edit one that already exists):
file; here the assistant can read and edit one that already exists). The same
three formats the viewer edits are supported — ``.xlsx``, ``.xlsm`` and
``.csv`` (#156-A14 used to be ``.xlsx`` only, which made a workbook the UI
edits invisible to the assistant):
* ``list_xlsx_sheets`` — READ, sheet names + dimensions;
* ``xlsx_to_markdown`` — READ, bounded markdown table for the LLM context;
* ``update_xlsx_cells`` — WRITE, batch cell edits (wraps the guarded service);
* ``append_xlsx_rows`` — WRITE, append whole rows at the end of a sheet.
* ``list_xlsx_sheets`` — READ, sheet names + dimensions;
* ``xlsx_to_markdown`` — READ, bounded markdown table for the LLM context;
* ``search_workbook`` — READ, find text across every sheet (#156-A14);
* ``analyze_range`` — READ, aggregate stats over an A1 range (#156-A14);
* ``update_xlsx_cells`` — WRITE, batch cell edits (guarded service);
* ``append_xlsx_rows`` — WRITE, append whole rows at the end of a sheet;
* ``edit_xlsx_structure`` — WRITE, structural CRUD (sheets/rows/columns).
Mutation tools go through :func:`backend.services.mutations.edit_xlsx_cells`,
which already carries the #153 P0 guards: per-file lock, atomic replace,
formula neutralisation (``allow_formula`` opt-in) and the lossy-write 409.
Mutation tools go through :func:`backend.services.mutations.edit_xlsx_cells`
(or ``save_csv_cells`` / ``mutate_xlsx_structure``), which already carry the
#153 P0 guards: per-file lock, atomic replace, formula neutralisation
(``allow_formula`` opt-in) and the lossy-write 409. Every write is a WRITE-risk
tool, so the registry keeps asking for an explicit confirmation.
"""
from __future__ import annotations
import logging
import re
from pathlib import Path
from typing import Any
@@ -25,8 +34,11 @@ from backend.services.vaults import get_vault_root
from backend.tools.context import ToolContext, ToolError, ToolRisk
from backend.tools.registry import tool
from backend.tools.schemas import (
AnalyzeRangeInput,
AppendXlsxRowsInput,
EditXlsxStructureInput,
ListXlsxSheetsInput,
SearchWorkbookInput,
UpdateXlsxCellsInput,
XlsxToMarkdownInput,
)
@@ -39,12 +51,28 @@ MAX_MD_ROWS = 100
MAX_MD_COLS = 20
MAX_MD_CHARS = 20_000
# #156-A14 — the newer read tools scan more than the markdown table (a search
# or an aggregate must not stop at row 100) but stay bounded all the same:
# a runaway scan would load a whole ledger into the model's context.
MAX_SCAN_ROWS = 5_000
MAX_SCAN_COLS = 100
MAX_SEARCH_RESULTS = 100
MAX_RANGE_CELLS = 10_000
MAX_RANGE_VALUES = 200
def _workbook_path(vault: str, path: str) -> Path:
"""Resolve and validate a vault-relative ``.xlsx`` path."""
# The formats the spreadsheet editor (and now the assistant) can handle.
SPREADSHEET_EXTENSIONS = (".xlsx", ".xlsm", ".csv")
_NUMBER_RE = re.compile(r"^-?\d+(?:[.,]\d+)?$")
def _spreadsheet_path(vault: str, path: str) -> Path:
"""Resolve and validate a vault-relative ``.xlsx``/``.xlsm``/``.csv`` path."""
path = (path or "").strip()
if not path.lower().endswith(".xlsx"):
raise ToolError("Extension attendue : .xlsx", code="invalid_arguments")
if not path.lower().endswith(SPREADSHEET_EXTENSIONS):
raise ToolError(
"Extension attendue : .xlsx, .xlsm ou .csv", code="invalid_arguments"
)
try:
root = get_vault_root(vault)
except ServiceError as e:
@@ -52,57 +80,175 @@ def _workbook_path(vault: str, path: str) -> Path:
return resolve_safe_path(root, path)
def _is_csv(file_path: Path) -> bool:
return file_path.suffix.lower() == ".csv"
def _map_service_error(e: ServiceError) -> ToolError:
return ToolError(e.message, code=e.code, details=e.details)
@tool(
name="list_xlsx_sheets",
description=(
"List the sheets of an .xlsx workbook with their dimensions "
"(rows x columns) and whether the display caps truncate them. "
"Use before editing to pick the right sheet name."
),
input_model=ListXlsxSheetsInput,
risk=ToolRisk.READ,
requires_vault=True,
)
def list_xlsx_sheets(ctx: ToolContext, params: ListXlsxSheetsInput) -> dict[str, Any]:
"""Return sheet names and extents of the workbook."""
from backend.xlsx_reader import MAX_COLS, MAX_ROWS, _sheet_extent
def _sheet_titles(file_path: Path) -> list[str]:
"""Sheet names of a workbook; a CSV has a single, unnamed “sheet”."""
if _is_csv(file_path):
return [""]
from openpyxl import load_workbook
file_path = _workbook_path(params.vault, params.path)
wb = load_workbook(str(file_path), read_only=True, data_only=True)
try:
from openpyxl import load_workbook
return list(wb.sheetnames)
finally:
wb.close()
def _read_grid(
file_path: Path, sheet: str, max_rows: int, max_cols: int
) -> tuple[str, list[list[str]], bool]:
"""Read one sheet (or a CSV) as bounded, formatted, trimmed rows.
Returns ``(title, rows, truncated)``; ``truncated`` is True when real data
sits just beyond the row cap (probed one row further) so the caller can say
so instead of silently dropping it.
"""
from openpyxl import load_workbook
from backend.xlsx_reader import _fmt
if _is_csv(file_path):
import csv as csv_mod
import io as io_mod
from backend.xlsx_reader import sniff_csv_delimiter
stem = file_path.stem
if sheet and sheet != stem:
raise ToolError(f"Feuille introuvable: {sheet}", code="not_found")
raw = file_path.read_text(encoding="utf-8-sig", errors="replace")
reader = csv_mod.reader(io_mod.StringIO(raw), delimiter=sniff_csv_delimiter(raw))
grid: list[list[str]] = []
truncated = False
for i, row in enumerate(reader):
if i >= max_rows:
truncated = any(str(c).strip() for c in row)
break
grid.append([str(c) for c in row][:max_cols])
while grid and not any(c.strip() for c in grid[-1]):
grid.pop()
return stem, grid, truncated
try:
wb = load_workbook(str(file_path), read_only=True, data_only=True)
except ServiceError as e:
raise _map_service_error(e) from e
except Exception as e:
raise ToolError(f"Classeur illisible: {e}", code="invalid") from e
try:
sheets = []
for ws in wb.worksheets:
total_rows, total_cols = _sheet_extent(ws)
sheets.append(
{
"name": ws.title,
"total_rows": total_rows,
"total_cols": total_cols,
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
}
if sheet:
if sheet not in wb.sheetnames:
raise ToolError(f"Feuille introuvable: {sheet}", code="not_found")
ws = wb[sheet]
else:
ws = wb.active
title = ws.title
grid = []
for row in ws.iter_rows(
min_row=1, max_row=max_rows, max_col=max_cols, values_only=True
):
grid.append([_fmt(v) for v in row])
probe = list(
ws.iter_rows(
min_row=max_rows + 1,
max_row=max_rows + 1,
max_col=max_cols,
values_only=True,
)
return {"vault": params.vault, "path": params.path, "sheets": sheets}
)
truncated = any(any(str(v or "").strip() for v in r) for r in probe)
finally:
wb.close()
while grid and not any(c.strip() for c in grid[-1]):
grid.pop()
return title, grid, truncated
def _to_number(text: str) -> float | None:
"""Coerce a displayed cell to a float, or ``None`` when it is not one."""
t = text.strip().replace("\u00a0", "").replace(" ", "")
if not _NUMBER_RE.match(t):
return None
try:
return float(t.replace(",", "."))
except ValueError: # pragma: no cover - regex already guarantees the shape
return None
@tool(
name="list_xlsx_sheets",
description=(
"List the sheets of a spreadsheet (.xlsx, .xlsm or .csv) with their "
"dimensions (rows x columns) and whether the display caps truncate "
"them. Use before editing to pick the right sheet name."
),
input_model=ListXlsxSheetsInput,
risk=ToolRisk.READ,
requires_vault=True,
)
def list_xlsx_sheets(ctx: ToolContext, params: ListXlsxSheetsInput) -> dict[str, Any]:
"""Return sheet names and extents of the workbook (or of the CSV)."""
from backend.xlsx_reader import MAX_COLS, MAX_ROWS
file_path = _spreadsheet_path(params.vault, params.path)
if not file_path.exists() or not file_path.is_file():
raise ToolError(f"Fichier introuvable: {params.path}", code="not_found")
extents: list[tuple[str, int, int]] = []
if _is_csv(file_path):
_, rows, _ = _read_grid(file_path, "", MAX_SCAN_ROWS, MAX_SCAN_COLS)
extents.append(
(
file_path.stem,
len(rows),
max((len(r) for r in rows), default=0),
)
)
else:
# Declared dimensions are enough here (and far cheaper than scanning
# every row): the caller just needs a size to decide what to read.
from openpyxl import load_workbook
from backend.xlsx_reader import _sheet_extent
try:
wb = load_workbook(str(file_path), read_only=True, data_only=True)
except ServiceError as e:
raise _map_service_error(e) from e
except Exception as e:
raise ToolError(f"Classeur illisible: {e}", code="invalid") from e
try:
extents = [
(ws.title, *_sheet_extent(ws)) for ws in wb.worksheets
]
finally:
wb.close()
sheets = [
{
"name": name,
"total_rows": total_rows,
"total_cols": total_cols,
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
}
for name, total_rows, total_cols in extents
]
return {"vault": params.vault, "path": params.path, "sheets": sheets}
@tool(
name="xlsx_to_markdown",
description=(
"Read a sheet of an .xlsx workbook as a bounded markdown table "
"(up to 100 rows x 20 columns). Use to inspect spreadsheet data "
"before answering or editing."
"Read a sheet of a spreadsheet (.xlsx, .xlsm or .csv) as a bounded "
"markdown table (up to 100 rows x 20 columns). Use to inspect "
"spreadsheet data before answering or editing."
),
input_model=XlsxToMarkdownInput,
risk=ToolRisk.READ,
@@ -110,49 +256,8 @@ def list_xlsx_sheets(ctx: ToolContext, params: ListXlsxSheetsInput) -> dict[str,
)
def xlsx_to_markdown(ctx: ToolContext, params: XlsxToMarkdownInput) -> dict[str, Any]:
"""Render one sheet as a markdown table for the LLM context."""
from openpyxl import load_workbook
from backend.xlsx_reader import _fmt
file_path = _workbook_path(params.vault, params.path)
try:
wb = load_workbook(str(file_path), read_only=True, data_only=True)
except ServiceError as e:
raise _map_service_error(e) from e
except Exception as e:
raise ToolError(f"Classeur illisible: {e}", code="invalid") from e
try:
if params.sheet:
if params.sheet not in wb.sheetnames:
raise ToolError(
f"Feuille introuvable: {params.sheet}", code="not_found"
)
ws = wb[params.sheet]
else:
ws = wb.active
title = ws.title
rows: list[list[str]] = []
truncated = False
for row in ws.iter_rows(
min_row=1, max_row=MAX_MD_ROWS, max_col=MAX_MD_COLS, values_only=True
):
cells = [_fmt(v) for v in row]
if not any(c.strip() for c in cells):
continue
rows.append(cells)
# Real tail beyond the caps? Probe one row further.
probe = list(
ws.iter_rows(
min_row=MAX_MD_ROWS + 1,
max_row=MAX_MD_ROWS + 1,
max_col=MAX_MD_COLS,
values_only=True,
)
)
if any(any(str(v or "").strip() for v in r) for r in probe):
truncated = True
finally:
wb.close()
file_path = _spreadsheet_path(params.vault, params.path)
title, rows, truncated = _read_grid(file_path, params.sheet, MAX_MD_ROWS, MAX_MD_COLS)
lines: list[str] = []
if rows:
@@ -174,34 +279,166 @@ def xlsx_to_markdown(ctx: ToolContext, params: XlsxToMarkdownInput) -> dict[str,
}
@tool(
name="search_workbook",
description=(
"Search a text across every sheet of a spreadsheet (.xlsx, .xlsm or "
".csv) and return the matching cells with their sheet and A1 "
"reference (max 100 matches). Use it to find where a value lives "
"without dumping whole sheets into the context."
),
input_model=SearchWorkbookInput,
risk=ToolRisk.READ,
requires_vault=True,
)
def search_workbook(ctx: ToolContext, params: SearchWorkbookInput) -> dict[str, Any]:
"""Find a needle across all sheets, bounded and counted per sheet."""
from openpyxl.utils import get_column_letter
file_path = _spreadsheet_path(params.vault, params.path)
needle = (params.query or "").strip()
if not needle:
raise ToolError("Requête vide", code="invalid_arguments")
titles = _sheet_titles(file_path)
if params.sheet:
if params.sheet not in titles:
raise ToolError(f"Feuille introuvable: {params.sheet}", code="not_found")
titles = [params.sheet]
hay = needle if params.case_sensitive else needle.lower()
matches: list[dict[str, Any]] = []
by_sheet: dict[str, int] = {}
total = 0
for title in titles:
sheet_title, rows, _ = _read_grid(
file_path, title, MAX_SCAN_ROWS, MAX_SCAN_COLS
)
label = sheet_title or file_path.stem
for r_i, row in enumerate(rows, start=1):
for c_i, value in enumerate(row, start=1):
if not value:
continue
haystack = value if params.case_sensitive else value.lower()
if hay not in haystack:
continue
total += 1
by_sheet[label] = by_sheet.get(label, 0) + 1
if len(matches) < MAX_SEARCH_RESULTS:
matches.append(
{
"sheet": label,
"cell": f"{get_column_letter(c_i)}{r_i}",
"value": value,
}
)
return {
"vault": params.vault,
"path": params.path,
"query": needle,
"total": total,
"truncated": total > MAX_SEARCH_RESULTS,
"by_sheet": by_sheet,
"matches": matches,
}
@tool(
name="analyze_range",
description=(
"Aggregate an A1 range of a sheet (.xlsx, .xlsm or .csv): count, sum, "
"mean, min and max of the numeric cells, plus a bounded sample of the "
"values. Use it to answer a question about a column without reading "
"the whole sheet."
),
input_model=AnalyzeRangeInput,
risk=ToolRisk.READ,
requires_vault=True,
)
def analyze_range(ctx: ToolContext, params: AnalyzeRangeInput) -> dict[str, Any]:
"""Numeric aggregates + value sample over an A1 range of the sheet."""
from openpyxl.utils.cell import range_boundaries
file_path = _spreadsheet_path(params.vault, params.path)
title, rows, _ = _read_grid(file_path, params.sheet, MAX_SCAN_ROWS, MAX_SCAN_COLS)
label = (params.range or "").strip()
if label:
try:
min_col, min_row, max_col, max_row = range_boundaries(label.upper())
except Exception as e:
raise ToolError(f"Plage invalide: {label}", code="invalid_arguments") from e
if (max_row - min_row + 1) * (max_col - min_col + 1) > MAX_RANGE_CELLS:
raise ToolError(
f"Plage trop grande (max {MAX_RANGE_CELLS} cellules)",
code="invalid_arguments",
)
selected = [row[min_col - 1 : max_col] for row in rows[min_row - 1 : max_row]]
else:
selected = rows
values = [v for row in selected for v in row if isinstance(v, str) and v.strip()]
numbers = [n for n in (_to_number(v) for v in values) if n is not None]
stats: dict[str, Any] = {"count": len(numbers)}
if numbers:
stats.update(
{
"sum": round(sum(numbers), 6),
"mean": round(sum(numbers) / len(numbers), 6),
"min": min(numbers),
"max": max(numbers),
}
)
return {
"vault": params.vault,
"path": params.path,
"sheet": title,
"range": params.range or "",
"rows": len(selected),
"cols": max((len(r) for r in selected), default=0),
"cells": len(values),
"numeric": stats,
"values": values[:MAX_RANGE_VALUES],
"truncated": len(values) > MAX_RANGE_VALUES,
}
@tool(
name="update_xlsx_cells",
description=(
"Edit cells of an existing .xlsx workbook. ``cells`` maps A1 "
"references to new values (max 500). A value starting with '=' or "
"'@' is stored as TEXT unless allow_formula is set (DDE guard). "
"Editing a workbook carrying features openpyxl cannot rewrite "
"requires force=true (cached formula results, slicers…)."
"Edit cells of an existing spreadsheet (.xlsx, .xlsm or .csv). "
"``cells`` maps A1 references to new values (max 500). A value "
"starting with '=' or '@' is stored as TEXT unless allow_formula is "
"set (DDE guard). Editing a workbook carrying features openpyxl "
"cannot rewrite requires force=true (cached formula results, "
"slicers…). A .csv has no sheet: any ``sheet`` value is ignored."
),
input_model=UpdateXlsxCellsInput,
risk=ToolRisk.WRITE,
requires_vault=True,
)
def update_xlsx_cells(ctx: ToolContext, params: UpdateXlsxCellsInput) -> dict[str, Any]:
"""Wrap the guarded cell-edit service."""
from backend.services.mutations import edit_xlsx_cells
"""Wrap the guarded cell-edit service (``.xlsx``/``.xlsm`` or ``.csv``)."""
from backend.services.mutations import edit_xlsx_cells, save_csv_cells
file_path = _spreadsheet_path(params.vault, params.path)
if not params.cells:
raise ToolError("Aucune cellule fournie", code="invalid_arguments")
if not _is_csv(file_path) and not params.sheet:
raise ToolError("Feuille requise pour un classeur", code="invalid_arguments")
try:
result = edit_xlsx_cells(
params.vault,
params.path,
params.sheet,
dict(params.cells),
allow_formula=params.allow_formula,
force=params.force,
)
if _is_csv(file_path):
result = save_csv_cells(params.vault, params.path, dict(params.cells))
else:
result = edit_xlsx_cells(
params.vault,
params.path,
params.sheet,
dict(params.cells),
allow_formula=params.allow_formula,
force=params.force,
)
except ServiceError as e:
raise _map_service_error(e) from e
return {
@@ -216,9 +453,10 @@ def update_xlsx_cells(ctx: ToolContext, params: UpdateXlsxCellsInput) -> dict[st
@tool(
name="append_xlsx_rows",
description=(
"Append rows at the end of a sheet of an existing .xlsx workbook. "
"Values are typed like in the viewer (numbers, TRUE/FALSE, FR dates "
"JJ/MM/AAAA). The workbook is rewritten atomically with a backup."
"Append rows at the end of a sheet of an existing spreadsheet "
"(.xlsx or .xlsm). Values are typed like in the viewer (numbers, "
"TRUE/FALSE, FR dates JJ/MM/AAAA). The workbook is rewritten "
"atomically with a backup. Not available for .csv."
),
input_model=AppendXlsxRowsInput,
risk=ToolRisk.WRITE,
@@ -236,7 +474,13 @@ def append_xlsx_rows(ctx: ToolContext, params: AppendXlsxRowsInput) -> dict[str,
if len(params.rows) > 500:
raise ToolError("Trop de lignes (max 500)", code="invalid_arguments")
file_path = _workbook_path(params.vault, params.path)
file_path = _spreadsheet_path(params.vault, params.path)
if _is_csv(file_path):
raise ToolError(
"Un .csv n'a pas de notion de fin de feuille : utilisez "
"update_xlsx_cells avec des références A1",
code="invalid_arguments",
)
try:
wb = load_workbook(str(file_path), read_only=True, data_only=True)
try:
@@ -284,3 +528,46 @@ def append_xlsx_rows(ctx: ToolContext, params: AppendXlsxRowsInput) -> dict[str,
"rows": len(params.rows),
"first_row": first_free,
}
@tool(
name="edit_xlsx_structure",
description=(
"Change the structure of an existing .xlsx/.xlsm workbook: add, "
"rename, duplicate or delete a sheet, or insert/delete rows and "
"columns. ``actions`` is an ordered list of "
'{"op": "sheet_add"|"sheet_rename"|"sheet_duplicate"|"sheet_delete"|'
'"row_insert"|"row_delete"|"col_insert"|"col_delete", …} '
"(1 to 50). Deletions drop data and cannot be undone from the "
"assistant — confirm with the user first."
),
input_model=EditXlsxStructureInput,
risk=ToolRisk.WRITE,
requires_vault=True,
)
def edit_xlsx_structure(
ctx: ToolContext, params: EditXlsxStructureInput
) -> dict[str, Any]:
"""Apply a batch of structural changes through the guarded service."""
from backend.services.mutations import mutate_xlsx_structure
if not params.actions:
raise ToolError("Aucune action fournie", code="invalid_arguments")
if len(params.actions) > 50:
raise ToolError("Trop d'actions (max 50)", code="invalid_arguments")
if str(params.path or "").lower().endswith(".csv"):
raise ToolError(
"Un .csv n'a pas de structure modifiable", code="invalid_arguments"
)
try:
result = mutate_xlsx_structure(
params.vault, params.path, [dict(a) for a in params.actions], force=params.force
)
except ServiceError as e:
raise _map_service_error(e) from e
return {
"status": "ok",
"vault": result["vault"],
"path": result["path"],
"actions": len(params.actions),
}
+146 -14
View File
@@ -14,9 +14,11 @@ read-only; :func:`render_csv_table` turns a CSV into the same table shape.
from __future__ import annotations
import csv
import html
import logging
import re
import threading
import zipfile
from datetime import date, datetime
from pathlib import Path
@@ -65,8 +67,22 @@ LOSSY_PARTS: dict[str, tuple[str, ...]] = {
# Excel recalculates.
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
# Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES).
_MAX_PROBE_BYTES = 8_000_000
# Sheet XML scanned by the cached-formula probe (CPU guard, like
# MAX_REPLACE_FILE_BYTES). BUG-099 — the allowance is PER SHEET: a single huge
# first sheet used to eat the whole budget and hide a cached formula sitting in
# the next one. The total ceiling still bounds the work on a many-sheet archive.
# Running out of budget is reported as *unverified* (never as "nothing to lose")
# so the write guard stays cautious instead of silently dropping the values.
_MAX_PROBE_BYTES_PER_SHEET = 4_000_000
_MAX_PROBE_BYTES_TOTAL = 32_000_000
# #156-A13 — metadata cache. `read_sheet_window()` serves one window at a time
# and used to reload the whole workbook (normal mode, data_only=False) for every
# window, just to read the style/merge/freeze maps of one sheet. The result is
# keyed by (path, mtime_ns, size): any write replaces the file, hence the key.
_META_CACHE_MAX = 8
_meta_cache: dict[str, tuple[tuple[int, int], dict[str, dict[str, Any]]]] = {}
_meta_cache_lock = threading.Lock()
# #153 A17 — OPC parts of chart / pivot objects, matched against the archive
# name list (xl/charts/chart1.xml, xl/pivotTables/pivotTable1.xml, …).
@@ -111,6 +127,10 @@ def _cell_fragments(cell: Any) -> tuple[list[str], str | None]:
fragments.append("font-weight:600")
if font and font.italic:
fragments.append("font-style:italic")
# #156-A8 — underline is written from the viewer too, so it is read back
# (the toggle in the formatting menu needs to see its own effect).
if font and font.underline:
fragments.append("text-decoration:underline")
fmt = cell.number_format
if fmt and fmt not in ("General", "@"):
# A custom number format is signalled typographically (mono font)
@@ -320,24 +340,65 @@ def _table(
return "".join(out)
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
"""True when at least one formula cell still carries its computed value."""
budget = _MAX_PROBE_BYTES
def _is_sheet_xml(name: str) -> bool:
"""True for the worksheet XML parts the cached-formula probe scans."""
return name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
def _scan_cached_formulas(zf: zipfile.ZipFile) -> tuple[bool, bool]:
"""Scan the sheet XML for a formula carrying a non-empty cached result.
Returns ``(found, unverified)``. ``unverified`` is True when the byte
budget stopped the scan before every sheet could be read to the end: a
negative result is then **not** proof that the workbook holds no cached
value (BUG-099), so callers must not treat it as a licence to write.
Each sheet gets its own :data:`_MAX_PROBE_BYTES_PER_SHEET` allowance (a
single huge sheet can no longer starve the others) while
:data:`_MAX_PROBE_BYTES_TOTAL` bounds the whole archive. A hit short-
circuits the scan: the answer is already known.
"""
budget = _MAX_PROBE_BYTES_TOTAL
unverified = False
for name in zf.namelist():
if not name.startswith("xl/worksheets/sheet") or not name.endswith(".xml"):
if not _is_sheet_xml(name):
continue
sheet_budget = min(_MAX_PROBE_BYTES_PER_SHEET, budget)
exhausted = False
try:
with zf.open(name) as fh:
while budget > 0:
chunk = fh.read(65536)
while sheet_budget > 0:
# Read at most what the sheet's allowance has left, so one
# large chunk can never consume the whole total budget.
chunk = fh.read(min(65536, sheet_budget))
if not chunk:
break
break # read to the end: this sheet is verified clean
sheet_budget -= len(chunk)
budget -= len(chunk)
if _CACHED_FORMULA_RE.search(chunk):
return True
return True, False
else:
# Left the loop on the budget, not on EOF.
exhausted = True
except (KeyError, OSError, zipfile.BadZipFile):
continue
return False
if exhausted:
unverified = True
if budget <= 0:
unverified = True
break
return False, unverified
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
"""True when at least one formula cell still carries its computed value.
Boolean view of :func:`_scan_cached_formulas` for the display path (the
second ``data_only=True`` read): a truncated scan simply skips the shadow
grid, it never claims the workbook is lossless.
"""
found, _ = _scan_cached_formulas(zf)
return found
def inspect_workbook(file_path: Path) -> list[str]:
@@ -349,6 +410,10 @@ def inspect_workbook(file_path: Path) -> list[str]:
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
the cached result, so the workbook stays correct once Excel recalculates it.
``cached_values_unverified`` (BUG-099) is the other synthetic key: the
cached-value probe ran out of budget, so a negative result is not proof —
the entry keeps the write guard cautious (409 + confirmation) rather than
promising a lossless round-trip it cannot vouch for.
"""
try:
with zipfile.ZipFile(file_path) as zf:
@@ -358,13 +423,27 @@ def inspect_workbook(file_path: Path) -> list[str]:
for key, prefixes in LOSSY_PARTS.items()
if any(name.startswith(prefix) for name in names for prefix in prefixes)
}
if _has_cached_formulas(zf):
cached_found, cached_unverified = _scan_cached_formulas(zf)
if cached_found:
found.add("cached_values")
elif cached_unverified:
found.add("cached_values_unverified")
return sorted(found)
except (OSError, zipfile.BadZipFile):
return []
def invalidate_workbook_meta(file_path: Path | str) -> None:
"""Drop the cached metadata of one workbook (#156-A13).
Called by the write path right after the atomic replace: the (mtime, size)
key already changes on a rewrite, this only closes the theoretical window
where a same-size write lands on the same timestamp tick.
"""
with _meta_cache_lock:
_meta_cache.pop(str(file_path), None)
def read_workbook_meta(file_path: Path) -> dict[str, dict[str, Any]]:
"""Return ``{sheet: {styles, aligns, merges, freeze}}`` for every sheet.
@@ -372,7 +451,31 @@ def read_workbook_meta(file_path: Path) -> dict[str, dict[str, Any]]:
fragments are the workbook's own values, a failure yields ``{}`` per sheet
so the viewer keeps its plain rendering. Styles are read with
``data_only=False`` — the edited value is the formula, not its result.
#156-A13 — the result is cached on ``(path, mtime_ns, size)``: a truncated
sheet is fetched window by window, and each window used to pay a full
workbook load for these maps alone. Callers only read from the mapping.
"""
try:
st = file_path.stat()
except OSError:
return {}
key = str(file_path)
stamp = (st.st_mtime_ns, st.st_size)
with _meta_cache_lock:
hit = _meta_cache.get(key)
if hit and hit[0] == stamp:
return hit[1]
out = _read_workbook_meta_uncached(file_path)
with _meta_cache_lock:
_meta_cache[key] = (stamp, out)
while len(_meta_cache) > _META_CACHE_MAX:
_meta_cache.pop(next(iter(_meta_cache)))
return out
def _read_workbook_meta_uncached(file_path: Path) -> dict[str, dict[str, Any]]:
"""Load the workbook once and build the per-sheet metadata maps."""
try:
wb = load_workbook(str(file_path), data_only=False)
except Exception:
@@ -781,17 +884,46 @@ def read_workbook_dashboard(file_path: Path) -> dict[str, Any]:
# ── #153 A16 — additional spreadsheet formats ───────────────────────────────
def render_csv_table(raw: str, *, delimiter: str = ",") -> str:
# BUG-098 — a CSV written by a French office suite is `;`-separated, and the
# delimiter must be *detected* (then reused on write-back), not assumed.
CSV_SNIFF_BYTES = 4096
_CSV_DELIMITERS = (",", ";", "\t", "|")
def sniff_csv_delimiter(raw: str) -> str:
"""Return the most likely field delimiter of *raw* (``,`` as fallback).
:class:`csv.Sniffer` handles quoted fields and multi-line values; it is
unreliable on short or single-column samples, so the first non-empty line
is counted as a tie-breaker and a comma remains the last resort.
"""
sample = raw[:CSV_SNIFF_BYTES]
try:
return csv.Sniffer().sniff(sample, delimiters="".join(_CSV_DELIMITERS)).delimiter
except csv.Error:
pass
first = next((line for line in sample.splitlines() if line.strip()), "")
counts = {delim: first.count(delim) for delim in _CSV_DELIMITERS}
best = max(counts, key=lambda delim: counts[delim])
return best if counts[best] else ","
def render_csv_table(raw: str, *, delimiter: str | None = None) -> str:
"""Render CSV text as the same HTML table shape the xlsx viewer consumes.
Row numbers replace the A1 column: a CSV has no fixed column count, so
the first row is a plain data row like the others (the viewer offers the
toolbar either way). Every cell is HTML-escaped at render time.
``delimiter`` defaults to the sniffed one (:func:`sniff_csv_delimiter`,
BUG-098): a `;`-separated file used to render as a single column.
"""
import csv as csv_mod
import io as io_mod
reader = csv_mod.reader(io_mod.StringIO(raw), delimiter=delimiter)
reader = csv_mod.reader(
io_mod.StringIO(raw), delimiter=delimiter or sniff_csv_delimiter(raw)
)
try:
rows = [row for row in reader]
except csv_mod.Error: