Files
ObsiGate/backend/xlsx_reader.py
T
bruno 31d4616baf feat: garde-fous d'écriture des classeurs Excel #153 (P0)
L'édition d'un .xlsx pouvait détruire une partie du classeur, le
concurrencer en silence, ou diffuser une injection de formule.

- BUG-085 : inspect_workbook() détecte ce qu'un round-trip openpyxl perd
  (valeurs calculées en cache, slicers, contrôles, connexions, custom
  XML, signature, commentaires enrichis, macros) → xlsx_lossy_features
  exposé en lecture, bandeau FR/EN, et 409 xlsx_lossy_content sans
  `force` (confirmation explicite puis reprise). Périmètre réel
  revalidé : graphiques, images et TCD survivent au round-trip.
- BUG-086 : écriture atomique (fichier .tmp + os.replace) : un plantage
  ne peut plus tronquer le classeur, le backup reste intact.
- BUG-087 : verrou par fichier autour du read-modify-write (timeout 15 s,
  409 conflict) ; endpoint xlsx/save devenu synchrone pour que
  l'attente s'exécute dans le threadpool.
- BUG-088 : une saisie en '=' ou '@' est stockée en texte, sauf opt-in
  `allow_formula` ou le bouton f(x) de la visionneuse. Le handler
  ServiceError expose désormais code + details, que api() propage.
- BUG-084 : la suppression d'une vault purge enfin l'index inversé
  (documents fantômes qui continuaient de matcher) et is_stale() devient
  is_ready(), le nom étant trompeur (la staleness n'existe plus).

Tests : 1390 pytest, 10 JSDOM (xlsx-viewer.test.mjs, branché au CI),
3 E2E Playwright, suite E2E complète verte, ruff/mypy 0.

🤖 Generated with Codebuff
Co-Authored-By: Codebuff <[email protected]>
2026-09-27 20:39:12 -04:00

160 lines
5.7 KiB
Python

"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx).
Read-only: formulas are shown as their text (``data_only=False``) so a
round-trip through the viewer never depends on Excel's cached values.
Write-side lives in ``backend.services.mutations.edit_xlsx_cells``.
:func:`inspect_workbook` lists the workbook features that an openpyxl
round-trip would drop (#153 A1) so the UI can warn before saving.
"""
from __future__ import annotations
import html
import re
import zipfile
from datetime import date, datetime
from pathlib import Path
from typing import Any
from openpyxl import load_workbook
from openpyxl.utils import get_column_letter
# ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet).
# Raise them, or paginate per sheet, if a real workbook needs more.
MAX_ROWS = 500
MAX_COLS = 40
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
# DO survive the round-trip, so they are deliberately absent from this map.
LOSSY_PARTS: dict[str, tuple[str, ...]] = {
"slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"),
"form_controls": ("xl/ctrlProps/", "xl/activeX/"),
"connections": ("xl/queryTables/", "xl/connections.xml"),
"custom_xml": ("customXml/",),
"signature": ("_xmlsignatures/",),
"rich_comments": ("xl/threadedComments/", "xl/persons/"),
"macros": ("xl/vbaProject.bin",),
}
# A formula cell carrying its last computed result: ``<f>…</f><v>…</v>``.
# openpyxl writes an EMPTY ``<v></v>`` itself, hence the ``[^<]`` guard: only a
# non-empty value counts. openpyxl keeps the formula but drops the cached result,
# so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until
# Excel recalculates.
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
# Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES).
_MAX_PROBE_BYTES = 8_000_000
def _fmt(value: Any) -> str:
if value is None:
return ""
if isinstance(value, datetime):
return value.strftime("%Y-%m-%d %H:%M")
if isinstance(value, date):
return value.isoformat()
return str(value)
def _trim(grid: list[list[str]]) -> list[list[str]]:
"""Drop trailing empty rows and columns (openpyxl pads to max_col)."""
while grid and not any(grid[-1]):
grid.pop()
if not grid:
return grid
width = 0
for row in grid:
for i in range(len(row) - 1, -1, -1):
if row[i]:
width = max(width, i + 1)
break
return [row[:width] for row in grid]
def _table(grid: list[list[str]]) -> str:
if not grid:
return "<p><em>Feuille vide</em></p>"
n_cols = max(len(row) for row in grid)
out = [
(
'<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
'<thead><tr><th class="xlsx-corner"></th>'
)
]
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
out.append("</tr></thead><tbody>")
for r, row in enumerate(grid, start=1):
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
for c, val in enumerate(row, start=1):
ref = f"{get_column_letter(c)}{r}"
out.append(f'<td data-cell="{ref}">{html.escape(val)}</td>')
out.append("</tr>")
out.append("</tbody></table></div>")
return "".join(out)
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
"""True when at least one formula cell still carries its computed value."""
budget = _MAX_PROBE_BYTES
for name in zf.namelist():
if not name.startswith("xl/worksheets/sheet") or not name.endswith(".xml"):
continue
try:
with zf.open(name) as fh:
while budget > 0:
chunk = fh.read(65536)
if not chunk:
break
budget -= len(chunk)
if _CACHED_FORMULA_RE.search(chunk):
return True
except (KeyError, OSError, zipfile.BadZipFile):
continue
return False
def inspect_workbook(file_path: Path) -> list[str]:
"""Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*.
Read-only inspection of the OPC package (central directory + a bounded scan
of the sheet XML). Never raises: an unreadable or encrypted workbook simply
yields ``[]`` and the save path keeps its current behaviour.
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
the cached result, so the workbook stays correct once Excel recalculates it.
"""
try:
with zipfile.ZipFile(file_path) as zf:
names = set(zf.namelist())
found = {
key
for key, prefixes in LOSSY_PARTS.items()
if any(name.startswith(prefix) for name in names for prefix in prefixes)
}
if _has_cached_formulas(zf):
found.add("cached_values")
return sorted(found)
except (OSError, zipfile.BadZipFile):
return []
def render_sheets(file_path: Path) -> list[dict[str, str]]:
"""Return ``[{"name": sheet_title, "html": table_html}, ...]``."""
wb = load_workbook(str(file_path), read_only=True, data_only=False)
try:
sheets = []
for ws in wb.worksheets:
grid = [
[_fmt(v) for v in row]
for row in ws.iter_rows(
min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS, values_only=True
)
]
sheets.append({"name": ws.title, "html": _table(_trim(grid))})
return sheets
finally:
wb.close()