L'édition d'un .xlsx pouvait détruire une partie du classeur, le concurrencer en silence, ou diffuser une injection de formule. - BUG-085 : inspect_workbook() détecte ce qu'un round-trip openpyxl perd (valeurs calculées en cache, slicers, contrôles, connexions, custom XML, signature, commentaires enrichis, macros) → xlsx_lossy_features exposé en lecture, bandeau FR/EN, et 409 xlsx_lossy_content sans `force` (confirmation explicite puis reprise). Périmètre réel revalidé : graphiques, images et TCD survivent au round-trip. - BUG-086 : écriture atomique (fichier .tmp + os.replace) : un plantage ne peut plus tronquer le classeur, le backup reste intact. - BUG-087 : verrou par fichier autour du read-modify-write (timeout 15 s, 409 conflict) ; endpoint xlsx/save devenu synchrone pour que l'attente s'exécute dans le threadpool. - BUG-088 : une saisie en '=' ou '@' est stockée en texte, sauf opt-in `allow_formula` ou le bouton f(x) de la visionneuse. Le handler ServiceError expose désormais code + details, que api() propage. - BUG-084 : la suppression d'une vault purge enfin l'index inversé (documents fantômes qui continuaient de matcher) et is_stale() devient is_ready(), le nom étant trompeur (la staleness n'existe plus). Tests : 1390 pytest, 10 JSDOM (xlsx-viewer.test.mjs, branché au CI), 3 E2E Playwright, suite E2E complète verte, ruff/mypy 0. 🤖 Generated with Codebuff Co-Authored-By: Codebuff <[email protected]>
160 lines
5.7 KiB
Python
160 lines
5.7 KiB
Python
"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx).
|
|
|
|
Read-only: formulas are shown as their text (``data_only=False``) so a
|
|
round-trip through the viewer never depends on Excel's cached values.
|
|
Write-side lives in ``backend.services.mutations.edit_xlsx_cells``.
|
|
|
|
:func:`inspect_workbook` lists the workbook features that an openpyxl
|
|
round-trip would drop (#153 A1) so the UI can warn before saving.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import html
|
|
import re
|
|
import zipfile
|
|
from datetime import date, datetime
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from openpyxl import load_workbook
|
|
from openpyxl.utils import get_column_letter
|
|
|
|
# ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet).
|
|
# Raise them, or paginate per sheet, if a real workbook needs more.
|
|
MAX_ROWS = 500
|
|
MAX_COLS = 40
|
|
|
|
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
|
|
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
|
|
# DO survive the round-trip, so they are deliberately absent from this map.
|
|
LOSSY_PARTS: dict[str, tuple[str, ...]] = {
|
|
"slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"),
|
|
"form_controls": ("xl/ctrlProps/", "xl/activeX/"),
|
|
"connections": ("xl/queryTables/", "xl/connections.xml"),
|
|
"custom_xml": ("customXml/",),
|
|
"signature": ("_xmlsignatures/",),
|
|
"rich_comments": ("xl/threadedComments/", "xl/persons/"),
|
|
"macros": ("xl/vbaProject.bin",),
|
|
}
|
|
|
|
# A formula cell carrying its last computed result: ``<f>…</f><v>…</v>``.
|
|
# openpyxl writes an EMPTY ``<v></v>`` itself, hence the ``[^<]`` guard: only a
|
|
# non-empty value counts. openpyxl keeps the formula but drops the cached result,
|
|
# so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until
|
|
# Excel recalculates.
|
|
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
|
|
|
|
# Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES).
|
|
_MAX_PROBE_BYTES = 8_000_000
|
|
|
|
|
|
def _fmt(value: Any) -> str:
|
|
if value is None:
|
|
return ""
|
|
if isinstance(value, datetime):
|
|
return value.strftime("%Y-%m-%d %H:%M")
|
|
if isinstance(value, date):
|
|
return value.isoformat()
|
|
return str(value)
|
|
|
|
|
|
def _trim(grid: list[list[str]]) -> list[list[str]]:
|
|
"""Drop trailing empty rows and columns (openpyxl pads to max_col)."""
|
|
while grid and not any(grid[-1]):
|
|
grid.pop()
|
|
if not grid:
|
|
return grid
|
|
width = 0
|
|
for row in grid:
|
|
for i in range(len(row) - 1, -1, -1):
|
|
if row[i]:
|
|
width = max(width, i + 1)
|
|
break
|
|
return [row[:width] for row in grid]
|
|
|
|
|
|
def _table(grid: list[list[str]]) -> str:
|
|
if not grid:
|
|
return "<p><em>Feuille vide</em></p>"
|
|
n_cols = max(len(row) for row in grid)
|
|
out = [
|
|
(
|
|
'<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
|
'<thead><tr><th class="xlsx-corner"></th>'
|
|
)
|
|
]
|
|
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
|
out.append("</tr></thead><tbody>")
|
|
for r, row in enumerate(grid, start=1):
|
|
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
|
for c, val in enumerate(row, start=1):
|
|
ref = f"{get_column_letter(c)}{r}"
|
|
out.append(f'<td data-cell="{ref}">{html.escape(val)}</td>')
|
|
out.append("</tr>")
|
|
out.append("</tbody></table></div>")
|
|
return "".join(out)
|
|
|
|
|
|
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
|
|
"""True when at least one formula cell still carries its computed value."""
|
|
budget = _MAX_PROBE_BYTES
|
|
for name in zf.namelist():
|
|
if not name.startswith("xl/worksheets/sheet") or not name.endswith(".xml"):
|
|
continue
|
|
try:
|
|
with zf.open(name) as fh:
|
|
while budget > 0:
|
|
chunk = fh.read(65536)
|
|
if not chunk:
|
|
break
|
|
budget -= len(chunk)
|
|
if _CACHED_FORMULA_RE.search(chunk):
|
|
return True
|
|
except (KeyError, OSError, zipfile.BadZipFile):
|
|
continue
|
|
return False
|
|
|
|
|
|
def inspect_workbook(file_path: Path) -> list[str]:
|
|
"""Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*.
|
|
|
|
Read-only inspection of the OPC package (central directory + a bounded scan
|
|
of the sheet XML). Never raises: an unreadable or encrypted workbook simply
|
|
yields ``[]`` and the save path keeps its current behaviour.
|
|
|
|
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
|
|
the cached result, so the workbook stays correct once Excel recalculates it.
|
|
"""
|
|
try:
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
names = set(zf.namelist())
|
|
found = {
|
|
key
|
|
for key, prefixes in LOSSY_PARTS.items()
|
|
if any(name.startswith(prefix) for name in names for prefix in prefixes)
|
|
}
|
|
if _has_cached_formulas(zf):
|
|
found.add("cached_values")
|
|
return sorted(found)
|
|
except (OSError, zipfile.BadZipFile):
|
|
return []
|
|
|
|
|
|
def render_sheets(file_path: Path) -> list[dict[str, str]]:
|
|
"""Return ``[{"name": sheet_title, "html": table_html}, ...]``."""
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
try:
|
|
sheets = []
|
|
for ws in wb.worksheets:
|
|
grid = [
|
|
[_fmt(v) for v in row]
|
|
for row in ws.iter_rows(
|
|
min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS, values_only=True
|
|
)
|
|
]
|
|
sheets.append({"name": ws.title, "html": _table(_trim(grid))})
|
|
return sheets
|
|
finally:
|
|
wb.close()
|