- BUG-090 (#153 A8) : render_sheets() expose total_rows/total_cols, max_rows/max_cols et truncated ; la visionneuse affiche un bandeau « Feuille tronquée » (i18n FR/EN) au lieu de couper en silence, et la ligne d'en-têtes devient sticky (top:auto sur les numéros de ligne). - #153 A9 : GET /api/file/{vault}/xlsx/sheet?sheet&offset&limit sert une fenêtre de 1 à 1000 lignes avec les vraies coordonnées A1, has_more de pagination et valeurs calculées A12 ; 404 feuille inconnue, 415 non-xlsx. - Tests : TestXlsxTruncationNotice (4) + TestXlsxSheetWindow (11) avec contre-preuves, xlsx-viewer.test.mjs 14/14, E2E 7/7 (fixture sample-xlsx-large.xlsx 520 lignes), suite 1417 passed / 6 skipped, ruff/mypy 0, i18n parity. 🤖 Generated with Codebuff Co-Authored-By: Codebuff <[email protected]>
447 lines
17 KiB
Python
447 lines
17 KiB
Python
"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx).
|
|
|
|
Read-only: formulas are shown as their text (``data_only=False``) so a
|
|
round-trip through the viewer never depends on Excel's cached values.
|
|
Write-side lives in ``backend.services.mutations.edit_xlsx_cells``.
|
|
|
|
:func:`inspect_workbook` lists the workbook features that an openpyxl
|
|
round-trip would drop (#153 A1) so the UI can warn before saving.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import html
|
|
import logging
|
|
import re
|
|
import zipfile
|
|
from datetime import date, datetime
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from openpyxl import load_workbook
|
|
from openpyxl.utils import get_column_letter
|
|
|
|
logger = logging.getLogger("obsigate.xlsx_reader")
|
|
|
|
# ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet).
|
|
# Raise them, or paginate per sheet, if a real workbook needs more.
|
|
MAX_ROWS = 500
|
|
MAX_COLS = 40
|
|
|
|
# #153 A9 — window size served by ``read_sheet_window()`` (lazy per-sheet
|
|
# loading). The endpoint is bounded so a single request can never ask for the
|
|
# whole workbook back in one JSON payload; the UI pages through the rest.
|
|
MAX_WINDOW_ROWS = 1_000
|
|
DEFAULT_WINDOW_ROWS = 200
|
|
|
|
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
|
|
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
|
|
# DO survive the round-trip, so they are deliberately absent from this map.
|
|
LOSSY_PARTS: dict[str, tuple[str, ...]] = {
|
|
"slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"),
|
|
"form_controls": ("xl/ctrlProps/", "xl/activeX/"),
|
|
"connections": ("xl/queryTables/", "xl/connections.xml"),
|
|
"custom_xml": ("customXml/",),
|
|
"signature": ("_xmlsignatures/",),
|
|
"rich_comments": ("xl/threadedComments/", "xl/persons/"),
|
|
"macros": ("xl/vbaProject.bin",),
|
|
}
|
|
|
|
# A formula cell carrying its last computed result: ``<f>…</f><v>…</v>``.
|
|
# openpyxl writes an EMPTY ``<v></v>`` itself, hence the ``[^<]`` guard: only a
|
|
# non-empty value counts. openpyxl keeps the formula but drops the cached result,
|
|
# so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until
|
|
# Excel recalculates.
|
|
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
|
|
|
|
# Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES).
|
|
_MAX_PROBE_BYTES = 8_000_000
|
|
|
|
# #153 A5 — ceiling on the text handed to the TF-IDF / semantic index. A workbook
|
|
# is a data dump, not prose: indexing every cell would flood the inverted index
|
|
# and bury the notes. Sheet names + the first rows are enough to make a
|
|
# spreadsheet findable by its headers.
|
|
MAX_INDEX_CHARS = 5_000
|
|
_INDEX_ROWS_PER_SHEET = 20
|
|
MAX_INDEX_SHEETS = 20
|
|
|
|
|
|
def _fmt(value: Any) -> str:
|
|
if value is None:
|
|
return ""
|
|
if isinstance(value, datetime):
|
|
return value.strftime("%Y-%m-%d %H:%M")
|
|
if isinstance(value, date):
|
|
return value.isoformat()
|
|
return str(value)
|
|
|
|
|
|
def _trim(grid: list[list[str]]) -> list[list[str]]:
|
|
"""Drop trailing empty rows and columns (openpyxl pads to max_col)."""
|
|
while grid and not any(grid[-1]):
|
|
grid.pop()
|
|
if not grid:
|
|
return grid
|
|
width = 0
|
|
for row in grid:
|
|
for i in range(len(row) - 1, -1, -1):
|
|
if row[i]:
|
|
width = max(width, i + 1)
|
|
break
|
|
return [row[:width] for row in grid]
|
|
|
|
|
|
def _cell_cached(cached: list[list[str]] | None, r: int, c: int) -> str:
|
|
"""Return the cached result for a 0-based cell, or ``""``.
|
|
|
|
The shadow grid is read positionally and may be narrower than the formula
|
|
grid (``_trim`` collapses the trailing empty columns of each grid
|
|
independently), so every lookup is bounds-checked rather than assumed.
|
|
"""
|
|
if not cached or r >= len(cached):
|
|
return ""
|
|
row = cached[r]
|
|
return row[c] if c < len(row) else ""
|
|
|
|
|
|
def _table(
|
|
grid: list[list[str]],
|
|
cached: list[list[str]] | None = None,
|
|
row_offset: int = 0,
|
|
) -> str:
|
|
"""Render a grid as an HTML table.
|
|
|
|
``cached`` is the same grid read with ``data_only=True`` (#153 A12): where a
|
|
formula cell still carries its last computed result, it is shown as a
|
|
discreet second line (``<span class="xlsx-cached">``) so the user sees the
|
|
number Excel last calculated instead of only the formula text. The span
|
|
carries ``data-cached-value`` and is titled client-side from
|
|
``xlsx.cached_value_title`` — the backend never emits UI text.
|
|
|
|
``row_offset`` is the number of rows skipped before this grid (#153 A9): the
|
|
row numbers and the ``data-cell`` references must stay the real A1
|
|
coordinates of the sheet, not of the window.
|
|
"""
|
|
if not grid:
|
|
return "<p><em>Feuille vide</em></p>"
|
|
n_cols = max(len(row) for row in grid)
|
|
out = [
|
|
(
|
|
'<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
|
'<thead><tr><th class="xlsx-corner"></th>'
|
|
)
|
|
]
|
|
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
|
out.append("</tr></thead><tbody>")
|
|
for r, row in enumerate(grid, start=row_offset + 1):
|
|
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
|
for c, val in enumerate(row, start=1):
|
|
ref = f"{get_column_letter(c)}{r}"
|
|
# The cached result only makes sense for a formula cell: on a plain
|
|
# value cell the two reads are identical and showing both would
|
|
# duplicate the text.
|
|
shadow = ""
|
|
if cached is not None and val.startswith("="):
|
|
# `c` is 1-based (A1 notation) and `r` too, while the grid is
|
|
# 0-based: translate both.
|
|
cval = _cell_cached(cached, r - 1, c - 1)
|
|
if cval and cval != val:
|
|
# The tooltip is translated client-side from
|
|
# `xlsx.cached_value_title`; never hardcode UI text here.
|
|
shadow = (
|
|
f'<span class="xlsx-cached" data-cached-value="1">'
|
|
f"{html.escape(cval)}</span>"
|
|
)
|
|
out.append(
|
|
f'<td data-cell="{ref}">{html.escape(val)}{shadow}</td>'
|
|
)
|
|
out.append("</tr>")
|
|
out.append("</tbody></table></div>")
|
|
return "".join(out)
|
|
|
|
|
|
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
|
|
"""True when at least one formula cell still carries its computed value."""
|
|
budget = _MAX_PROBE_BYTES
|
|
for name in zf.namelist():
|
|
if not name.startswith("xl/worksheets/sheet") or not name.endswith(".xml"):
|
|
continue
|
|
try:
|
|
with zf.open(name) as fh:
|
|
while budget > 0:
|
|
chunk = fh.read(65536)
|
|
if not chunk:
|
|
break
|
|
budget -= len(chunk)
|
|
if _CACHED_FORMULA_RE.search(chunk):
|
|
return True
|
|
except (KeyError, OSError, zipfile.BadZipFile):
|
|
continue
|
|
return False
|
|
|
|
|
|
def inspect_workbook(file_path: Path) -> list[str]:
|
|
"""Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*.
|
|
|
|
Read-only inspection of the OPC package (central directory + a bounded scan
|
|
of the sheet XML). Never raises: an unreadable or encrypted workbook simply
|
|
yields ``[]`` and the save path keeps its current behaviour.
|
|
|
|
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
|
|
the cached result, so the workbook stays correct once Excel recalculates it.
|
|
"""
|
|
try:
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
names = set(zf.namelist())
|
|
found = {
|
|
key
|
|
for key, prefixes in LOSSY_PARTS.items()
|
|
if any(name.startswith(prefix) for name in names for prefix in prefixes)
|
|
}
|
|
if _has_cached_formulas(zf):
|
|
found.add("cached_values")
|
|
return sorted(found)
|
|
except (OSError, zipfile.BadZipFile):
|
|
return []
|
|
|
|
|
|
def render_sheets(file_path: Path) -> list[dict[str, Any]]:
|
|
"""Return one dict per sheet: ``{name, html, rows, cols, total_*, truncated}``.
|
|
|
|
Reads the workbook twice: once with ``data_only=False`` for the formulas
|
|
(what the user must edit) and, when any formula carries a cached result
|
|
(#153 A12), once with ``data_only=True`` to show what Excel last computed.
|
|
The second pass is skipped entirely when the archive holds no cached value,
|
|
so the common case still costs a single load.
|
|
|
|
``total_rows``/``total_cols`` are the dimensions the sheet declares and
|
|
``truncated`` says whether the hard caps actually cut it (#153 A8) — the
|
|
viewer needs both to stop silently hiding the tail of a sheet.
|
|
"""
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
try:
|
|
formulas = [_sheet_grid(ws) for ws in wb.worksheets]
|
|
titles = [ws.title for ws in wb.worksheets]
|
|
extents = [_sheet_extent(ws) for ws in wb.worksheets]
|
|
finally:
|
|
wb.close()
|
|
|
|
cached: list[list[list[str]]] | None = None
|
|
if _has_cached_values(file_path):
|
|
cached = _read_cached_grids(file_path, titles)
|
|
|
|
sheets = []
|
|
for i, title in enumerate(titles):
|
|
grid = _trim(formulas[i])
|
|
# The shadow grid is NOT trimmed independently: _trim drops the
|
|
# trailing empty columns of each grid on its own width, which would
|
|
# shift every cached value left of its formula. Indexing it
|
|
# positionally against the untrimmed grid keeps the two aligned.
|
|
shadow = cached[i] if cached is not None and i < len(cached) else None
|
|
total_rows, total_cols = extents[i]
|
|
sheets.append(
|
|
{
|
|
"name": title,
|
|
"html": _table(grid, shadow),
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": total_rows,
|
|
"total_cols": total_cols,
|
|
# Coverage, not display size: `rows`/`cols` are post-trim (a
|
|
# sheet of 3 filled cells in a 500-row block renders 1x1), and
|
|
# the client must announce the cap it stopped at, not how many
|
|
# cells happen to be non-empty.
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
# A sheet is truncated when the caps, not the trailing blanks,
|
|
# decided its shape: comparing against the *rendered* size would
|
|
# flag every sheet carrying a few empty formatted rows.
|
|
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
|
}
|
|
)
|
|
return sheets
|
|
|
|
|
|
def read_sheet_window(
|
|
file_path: Path,
|
|
sheet: str,
|
|
offset: int = 0,
|
|
limit: int = DEFAULT_WINDOW_ROWS,
|
|
) -> dict[str, Any] | None:
|
|
"""Return a window of rows of one sheet, or ``None`` if the sheet is unknown.
|
|
|
|
Backs the lazy per-sheet loading of #153 A9: the viewer asks for the rows
|
|
it is about to display instead of shipping every sheet in the initial file
|
|
payload. ``offset`` is 0-based; the row numbers and the ``data-cell``
|
|
references in the returned ``html`` are the real A1 coordinates of the
|
|
sheet, so a window is indistinguishable from a full render.
|
|
|
|
``limit`` is clamped to :data:`MAX_WINDOW_ROWS`. Raises nothing: an unknown
|
|
sheet yields ``None`` and a broken workbook propagates the caller's usual
|
|
500.
|
|
"""
|
|
offset = max(int(offset), 0)
|
|
limit = min(max(int(limit), 1), MAX_WINDOW_ROWS)
|
|
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return None
|
|
ws = wb[sheet]
|
|
total_rows, total_cols = _sheet_extent(ws)
|
|
grid = _trim(
|
|
_sheet_grid(ws, min_row=offset + 1, max_row=offset + limit)
|
|
)
|
|
finally:
|
|
wb.close()
|
|
|
|
shadow: list[list[str]] | None = None
|
|
# Same A12 rule as the full render: the second read only happens when the
|
|
# archive really holds cached results.
|
|
if _has_cached_values(file_path):
|
|
shadow = _read_cached_window(file_path, sheet, offset, limit)
|
|
return {
|
|
"sheet": sheet,
|
|
"offset": offset,
|
|
"limit": limit,
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": total_rows,
|
|
"total_cols": total_cols,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
|
"has_more": offset + len(grid) < total_rows,
|
|
"html": _table(grid, shadow, row_offset=offset),
|
|
}
|
|
|
|
|
|
def _read_cached_window(
|
|
file_path: Path, sheet: str, offset: int, limit: int
|
|
) -> list[list[str]] | None:
|
|
"""``data_only=True`` grid for one window, or ``None`` if unavailable.
|
|
|
|
Best effort like :func:`_read_cached_grids`: a workbook Excel opens but
|
|
openpyxl cannot re-read must still display (formulas only).
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
return None
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return None
|
|
return _sheet_grid(
|
|
wb[sheet], min_row=offset + 1, max_row=offset + limit
|
|
)
|
|
except Exception:
|
|
logger.debug("xlsx cached window unavailable", exc_info=True)
|
|
return None
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def _sheet_extent(ws: Any) -> tuple[int, int]:
|
|
"""Rows and columns the worksheet declares, never negative.
|
|
|
|
``max_row``/``max_column`` come from the sheet's dimension record; a
|
|
hand-edited file may omit it, hence the defensive coercion.
|
|
"""
|
|
try:
|
|
rows = max(int(getattr(ws, "max_row", 0) or 0), 0)
|
|
except (TypeError, ValueError):
|
|
rows = 0
|
|
try:
|
|
cols = max(int(getattr(ws, "max_column", 0) or 0), 0)
|
|
except (TypeError, ValueError):
|
|
cols = 0
|
|
return rows, cols
|
|
|
|
|
|
def _sheet_grid(
|
|
ws: Any, min_row: int = 1, max_row: int = MAX_ROWS, max_col: int = MAX_COLS
|
|
) -> list[list[str]]:
|
|
"""Read a worksheet window into a grid of formatted strings, bounded by the caps."""
|
|
return [
|
|
[_fmt(v) for v in row]
|
|
for row in ws.iter_rows(
|
|
min_row=min_row, max_row=max_row, max_col=max_col, values_only=True
|
|
)
|
|
]
|
|
|
|
|
|
def _has_cached_values(file_path: Path) -> bool:
|
|
"""True when the archive holds at least one ``<f>…</f><v>…</v>``."""
|
|
try:
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
return _has_cached_formulas(zf)
|
|
except (OSError, zipfile.BadZipFile):
|
|
return False
|
|
|
|
|
|
def _read_cached_grids(
|
|
file_path: Path, titles: list[str]
|
|
) -> list[list[list[str]]] | None:
|
|
"""Read every sheet with ``data_only=True`` (what Excel last computed).
|
|
|
|
Best effort: returns ``None`` on any failure so the viewer falls back to the
|
|
formula-only rendering. A workbook Excel opens but openpyxl cannot re-read
|
|
must still display.
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
return None
|
|
try:
|
|
grids = [_sheet_grid(ws) for ws in wb.worksheets]
|
|
if [ws.title for ws in wb.worksheets] != titles:
|
|
return None
|
|
return grids
|
|
except Exception:
|
|
logger.debug("xlsx cached values unavailable", exc_info=True)
|
|
return None
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def extract_indexable_text(file_path: Path) -> str:
|
|
"""Return searchable text for the TF-IDF / semantic index (#153 A5).
|
|
|
|
Sheet names plus the first :data:`_INDEX_ROWS_PER_SHEET` rows of each
|
|
sheet, capped at :data:`MAX_INDEX_CHARS`. Rows are tab-joined so a search
|
|
for a header matches the sheet it belongs to.
|
|
|
|
Never raises: a corrupt, encrypted or unsupported workbook yields ``""`` so
|
|
the file still gets indexed by name (same contract as :func:`inspect_workbook`).
|
|
"""
|
|
chunks: list[str] = []
|
|
budget = MAX_INDEX_CHARS
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
# Encrypted (BadZipFile) or not a real workbook: name-only indexing.
|
|
return ""
|
|
try:
|
|
for ws in wb.worksheets[:MAX_INDEX_SHEETS]:
|
|
if budget <= 0:
|
|
break
|
|
# The sheet title alone is a strong signal ("Recettes", "Budget").
|
|
block = [ws.title]
|
|
for row in ws.iter_rows(
|
|
min_row=1, max_row=_INDEX_ROWS_PER_SHEET, max_col=MAX_COLS, values_only=True
|
|
):
|
|
cells = [_fmt(v) for v in row]
|
|
# Skip blank rows instead of emitting runs of tabs.
|
|
if not any(c.strip() for c in cells):
|
|
continue
|
|
block.append("\t".join(cells).rstrip())
|
|
text = "\n".join(block)
|
|
chunks.append(text[:budget])
|
|
budget -= len(text)
|
|
except Exception:
|
|
# Truncated but still useful: keep whatever was collected.
|
|
pass
|
|
finally:
|
|
wb.close()
|
|
return "\n".join(c for c in chunks if c).strip()
|