"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx). Read-only: formulas are shown as their text (``data_only=False``) so a round-trip through the viewer never depends on Excel's cached values. Write-side lives in ``backend.services.mutations.edit_xlsx_cells``. :func:`inspect_workbook` lists the workbook features that an openpyxl round-trip would drop (#153 A1) so the UI can warn before saving. """ from __future__ import annotations import html import logging import re import zipfile from datetime import date, datetime from pathlib import Path from typing import Any from openpyxl import load_workbook from openpyxl.utils import get_column_letter logger = logging.getLogger("obsigate.xlsx_reader") # ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet). # Raise them, or paginate per sheet, if a real workbook needs more. MAX_ROWS = 500 MAX_COLS = 40 # #153 A9 — window size served by ``read_sheet_window()`` (lazy per-sheet # loading). The endpoint is bounded so a single request can never ask for the # whole workbook back in one JSON payload; the UI pages through the rest. MAX_WINDOW_ROWS = 1_000 DEFAULT_WINDOW_ROWS = 200 # #153 A1 — workbook parts openpyxl does not re-serialize on load+save. # Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables # DO survive the round-trip, so they are deliberately absent from this map. LOSSY_PARTS: dict[str, tuple[str, ...]] = { "slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"), "form_controls": ("xl/ctrlProps/", "xl/activeX/"), "connections": ("xl/queryTables/", "xl/connections.xml"), "custom_xml": ("customXml/",), "signature": ("_xmlsignatures/",), "rich_comments": ("xl/threadedComments/", "xl/persons/"), "macros": ("xl/vbaProject.bin",), } # A formula cell carrying its last computed result: ``……``. # openpyxl writes an EMPTY ```` itself, hence the ``[^<]`` guard: only a # non-empty value counts. openpyxl keeps the formula but drops the cached result, # so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until # Excel recalculates. _CACHED_FORMULA_RE = re.compile(rb"][^<]*\s*[^<]") # Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES). _MAX_PROBE_BYTES = 8_000_000 # #153 A5 — ceiling on the text handed to the TF-IDF / semantic index. A workbook # is a data dump, not prose: indexing every cell would flood the inverted index # and bury the notes. Sheet names + the first rows are enough to make a # spreadsheet findable by its headers. MAX_INDEX_CHARS = 5_000 _INDEX_ROWS_PER_SHEET = 20 MAX_INDEX_SHEETS = 20 def _fmt(value: Any) -> str: if value is None: return "" if isinstance(value, datetime): return value.strftime("%Y-%m-%d %H:%M") if isinstance(value, date): return value.isoformat() return str(value) def _trim(grid: list[list[str]]) -> list[list[str]]: """Drop trailing empty rows and columns (openpyxl pads to max_col).""" while grid and not any(grid[-1]): grid.pop() if not grid: return grid width = 0 for row in grid: for i in range(len(row) - 1, -1, -1): if row[i]: width = max(width, i + 1) break return [row[:width] for row in grid] def _cell_cached(cached: list[list[str]] | None, r: int, c: int) -> str: """Return the cached result for a 0-based cell, or ``""``. The shadow grid is read positionally and may be narrower than the formula grid (``_trim`` collapses the trailing empty columns of each grid independently), so every lookup is bounds-checked rather than assumed. """ if not cached or r >= len(cached): return "" row = cached[r] return row[c] if c < len(row) else "" def _table( grid: list[list[str]], cached: list[list[str]] | None = None, row_offset: int = 0, ) -> str: """Render a grid as an HTML table. ``cached`` is the same grid read with ``data_only=True`` (#153 A12): where a formula cell still carries its last computed result, it is shown as a discreet second line (````) so the user sees the number Excel last calculated instead of only the formula text. The span carries ``data-cached-value`` and is titled client-side from ``xlsx.cached_value_title`` — the backend never emits UI text. ``row_offset`` is the number of rows skipped before this grid (#153 A9): the row numbers and the ``data-cell`` references must stay the real A1 coordinates of the sheet, not of the window. """ if not grid: return "

Feuille vide

" n_cols = max(len(row) for row in grid) out = [ ( '
' '' ) ] out += [f"" for c in range(1, n_cols + 1)] out.append("") for r, row in enumerate(grid, start=row_offset + 1): out.append(f'') for c, val in enumerate(row, start=1): ref = f"{get_column_letter(c)}{r}" # The cached result only makes sense for a formula cell: on a plain # value cell the two reads are identical and showing both would # duplicate the text. shadow = "" if cached is not None and val.startswith("="): # `c` is 1-based (A1 notation) and `r` too, while the grid is # 0-based: translate both. cval = _cell_cached(cached, r - 1, c - 1) if cval and cval != val: # The tooltip is translated client-side from # `xlsx.cached_value_title`; never hardcode UI text here. shadow = ( f'' f"{html.escape(cval)}" ) out.append( f'' ) out.append("") out.append("
{get_column_letter(c)}
{r}{html.escape(val)}{shadow}
") return "".join(out) def _has_cached_formulas(zf: zipfile.ZipFile) -> bool: """True when at least one formula cell still carries its computed value.""" budget = _MAX_PROBE_BYTES for name in zf.namelist(): if not name.startswith("xl/worksheets/sheet") or not name.endswith(".xml"): continue try: with zf.open(name) as fh: while budget > 0: chunk = fh.read(65536) if not chunk: break budget -= len(chunk) if _CACHED_FORMULA_RE.search(chunk): return True except (KeyError, OSError, zipfile.BadZipFile): continue return False def inspect_workbook(file_path: Path) -> list[str]: """Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*. Read-only inspection of the OPC package (central directory + a bounded scan of the sheet XML). Never raises: an unreadable or encrypted workbook simply yields ``[]`` and the save path keeps its current behaviour. ``cached_values`` is a synthetic key: openpyxl keeps the formula but drops the cached result, so the workbook stays correct once Excel recalculates it. """ try: with zipfile.ZipFile(file_path) as zf: names = set(zf.namelist()) found = { key for key, prefixes in LOSSY_PARTS.items() if any(name.startswith(prefix) for name in names for prefix in prefixes) } if _has_cached_formulas(zf): found.add("cached_values") return sorted(found) except (OSError, zipfile.BadZipFile): return [] def render_sheets(file_path: Path) -> list[dict[str, Any]]: """Return one dict per sheet: ``{name, html, rows, cols, total_*, truncated}``. Reads the workbook twice: once with ``data_only=False`` for the formulas (what the user must edit) and, when any formula carries a cached result (#153 A12), once with ``data_only=True`` to show what Excel last computed. The second pass is skipped entirely when the archive holds no cached value, so the common case still costs a single load. ``total_rows``/``total_cols`` are the dimensions the sheet declares and ``truncated`` says whether the hard caps actually cut it (#153 A8) — the viewer needs both to stop silently hiding the tail of a sheet. """ wb = load_workbook(str(file_path), read_only=True, data_only=False) try: formulas = [_sheet_grid(ws) for ws in wb.worksheets] titles = [ws.title for ws in wb.worksheets] extents = [_sheet_extent(ws) for ws in wb.worksheets] finally: wb.close() cached: list[list[list[str]]] | None = None if _has_cached_values(file_path): cached = _read_cached_grids(file_path, titles) sheets = [] for i, title in enumerate(titles): grid = _trim(formulas[i]) # The shadow grid is NOT trimmed independently: _trim drops the # trailing empty columns of each grid on its own width, which would # shift every cached value left of its formula. Indexing it # positionally against the untrimmed grid keeps the two aligned. shadow = cached[i] if cached is not None and i < len(cached) else None total_rows, total_cols = extents[i] sheets.append( { "name": title, "html": _table(grid, shadow), "rows": len(grid), "cols": max((len(r) for r in grid), default=0), "total_rows": total_rows, "total_cols": total_cols, # Coverage, not display size: `rows`/`cols` are post-trim (a # sheet of 3 filled cells in a 500-row block renders 1x1), and # the client must announce the cap it stopped at, not how many # cells happen to be non-empty. "max_rows": MAX_ROWS, "max_cols": MAX_COLS, # A sheet is truncated when the caps, not the trailing blanks, # decided its shape: comparing against the *rendered* size would # flag every sheet carrying a few empty formatted rows. "truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS, } ) return sheets def read_sheet_window( file_path: Path, sheet: str, offset: int = 0, limit: int = DEFAULT_WINDOW_ROWS, ) -> dict[str, Any] | None: """Return a window of rows of one sheet, or ``None`` if the sheet is unknown. Backs the lazy per-sheet loading of #153 A9: the viewer asks for the rows it is about to display instead of shipping every sheet in the initial file payload. ``offset`` is 0-based; the row numbers and the ``data-cell`` references in the returned ``html`` are the real A1 coordinates of the sheet, so a window is indistinguishable from a full render. ``limit`` is clamped to :data:`MAX_WINDOW_ROWS`. Raises nothing: an unknown sheet yields ``None`` and a broken workbook propagates the caller's usual 500. """ offset = max(int(offset), 0) limit = min(max(int(limit), 1), MAX_WINDOW_ROWS) wb = load_workbook(str(file_path), read_only=True, data_only=False) try: if sheet not in wb.sheetnames: return None ws = wb[sheet] total_rows, total_cols = _sheet_extent(ws) grid = _trim( _sheet_grid(ws, min_row=offset + 1, max_row=offset + limit) ) finally: wb.close() shadow: list[list[str]] | None = None # Same A12 rule as the full render: the second read only happens when the # archive really holds cached results. if _has_cached_values(file_path): shadow = _read_cached_window(file_path, sheet, offset, limit) return { "sheet": sheet, "offset": offset, "limit": limit, "rows": len(grid), "cols": max((len(r) for r in grid), default=0), "total_rows": total_rows, "total_cols": total_cols, "max_rows": MAX_ROWS, "max_cols": MAX_COLS, "truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS, "has_more": offset + len(grid) < total_rows, "html": _table(grid, shadow, row_offset=offset), } def _read_cached_window( file_path: Path, sheet: str, offset: int, limit: int ) -> list[list[str]] | None: """``data_only=True`` grid for one window, or ``None`` if unavailable. Best effort like :func:`_read_cached_grids`: a workbook Excel opens but openpyxl cannot re-read must still display (formulas only). """ try: wb = load_workbook(str(file_path), read_only=True, data_only=True) except Exception: return None try: if sheet not in wb.sheetnames: return None return _sheet_grid( wb[sheet], min_row=offset + 1, max_row=offset + limit ) except Exception: logger.debug("xlsx cached window unavailable", exc_info=True) return None finally: wb.close() def _sheet_extent(ws: Any) -> tuple[int, int]: """Rows and columns the worksheet declares, never negative. ``max_row``/``max_column`` come from the sheet's dimension record; a hand-edited file may omit it, hence the defensive coercion. """ try: rows = max(int(getattr(ws, "max_row", 0) or 0), 0) except (TypeError, ValueError): rows = 0 try: cols = max(int(getattr(ws, "max_column", 0) or 0), 0) except (TypeError, ValueError): cols = 0 return rows, cols def _sheet_grid( ws: Any, min_row: int = 1, max_row: int = MAX_ROWS, max_col: int = MAX_COLS ) -> list[list[str]]: """Read a worksheet window into a grid of formatted strings, bounded by the caps.""" return [ [_fmt(v) for v in row] for row in ws.iter_rows( min_row=min_row, max_row=max_row, max_col=max_col, values_only=True ) ] def _has_cached_values(file_path: Path) -> bool: """True when the archive holds at least one ``……``.""" try: with zipfile.ZipFile(file_path) as zf: return _has_cached_formulas(zf) except (OSError, zipfile.BadZipFile): return False def _read_cached_grids( file_path: Path, titles: list[str] ) -> list[list[list[str]]] | None: """Read every sheet with ``data_only=True`` (what Excel last computed). Best effort: returns ``None`` on any failure so the viewer falls back to the formula-only rendering. A workbook Excel opens but openpyxl cannot re-read must still display. """ try: wb = load_workbook(str(file_path), read_only=True, data_only=True) except Exception: return None try: grids = [_sheet_grid(ws) for ws in wb.worksheets] if [ws.title for ws in wb.worksheets] != titles: return None return grids except Exception: logger.debug("xlsx cached values unavailable", exc_info=True) return None finally: wb.close() def extract_indexable_text(file_path: Path) -> str: """Return searchable text for the TF-IDF / semantic index (#153 A5). Sheet names plus the first :data:`_INDEX_ROWS_PER_SHEET` rows of each sheet, capped at :data:`MAX_INDEX_CHARS`. Rows are tab-joined so a search for a header matches the sheet it belongs to. Never raises: a corrupt, encrypted or unsupported workbook yields ``""`` so the file still gets indexed by name (same contract as :func:`inspect_workbook`). """ chunks: list[str] = [] budget = MAX_INDEX_CHARS try: wb = load_workbook(str(file_path), read_only=True, data_only=True) except Exception: # Encrypted (BadZipFile) or not a real workbook: name-only indexing. return "" try: for ws in wb.worksheets[:MAX_INDEX_SHEETS]: if budget <= 0: break # The sheet title alone is a strong signal ("Recettes", "Budget"). block = [ws.title] for row in ws.iter_rows( min_row=1, max_row=_INDEX_ROWS_PER_SHEET, max_col=MAX_COLS, values_only=True ): cells = [_fmt(v) for v in row] # Skip blank rows instead of emitting runs of tabs. if not any(c.strip() for c in cells): continue block.append("\t".join(cells).rstrip()) text = "\n".join(block) chunks.append(text[:budget]) budget -= len(text) except Exception: # Truncated but still useful: keep whatever was collected. pass finally: wb.close() return "\n".join(c for c in chunks if c).strip()