"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx). Read-only: formulas are shown as their text (``data_only=False``) so a round-trip through the viewer never depends on Excel's cached values. Write-side lives in ``backend.services.mutations.edit_xlsx_cells``. :func:`inspect_workbook` lists the workbook features that an openpyxl round-trip would drop (#153 A1) so the UI can warn before saving. #153 A16 — :func:`render_sheets` also accepts ``.xlsm`` (macros preserved on save via ``keep_vba``), ``.xls`` (xlrd) and ``.ods`` (odfpy), both served read-only; :func:`render_csv_table` turns a CSV into the same table shape. """ from __future__ import annotations import csv import html import logging import re import threading import zipfile from datetime import date, datetime from pathlib import Path from typing import Any from openpyxl import load_workbook from openpyxl.utils import get_column_letter logger = logging.getLogger("obsigate.xlsx_reader") # ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet). # Raise them, or paginate per sheet, if a real workbook needs more. MAX_ROWS = 500 MAX_COLS = 40 # BUG-094 — an empty sheet used to render as a bare "Feuille vide" paragraph # with no cell at all, so a freshly added sheet had nothing to click and no way # to insert a row/column. Render a small blank grid instead (Excel-like), with # real A1 coordinates, so the cells are editable and the structure actions work. EMPTY_SHEET_ROWS = 20 EMPTY_SHEET_COLS = 8 # #153 A9 — window size served by ``read_sheet_window()`` (lazy per-sheet # loading). The endpoint is bounded so a single request can never ask for the # whole workbook back in one JSON payload; the UI pages through the rest. MAX_WINDOW_ROWS = 1_000 DEFAULT_WINDOW_ROWS = 200 # #153 A1 — workbook parts openpyxl does not re-serialize on load+save. # Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables # DO survive the round-trip, so they are deliberately absent from this map. LOSSY_PARTS: dict[str, tuple[str, ...]] = { "slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"), "form_controls": ("xl/ctrlProps/", "xl/activeX/"), "connections": ("xl/queryTables/", "xl/connections.xml"), "custom_xml": ("customXml/",), "signature": ("_xmlsignatures/",), "rich_comments": ("xl/threadedComments/", "xl/persons/"), "macros": ("xl/vbaProject.bin",), } # A formula cell carrying its last computed result: ``……``. # openpyxl writes an EMPTY ```` itself, hence the ``[^<]`` guard: only a # non-empty value counts. openpyxl keeps the formula but drops the cached result, # so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until # Excel recalculates. _CACHED_FORMULA_RE = re.compile(rb"][^<]*\s*[^<]") # Sheet XML scanned by the cached-formula probe (CPU guard, like # MAX_REPLACE_FILE_BYTES). BUG-099 — the allowance is PER SHEET: a single huge # first sheet used to eat the whole budget and hide a cached formula sitting in # the next one. The total ceiling still bounds the work on a many-sheet archive. # Running out of budget is reported as *unverified* (never as "nothing to lose") # so the write guard stays cautious instead of silently dropping the values. _MAX_PROBE_BYTES_PER_SHEET = 4_000_000 _MAX_PROBE_BYTES_TOTAL = 32_000_000 # #156-A13 — metadata cache. `read_sheet_window()` serves one window at a time # and used to reload the whole workbook (normal mode, data_only=False) for every # window, just to read the style/merge/freeze maps of one sheet. The result is # keyed by (path, mtime_ns, size): any write replaces the file, hence the key. _META_CACHE_MAX = 8 _meta_cache: dict[str, tuple[tuple[int, int], dict[str, dict[str, Any]]]] = {} _meta_cache_lock = threading.Lock() # #153 A17 — OPC parts of chart / pivot objects, matched against the archive # name list (xl/charts/chart1.xml, xl/pivotTables/pivotTable1.xml, …). _CHART_PART_RE = re.compile(r"^xl/charts/chart\d+\.xml$") _PIVOT_PART_RE = re.compile(r"^xl/pivotTables/pivotTable\d+\.xml$") # #153 A5 — ceiling on the text handed to the TF-IDF / semantic index. A workbook # is a data dump, not prose: indexing every cell would flood the inverted index # and bury the notes. Sheet names + the first rows are enough to make a # spreadsheet findable by its headers. MAX_INDEX_CHARS = 5_000 _INDEX_ROWS_PER_SHEET = 20 MAX_INDEX_SHEETS = 20 # #153 A15 — reading styles is a second (non-read_only) pass on the sheet XML. # Bounded like everything else: a cell must be INSIDE the rendered window to # deserve an inline style, so a huge workbook never triggers a huge payload. # Only data-driven fragments are emitted: the hex values come from the file, # never from a hardcoded color table. def _cell_fragments(cell: Any) -> tuple[list[str], str | None]: """Inline CSS fragments of one cell plus its horizontal alignment. Fixed, color-first order: the API contract documents ``color:...`` as the first fragment of a styled cell. Only data-driven values are emitted — every hex comes from the workbook itself, never a hardcoded table. """ fragments: list[str] = [] font = cell.font if font and font.color is not None and isinstance(font.color.rgb, str): # ARGB from the workbook itself — never a hardcoded table. rgb = font.color.rgb if len(rgb) == 8 and rgb != "FF000000": fragments.append(f"color:#{rgb[2:].lower()}") fill = cell.fill if fill and fill.fgColor is not None and isinstance(fill.fgColor.rgb, str): rgb = fill.fgColor.rgb if len(rgb) == 8 and rgb not in ("00000000", "FFFFFFFF"): fragments.append(f"background:#{rgb[2:].lower()}") if font and font.bold: fragments.append("font-weight:600") if font and font.italic: fragments.append("font-style:italic") # #156-A8 — underline is written from the viewer too, so it is read back # (the toggle in the formatting menu needs to see its own effect). decorations = [] if font and font.underline: decorations.append("underline") if font and getattr(font, "strike", False): decorations.append("line-through") if decorations: fragments.append("text-decoration:" + " ".join(decorations)) size = getattr(font, "size", None) if font else None if isinstance(size, (int, float)) and size != 11: fragments.append(f"font-size:{size:g}pt") fmt = cell.number_format if fmt and fmt not in ("General", "@"): # A custom number format is signalled typographically (mono font) # rather than rendered: the displayed value already carries the # formatting from _fmt(). Single quotes: the fragment lands inside a # double-quoted HTML attribute. fragments.append("font-family:'JetBrains Mono',monospace") alignment = cell.alignment align = alignment.horizontal if alignment else None if alignment is not None: vertical = getattr(alignment, "vertical", None) if vertical == "center": fragments.append("vertical-align:middle") elif vertical in ("top", "bottom"): fragments.append(f"vertical-align:{vertical}") if getattr(alignment, "wrap_text", None) is True: fragments.append("white-space:normal") elif getattr(alignment, "wrap_text", None) is False: fragments.append("white-space:nowrap") rotation = getattr(alignment, "text_rotation", None) or 0 if rotation == 90 or rotation == 255: fragments.append("writing-mode:vertical-rl") elif rotation == 180: fragments.append("writing-mode:vertical-rl;transform:scale(-1)") else: # Angles diagonaux (ex. ±45° du menu Format › Rotation) : OOXML # stocke l'anti-horaire sur 0..90 et l'horaire sur 91..180 (-45° # → 135, cf. mutations). Le CSS tourne en sens horaire : signe # inversé pour un rendu fidèle. excel = rotation if rotation <= 90 else -(180 - rotation) if excel: fragments.append(f"transform:rotate({-excel}deg)") border = getattr(cell, "border", None) if border is not None: sides = [] for side_name in ("left", "right", "top", "bottom"): side = getattr(border, side_name, None) style = getattr(side, "style", None) if not style: continue color = getattr(getattr(side, "color", None), "rgb", None) hexpart = "" if isinstance(color, str) and len(color) == 8: hexpart = f" #{color[2:].lower()}" width = {"medium": "2px", "thick": "3px"}.get(style, "1px") css_style = {"dashed": "dashed", "dotted": "dotted", "double": "double"}.get( style, "solid" ) sides.append(f"border-{side_name}:{width} {css_style}{hexpart}") fragments.extend(sides) return fragments, (align if align in ("left", "right", "center", "justify") else None) def _sheet_style_maps( ws: Any, ) -> tuple[dict[str, str], dict[str, str], dict[str, str], dict[str, str]]: """Flat maps of one worksheet: css, align, comment text, number format. The flat string is what the API serves and what the viewer applies verbatim to ``td.style``; a plain cell is simply absent from the map. ``left`` is the table default and never included. Bounded by ``MAX_ROWS x MAX_COLS`` like the render itself. Comment texts are capped (tooltip use) and formats only list non-General number formats. """ styles: dict[str, str] = {} aligns: dict[str, str] = {} comments: dict[str, str] = {} formats: dict[str, str] = {} for row in ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS): for cell in row: comment = getattr(cell, "comment", None) comment_text = getattr(comment, "text", None) if comment is not None else None # Une cellule vide mais stylée (gras, diagonale, commentaire…) # doit survivre à la relecture : `has_style` (et le commentaire) # sont les seuls tests rapides qui la distinguent d'une cellule # vraiment vierge — sinon le style semble « ne pas fonctionner » # dès qu'on recharge le classeur. if cell.value is None and cell.number_format == "General" and not cell.has_style: if comment_text: comments[cell.coordinate] = str(comment_text)[:500] continue fragments, align = _cell_fragments(cell) if fragments: styles[cell.coordinate] = ";".join(fragments) if align and align != "left": aligns[cell.coordinate] = align if comment_text: comments[cell.coordinate] = str(comment_text)[:500] fmt = getattr(cell, "number_format", None) if fmt and fmt not in ("General", "@", None): formats[cell.coordinate] = str(fmt) return styles, aligns, comments, formats def _sheet_dim_maps(ws: Any) -> tuple[dict[str, float], dict[str, float]]: """Explicit column widths (units) and row heights (pt), by letter/number. Only dimensions the file sets explicitly are listed (defaults are a rendering concern of the client). Bounded like the render itself. """ from openpyxl.utils import column_index_from_string colwidths: dict[str, float] = {} try: for key, dim in (getattr(ws, "column_dimensions", {}) or {}).items(): width = getattr(dim, "width", None) if not isinstance(width, (int, float)): continue letters = str(key).split(":") try: cols = [column_index_from_string(a.strip()) for a in letters] except ValueError: continue lo, hi = min(cols), max(cols) for c in range(lo, min(hi, MAX_COLS) + 1): if c >= 1: from openpyxl.utils import get_column_letter colwidths[get_column_letter(c)] = float(width) except Exception: logger.debug("xlsx column widths unavailable", exc_info=True) rowheights: dict[str, float] = {} try: for r, dim in (getattr(ws, "row_dimensions", {}) or {}).items(): if not isinstance(r, int) or not 1 <= r <= MAX_ROWS: continue height = getattr(dim, "height", None) if isinstance(height, (int, float)): rowheights[str(r)] = float(height) except Exception: logger.debug("xlsx row heights unavailable", exc_info=True) return colwidths, rowheights def read_sheet_styles(file_path: Path, sheet: str) -> dict[str, dict[str, Any]]: """Return ``{ref: {style, align}}`` for the styled cells of one sheet. ``style`` is the flat CSS fragment the viewer applies verbatim and ``align`` the horizontal text-align when it is not the table default. Normal (non-streaming) load — styles are unavailable in read_only mode; a failure yields ``{}`` so the viewer falls back to the plain rendering. """ meta = read_workbook_meta(file_path).get(sheet, {}) styles_map = meta.get("styles", {}) aligns = meta.get("aligns", {}) out: dict[str, dict[str, Any]] = {} for ref, css in styles_map.items(): entry: dict[str, Any] = {"style": css} if ref in aligns: entry["align"] = aligns[ref] out[ref] = entry return out def read_sheet_merges(file_path: Path, sheet: str) -> list[str]: """Return the merged ranges of one sheet as ``A1:C3`` strings.""" try: # Styles and merges are only fully materialised in normal mode # (read_only=True leaves merged_cells empty). wb = load_workbook(str(file_path)) except Exception: return [] try: if sheet not in wb.sheetnames: return [] merged = getattr(wb[sheet], "merged_cells", None) ranges = getattr(merged, "ranges", None) or [] return [str(r) for r in ranges] except Exception: logger.debug("xlsx merges unavailable", exc_info=True) return [] finally: wb.close() def read_sheet_freeze(file_path: Path, sheet: str) -> str: """Return the freeze-panes anchor of one sheet ('' when not frozen). Normal (non-streaming) load: `freeze_panes` is NOT materialised on ReadOnlyWorksheet in openpyxl 3.1.x — read_only=True always yields ''. """ try: wb = load_workbook(str(file_path)) except Exception: return "" try: if sheet not in wb.sheetnames: return "" return str(getattr(wb[sheet], "freeze_panes", None) or "") except Exception: return "" finally: wb.close() def _fmt(value: Any) -> str: if value is None: return "" if isinstance(value, datetime): return value.strftime("%Y-%m-%d %H:%M") if isinstance(value, date): return value.isoformat() return str(value) def _trim(grid: list[list[str]]) -> list[list[str]]: """Drop trailing empty rows and columns (openpyxl pads to max_col).""" while grid and not any(grid[-1]): grid.pop() if not grid: return grid width = 0 for row in grid: for i in range(len(row) - 1, -1, -1): if row[i]: width = max(width, i + 1) break return [row[:width] for row in grid] def _cell_cached(cached: list[list[str]] | None, r: int, c: int) -> str: """Return the cached result for a 0-based cell, or ``""``. The shadow grid is read positionally and may be narrower than the formula grid (``_trim`` collapses the trailing empty columns of each grid independently), so every lookup is bounds-checked rather than assumed. """ if not cached or r >= len(cached): return "" row = cached[r] return row[c] if c < len(row) else "" def _table( grid: list[list[str]], cached: list[list[str]] | None = None, row_offset: int = 0, styles: dict[str, dict[str, Any]] | None = None, ) -> str: """Render a grid as an HTML table. ``cached`` is the same grid read with ``data_only=True`` (#153 A12): where a formula cell still carries its last computed result, it is shown as a discreet second line (````) so the user sees the number Excel last calculated instead of only the formula text. The span carries ``data-cached-value`` and is titled client-side from ``xlsx.cached_value_title`` — the backend never emits UI text. ``row_offset`` is the number of rows skipped before this grid (#153 A9): the row numbers and the ``data-cell`` references must stay the real A1 coordinates of the sheet, not of the window. ``styles`` maps A1 references to ``{style, align}`` fragments (#153 A15: bold, italic, background, alignment) — the backend only reads the workbook, the fragments are built from it and always data-driven, never hardcoded colors. A plain ``str`` value is tolerated (legacy callers). """ if not grid: return "

Feuille vide

" n_cols = max(len(row) for row in grid) out = [ ( '
' '' ) ] out += [f"" for c in range(1, n_cols + 1)] out.append("") for r, row in enumerate(grid, start=row_offset + 1): out.append(f'') for c, val in enumerate(row, start=1): ref = f"{get_column_letter(c)}{r}" meta = (styles or {}).get(ref) if meta is None: style_attr = "" else: # Legacy callers may still pass a bare CSS string. if isinstance(meta, str): meta = {"style": meta} fragment = meta.get("style", "") align = meta.get("align") if align and align not in ("left",): # left is the table default; only non-default alignments # need an explicit declaration. fragment = f"{fragment};text-align:{align}" if fragment else f"text-align:{align}" style_attr = f' style="{fragment}"' if fragment else "" # The cached result only makes sense for a formula cell: on a plain # value cell the two reads are identical and showing both would # duplicate the text. shadow = "" if cached is not None and val.startswith("="): # `c` is 1-based (A1 notation) and `r` too, while the grid is # 0-based: translate both. cval = _cell_cached(cached, r - 1, c - 1) if cval and cval != val: # The tooltip is translated client-side from # `xlsx.cached_value_title`; never hardcode UI text here. shadow = ( f'' f"{html.escape(cval)}" ) out.append( f'' ) out.append("") out.append("
{get_column_letter(c)}
{r}{html.escape(val)}{shadow}
") return "".join(out) def _is_sheet_xml(name: str) -> bool: """True for the worksheet XML parts the cached-formula probe scans.""" return name.startswith("xl/worksheets/sheet") and name.endswith(".xml") def _scan_cached_formulas(zf: zipfile.ZipFile) -> tuple[bool, bool]: """Scan the sheet XML for a formula carrying a non-empty cached result. Returns ``(found, unverified)``. ``unverified`` is True when the byte budget stopped the scan before every sheet could be read to the end: a negative result is then **not** proof that the workbook holds no cached value (BUG-099), so callers must not treat it as a licence to write. Each sheet gets its own :data:`_MAX_PROBE_BYTES_PER_SHEET` allowance (a single huge sheet can no longer starve the others) while :data:`_MAX_PROBE_BYTES_TOTAL` bounds the whole archive. A hit short- circuits the scan: the answer is already known. """ budget = _MAX_PROBE_BYTES_TOTAL unverified = False for name in zf.namelist(): if not _is_sheet_xml(name): continue sheet_budget = min(_MAX_PROBE_BYTES_PER_SHEET, budget) exhausted = False try: with zf.open(name) as fh: while sheet_budget > 0: # Read at most what the sheet's allowance has left, so one # large chunk can never consume the whole total budget. chunk = fh.read(min(65536, sheet_budget)) if not chunk: break # read to the end: this sheet is verified clean sheet_budget -= len(chunk) budget -= len(chunk) if _CACHED_FORMULA_RE.search(chunk): return True, False else: # Left the loop on the budget, not on EOF. exhausted = True except (KeyError, OSError, zipfile.BadZipFile): continue if exhausted: unverified = True if budget <= 0: unverified = True break return False, unverified def _has_cached_formulas(zf: zipfile.ZipFile) -> bool: """True when at least one formula cell still carries its computed value. Boolean view of :func:`_scan_cached_formulas` for the display path (the second ``data_only=True`` read): a truncated scan simply skips the shadow grid, it never claims the workbook is lossless. """ found, _ = _scan_cached_formulas(zf) return found def inspect_workbook(file_path: Path) -> list[str]: """Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*. Read-only inspection of the OPC package (central directory + a bounded scan of the sheet XML). Never raises: an unreadable or encrypted workbook simply yields ``[]`` and the save path keeps its current behaviour. ``cached_values`` is a synthetic key: openpyxl keeps the formula but drops the cached result, so the workbook stays correct once Excel recalculates it. ``cached_values_unverified`` (BUG-099) is the other synthetic key: the cached-value probe ran out of budget, so a negative result is not proof — the entry keeps the write guard cautious (409 + confirmation) rather than promising a lossless round-trip it cannot vouch for. """ try: with zipfile.ZipFile(file_path) as zf: names = set(zf.namelist()) found = { key for key, prefixes in LOSSY_PARTS.items() if any(name.startswith(prefix) for name in names for prefix in prefixes) } cached_found, cached_unverified = _scan_cached_formulas(zf) if cached_found: found.add("cached_values") elif cached_unverified: found.add("cached_values_unverified") return sorted(found) except (OSError, zipfile.BadZipFile): return [] def invalidate_workbook_meta(file_path: Path | str) -> None: """Drop the cached metadata of one workbook (#156-A13). Called by the write path right after the atomic replace: the (mtime, size) key already changes on a rewrite, this only closes the theoretical window where a same-size write lands on the same timestamp tick. """ with _meta_cache_lock: _meta_cache.pop(str(file_path), None) def read_workbook_meta(file_path: Path) -> dict[str, dict[str, Any]]: """Return ``{sheet: {styles, aligns, comments, formats, colwidths, rowheights, merges, freeze}}``. One normal (non-streaming) load serves the three A15 metadata maps: the fragments are the workbook's own values, a failure yields ``{}`` per sheet so the viewer keeps its plain rendering. Styles are read with ``data_only=False`` — the edited value is the formula, not its result. #156-A13 — the result is cached on ``(path, mtime_ns, size)``: a truncated sheet is fetched window by window, and each window used to pay a full workbook load for these maps alone. Callers only read from the mapping. """ try: st = file_path.stat() except OSError: return {} key = str(file_path) stamp = (st.st_mtime_ns, st.st_size) with _meta_cache_lock: hit = _meta_cache.get(key) if hit and hit[0] == stamp: return hit[1] out = _read_workbook_meta_uncached(file_path) with _meta_cache_lock: _meta_cache[key] = (stamp, out) while len(_meta_cache) > _META_CACHE_MAX: _meta_cache.pop(next(iter(_meta_cache))) return out def _read_workbook_meta_uncached(file_path: Path) -> dict[str, dict[str, Any]]: """Load the workbook once and build the per-sheet metadata maps.""" try: wb = load_workbook(str(file_path), data_only=False) except Exception: return {} out: dict[str, dict[str, Any]] = {} try: for ws in wb.worksheets: styles, aligns, comments, formats = _sheet_style_maps(ws) colwidths, rowheights = _sheet_dim_maps(ws) merged = getattr(ws, "merged_cells", None) ranges = getattr(merged, "ranges", None) or [] out[ws.title] = { "styles": styles, "aligns": aligns, "comments": comments, "formats": formats, "colwidths": colwidths, "rowheights": rowheights, "merges": [str(r) for r in ranges], "freeze": str(getattr(ws, "freeze_panes", None) or ""), } return out except Exception: logger.debug("xlsx meta unavailable", exc_info=True) for t in wb.sheetnames: out.setdefault( t, { "styles": {}, "aligns": {}, "comments": {}, "formats": {}, "colwidths": {}, "rowheights": {}, "merges": [], "freeze": "", }, ) return out finally: wb.close() def render_sheets(file_path: Path) -> list[dict[str, Any]]: """Return one dict per sheet: ``{name, html, rows, cols, total_*, truncated}``. Reads the workbook twice: once with ``data_only=False`` for the formulas (what the user must edit) and, when any formula carries a cached result (#153 A12), once with ``data_only=True`` to show what Excel last computed. The second pass is skipped entirely when the archive holds no cached value, so the common case still costs a single load. ``total_rows``/``total_cols`` are the dimensions the sheet declares and ``truncated`` says whether the hard caps actually cut it (#153 A8) — the viewer needs both to stop silently hiding the tail of a sheet. """ wb = load_workbook(str(file_path), read_only=True, data_only=False) try: formulas = [_sheet_grid(ws) for ws in wb.worksheets] titles = [ws.title for ws in wb.worksheets] extents = [_sheet_extent(ws) for ws in wb.worksheets] finally: wb.close() cached: list[list[list[str]]] | None = None if _has_cached_values(file_path): cached = _read_cached_grids(file_path, titles) # #153 A15 — one extra normal-mode load serves the styles/merges/freeze # metadata of every sheet; the HTML then carries the fragments itself. meta = read_workbook_meta(file_path) sheets = [] for i, title in enumerate(titles): grid = _trim(formulas[i]) # BUG-094 — a blank sheet still needs an editable grid (see constants): # the viewer's cell editing and structure actions all hang off a cell. if not grid: grid = [[""] * EMPTY_SHEET_COLS for _ in range(EMPTY_SHEET_ROWS)] # The shadow grid is NOT trimmed independently: _trim drops the # trailing empty columns of each grid on its own width, which would # shift every cached value left of its formula. Indexing it # positionally against the untrimmed grid keeps the two aligned. shadow = cached[i] if cached is not None and i < len(cached) else None total_rows, total_cols = extents[i] sheet_meta = meta.get(title, {}) sheets.append( { "name": title, "html": _table(grid, shadow, styles=sheet_meta.get("styles")), "rows": len(grid), "cols": max((len(r) for r in grid), default=0), "total_rows": total_rows, "total_cols": total_cols, # Coverage, not display size: `rows`/`cols` are post-trim (a # sheet of 3 filled cells in a 500-row block renders 1x1), and # the client must announce the cap it stopped at, not how many # cells happen to be non-empty. "max_rows": MAX_ROWS, "max_cols": MAX_COLS, # A sheet is truncated when the caps, not the trailing blanks, # decided its shape: comparing against the *rendered* size would # flag every sheet carrying a few empty formatted rows. "truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS, "styles": sheet_meta.get("styles", {}), "aligns": sheet_meta.get("aligns", {}), "comments": sheet_meta.get("comments", {}), "formats": sheet_meta.get("formats", {}), "colwidths": sheet_meta.get("colwidths", {}), "rowheights": sheet_meta.get("rowheights", {}), "merges": sheet_meta.get("merges", []), "freeze": sheet_meta.get("freeze", ""), } ) return sheets def read_sheet_window( file_path: Path, sheet: str, offset: int = 0, limit: int = DEFAULT_WINDOW_ROWS, ) -> dict[str, Any] | None: """Return a window of rows of one sheet, or ``None`` if the sheet is unknown. Backs the lazy per-sheet loading of #153 A9: the viewer asks for the rows it is about to display instead of shipping every sheet in the initial file payload. ``offset`` is 0-based; the row numbers and the ``data-cell`` references in the returned ``html`` are the real A1 coordinates of the sheet, so a window is indistinguishable from a full render. ``limit`` is clamped to :data:`MAX_WINDOW_ROWS`. Raises nothing: an unknown sheet yields ``None`` and a broken workbook propagates the caller's usual 500. """ offset = max(int(offset), 0) limit = min(max(int(limit), 1), MAX_WINDOW_ROWS) wb = load_workbook(str(file_path), read_only=True, data_only=False) try: if sheet not in wb.sheetnames: return None ws = wb[sheet] total_rows, total_cols = _sheet_extent(ws) grid = _trim( _sheet_grid(ws, min_row=offset + 1, max_row=offset + limit) ) finally: wb.close() shadow: list[list[str]] | None = None # Same A12 rule as the full render: the second read only happens when the # archive really holds cached results. if _has_cached_values(file_path): shadow = _read_cached_window(file_path, sheet, offset, limit) # #153 A15 — same metadata as the full render, so a lazy window is # indistinguishable from it (styles in the HTML, merges/freeze for the # client-side spanning). meta = read_workbook_meta(file_path).get(sheet, {}) return { "sheet": sheet, "offset": offset, "limit": limit, "rows": len(grid), "cols": max((len(r) for r in grid), default=0), "total_rows": total_rows, "total_cols": total_cols, "max_rows": MAX_ROWS, "max_cols": MAX_COLS, "truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS, "has_more": offset + len(grid) < total_rows, "html": _table(grid, shadow, row_offset=offset, styles=meta.get("styles")), "styles": meta.get("styles", {}), "aligns": meta.get("aligns", {}), "comments": meta.get("comments", {}), "formats": meta.get("formats", {}), "colwidths": meta.get("colwidths", {}), "rowheights": meta.get("rowheights", {}), "merges": meta.get("merges", []), "freeze": meta.get("freeze", ""), } def _read_cached_window( file_path: Path, sheet: str, offset: int, limit: int ) -> list[list[str]] | None: """``data_only=True`` grid for one window, or ``None`` if unavailable. Best effort like :func:`_read_cached_grids`: a workbook Excel opens but openpyxl cannot re-read must still display (formulas only). """ try: wb = load_workbook(str(file_path), read_only=True, data_only=True) except Exception: return None try: if sheet not in wb.sheetnames: return None return _sheet_grid( wb[sheet], min_row=offset + 1, max_row=offset + limit ) except Exception: logger.debug("xlsx cached window unavailable", exc_info=True) return None finally: wb.close() def _sheet_extent(ws: Any) -> tuple[int, int]: """Rows and columns the worksheet declares, never negative. ``max_row``/``max_column`` come from the sheet's dimension record; a hand-edited file may omit it, hence the defensive coercion. """ try: rows = max(int(getattr(ws, "max_row", 0) or 0), 0) except (TypeError, ValueError): rows = 0 try: cols = max(int(getattr(ws, "max_column", 0) or 0), 0) except (TypeError, ValueError): cols = 0 return rows, cols def _sheet_grid( ws: Any, min_row: int = 1, max_row: int = MAX_ROWS, max_col: int = MAX_COLS ) -> list[list[str]]: """Read a worksheet window into a grid of formatted strings, bounded by the caps.""" return [ [_fmt(v) for v in row] for row in ws.iter_rows( min_row=min_row, max_row=max_row, max_col=max_col, values_only=True ) ] def _has_cached_values(file_path: Path) -> bool: """True when the archive holds at least one ``……``.""" try: with zipfile.ZipFile(file_path) as zf: return _has_cached_formulas(zf) except (OSError, zipfile.BadZipFile): return False def _read_cached_grids( file_path: Path, titles: list[str] ) -> list[list[list[str]]] | None: """Read every sheet with ``data_only=True`` (what Excel last computed). Best effort: returns ``None`` on any failure so the viewer falls back to the formula-only rendering. A workbook Excel opens but openpyxl cannot re-read must still display. """ try: wb = load_workbook(str(file_path), read_only=True, data_only=True) except Exception: return None try: grids = [_sheet_grid(ws) for ws in wb.worksheets] if [ws.title for ws in wb.worksheets] != titles: return None return grids except Exception: logger.debug("xlsx cached values unavailable", exc_info=True) return None finally: wb.close() def extract_indexable_text(file_path: Path) -> str: """Return searchable text for the TF-IDF / semantic index (#153 A5). Sheet names plus the first :data:`_INDEX_ROWS_PER_SHEET` rows of each sheet, capped at :data:`MAX_INDEX_CHARS`. Rows are tab-joined so a search for a header matches the sheet it belongs to. Never raises: a corrupt, encrypted or unsupported workbook yields ``""`` so the file still gets indexed by name (same contract as :func:`inspect_workbook`). """ chunks: list[str] = [] budget = MAX_INDEX_CHARS try: wb = load_workbook(str(file_path), read_only=True, data_only=True) except Exception: # Encrypted (BadZipFile) or not a real workbook: name-only indexing. return "" try: for ws in wb.worksheets[:MAX_INDEX_SHEETS]: if budget <= 0: break # The sheet title alone is a strong signal ("Recettes", "Budget"). block = [ws.title] for row in ws.iter_rows( min_row=1, max_row=_INDEX_ROWS_PER_SHEET, max_col=MAX_COLS, values_only=True ): cells = [_fmt(v) for v in row] # Skip blank rows instead of emitting runs of tabs. if not any(c.strip() for c in cells): continue block.append("\t".join(cells).rstrip()) text = "\n".join(block) chunks.append(text[:budget]) budget -= len(text) except Exception: # Truncated but still useful: keep whatever was collected. pass finally: wb.close() return "\n".join(c for c in chunks if c).strip() # ── #153 A17 — dashboard metadata ─────────────────────────────────── def read_workbook_dashboard(file_path: Path) -> dict[str, Any]: """Return the dashboard metadata of a workbook (#153 A17). Shape:: { "named_ranges": [{"name", "scope", "ref"}], "objects": {"charts": int, "pivots": int}, "sheets": [{ "name": str, "cells": int, # non-empty cells inside the caps "rows": int, # rows carrying at least one non-empty cell "cols": int, # columns carrying at least one non-empty cell "formulas": int, "numeric": int, "kpi": [ # first 8 numeric cells as {"label", "value"} {"label": str, "value": float} ], }], } Named ranges come from the streaming load (available read-only), cell stats from ``iter_rows(values_only=True)``. Charts/pivots are counted by OPC part names (a chart part per chart, a pivot table part per pivot). Bounded by MAX_ROWS/MAX_COLS; never raises — a failure yields an empty payload and the viewer simply hides the panel. """ payload: dict[str, Any] = { "named_ranges": [], "objects": {"charts": 0, "pivots": 0}, "sheets": [], } try: wb = load_workbook(str(file_path), read_only=True, data_only=False) except Exception: return payload try: dn = getattr(wb, "defined_names", None) items: list[tuple[Any, Any]] = ( list(dn.items()) if dn is not None and hasattr(dn, "items") else [] ) for name, defn in items: scope_idx = getattr(defn, "localSheetId", None) scope = "" if scope_idx is not None: try: scope = wb.sheetnames[int(scope_idx)] except (IndexError, ValueError): scope = "" payload["named_ranges"].append( { "name": str(name), "scope": scope, "ref": str(getattr(defn, "attr_text", "") or ""), } ) payload["named_ranges"].sort(key=lambda d: d["name"].lower()) for ws in wb.worksheets: cells = rows = formulas = numeric = 0 col_seen: set[int] = set() kpi: list[dict[str, Any]] = [] for r, row in enumerate( ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS, values_only=True), start=1, ): row_has_value = False for c, value in enumerate(row, start=1): if value is None or (isinstance(value, str) and not value.strip()): continue cells += 1 col_seen.add(c) row_has_value = True if isinstance(value, str) and value.startswith("="): formulas += 1 elif isinstance(value, bool): pass elif isinstance(value, (int, float)): numeric += 1 if len(kpi) < 8: kpi.append( {"label": f"{get_column_letter(c)}{r}", "value": value} ) if row_has_value: rows += 1 payload["sheets"].append( { "name": ws.title, "cells": cells, "rows": rows, "cols": len(col_seen), "formulas": formulas, "numeric": numeric, "kpi": kpi, } ) # Chart/pivot parts, counted from the archive (chart XML parts are # one per chart; pivot parts one per pivot table/cache). with zipfile.ZipFile(file_path) as zf: names = zf.namelist() payload["objects"]["charts"] = sum(1 for n in names if _CHART_PART_RE.match(n)) payload["objects"]["pivots"] = sum(1 for n in names if _PIVOT_PART_RE.match(n)) return payload except Exception: logger.debug("xlsx dashboard unavailable", exc_info=True) return { "named_ranges": [], "objects": {"charts": 0, "pivots": 0}, "sheets": [], } finally: wb.close() # ── #153 A16 — additional spreadsheet formats ─────────────────────────────── # BUG-098 — a CSV written by a French office suite is `;`-separated, and the # delimiter must be *detected* (then reused on write-back), not assumed. CSV_SNIFF_BYTES = 4096 _CSV_DELIMITERS = (",", ";", "\t", "|") def sniff_csv_delimiter(raw: str) -> str: """Return the most likely field delimiter of *raw* (``,`` as fallback). :class:`csv.Sniffer` handles quoted fields and multi-line values; it is unreliable on short or single-column samples, so the first non-empty line is counted as a tie-breaker and a comma remains the last resort. """ sample = raw[:CSV_SNIFF_BYTES] try: return csv.Sniffer().sniff(sample, delimiters="".join(_CSV_DELIMITERS)).delimiter except csv.Error: pass first = next((line for line in sample.splitlines() if line.strip()), "") counts = {delim: first.count(delim) for delim in _CSV_DELIMITERS} best = max(counts, key=lambda delim: counts[delim]) return best if counts[best] else "," def render_csv_table(raw: str, *, delimiter: str | None = None) -> str: """Render CSV text as the same HTML table shape the xlsx viewer consumes. Row numbers replace the A1 column: a CSV has no fixed column count, so the first row is a plain data row like the others (the viewer offers the toolbar either way). Every cell is HTML-escaped at render time. ``delimiter`` defaults to the sniffed one (:func:`sniff_csv_delimiter`, BUG-098): a `;`-separated file used to render as a single column. """ import csv as csv_mod import io as io_mod reader = csv_mod.reader( io_mod.StringIO(raw), delimiter=delimiter or sniff_csv_delimiter(raw) ) try: rows = [row for row in reader] except csv_mod.Error: # A malformed CSV still renders: each line becomes a one-cell row. rows = [[line] for line in raw.splitlines()] if not rows: return "

Feuille vide

" n_cols = max(len(r) for r in rows) out = [ ('
' '') ] out += [f"" for c in range(1, n_cols + 1)] out.append("") for r, row in enumerate(rows, start=1): out.append(f'') for c in range(1, n_cols + 1): val = row[c - 1] if c - 1 < len(row) else "" out.append(f'') out.append("") out.append("
{get_column_letter(c)}
{r}{html.escape(val)}
") return "".join(out) def render_legacy_workbook(file_path: Path, ext: str) -> list[dict[str, Any]]: """Render ``.xls``/``.ods`` sheets with the same dict shape as xlsx. Read-only formats (#153 A16): ``styles``/``aligns``/``merges``/``freeze`` are served empty so the client-side wiring keeps one code path. Raises nothing to the render path: an unreadable file yields one error sheet. """ name = file_path.name try: if ext == ".xls": import xlrd book = xlrd.open_workbook(str(file_path)) titles = book.sheet_names() grids = [] for si in range(book.nsheets): sh = book.sheet_by_index(si) grid = [ [_fmt(sh.cell_value(r, c)) for c in range(min(sh.ncols, MAX_COLS))] for r in range(min(sh.nrows, MAX_ROWS)) ] grids.append(_trim(grid)) total = [(sh.nrows, sh.ncols) for sh in (book.sheet_by_index(i) for i in range(book.nsheets))] elif ext == ".ods": from odf.opendocument import load as odf_load from odf.table import Table, TableCell, TableRow from odf.teletype import extractText doc = odf_load(str(file_path)) titles = [] grids = [] total = [] for table in doc.getElementsByType(Table): title = table.getAttribute("name") or f"Feuille {len(titles) + 1}" titles.append(title) grid = [] for row in table.getElementsByType(TableRow)[:MAX_ROWS]: row_cells = row.getElementsByType(TableCell) values: list[str] = [] for tc in row_cells[:MAX_COLS]: repeat = int(tc.getAttribute("numbercolumnsrepeated") or 1) values.extend([extractText(tc)] * min(repeat, MAX_COLS - len(values))) grid.append(values) grids.append(_trim(grid)) total.append((len(grid), max((len(r) for r in grid), default=0))) else: raise ValueError(f"Unsupported legacy format: {ext}") except Exception as exc: logger.warning("legacy workbook render failed for %s: %s", name, exc) return [ { "name": name, "html": ( '

Feuille vide

' ), "rows": 0, "cols": 0, "total_rows": 0, "total_cols": 0, "max_rows": MAX_ROWS, "max_cols": MAX_COLS, "truncated": False, "styles": {}, "aligns": {}, "merges": [], "freeze": "", } ] sheets: list[dict[str, Any]] = [] for i, title in enumerate(titles): grid = grids[i] if i < len(grids) else [] t_rows, t_cols = total[i] if i < len(total) else (0, 0) sheets.append( { "name": title, "html": _table(grid), "rows": len(grid), "cols": max((len(r) for r in grid), default=0), "total_rows": t_rows, "total_cols": t_cols, "max_rows": MAX_ROWS, "max_cols": MAX_COLS, "truncated": t_rows > MAX_ROWS or t_cols > MAX_COLS, "styles": {}, "aligns": {}, "merges": [], "freeze": "", } ) return sheets