Files
ObsiGate/backend/xlsx_reader.py
T
bruno 69927176df
CI / lint (push) Canceled after 0s
CI / test (push) Canceled after 0s
CI / security (push) Canceled after 0s
CI / build (push) Canceled after 0s
CI / e2e (push) Canceled after 0s
fix: feuille xlsx vide editable avec quadrillage vierge BUG-094
2026-09-29 16:34:33 -04:00

910 lines
36 KiB
Python

"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx).
Read-only: formulas are shown as their text (``data_only=False``) so a
round-trip through the viewer never depends on Excel's cached values.
Write-side lives in ``backend.services.mutations.edit_xlsx_cells``.
:func:`inspect_workbook` lists the workbook features that an openpyxl
round-trip would drop (#153 A1) so the UI can warn before saving.
#153 A16 — :func:`render_sheets` also accepts ``.xlsm`` (macros preserved on
save via ``keep_vba``), ``.xls`` (xlrd) and ``.ods`` (odfpy), both served
read-only; :func:`render_csv_table` turns a CSV into the same table shape.
"""
from __future__ import annotations
import html
import logging
import re
import zipfile
from datetime import date, datetime
from pathlib import Path
from typing import Any
from openpyxl import load_workbook
from openpyxl.utils import get_column_letter
logger = logging.getLogger("obsigate.xlsx_reader")
# ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet).
# Raise them, or paginate per sheet, if a real workbook needs more.
MAX_ROWS = 500
MAX_COLS = 40
# BUG-094 — an empty sheet used to render as a bare "Feuille vide" paragraph
# with no cell at all, so a freshly added sheet had nothing to click and no way
# to insert a row/column. Render a small blank grid instead (Excel-like), with
# real A1 coordinates, so the cells are editable and the structure actions work.
EMPTY_SHEET_ROWS = 20
EMPTY_SHEET_COLS = 8
# #153 A9 — window size served by ``read_sheet_window()`` (lazy per-sheet
# loading). The endpoint is bounded so a single request can never ask for the
# whole workbook back in one JSON payload; the UI pages through the rest.
MAX_WINDOW_ROWS = 1_000
DEFAULT_WINDOW_ROWS = 200
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
# DO survive the round-trip, so they are deliberately absent from this map.
LOSSY_PARTS: dict[str, tuple[str, ...]] = {
"slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"),
"form_controls": ("xl/ctrlProps/", "xl/activeX/"),
"connections": ("xl/queryTables/", "xl/connections.xml"),
"custom_xml": ("customXml/",),
"signature": ("_xmlsignatures/",),
"rich_comments": ("xl/threadedComments/", "xl/persons/"),
"macros": ("xl/vbaProject.bin",),
}
# A formula cell carrying its last computed result: ``<f>…</f><v>…</v>``.
# openpyxl writes an EMPTY ``<v></v>`` itself, hence the ``[^<]`` guard: only a
# non-empty value counts. openpyxl keeps the formula but drops the cached result,
# so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until
# Excel recalculates.
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
# Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES).
_MAX_PROBE_BYTES = 8_000_000
# #153 A17 — OPC parts of chart / pivot objects, matched against the archive
# name list (xl/charts/chart1.xml, xl/pivotTables/pivotTable1.xml, …).
_CHART_PART_RE = re.compile(r"^xl/charts/chart\d+\.xml$")
_PIVOT_PART_RE = re.compile(r"^xl/pivotTables/pivotTable\d+\.xml$")
# #153 A5 — ceiling on the text handed to the TF-IDF / semantic index. A workbook
# is a data dump, not prose: indexing every cell would flood the inverted index
# and bury the notes. Sheet names + the first rows are enough to make a
# spreadsheet findable by its headers.
MAX_INDEX_CHARS = 5_000
_INDEX_ROWS_PER_SHEET = 20
MAX_INDEX_SHEETS = 20
# #153 A15 — reading styles is a second (non-read_only) pass on the sheet XML.
# Bounded like everything else: a cell must be INSIDE the rendered window to
# deserve an inline style, so a huge workbook never triggers a huge payload.
# Only data-driven fragments are emitted: the hex values come from the file,
# never from a hardcoded color table.
def _cell_fragments(cell: Any) -> tuple[list[str], str | None]:
"""Inline CSS fragments of one cell plus its horizontal alignment.
Fixed, color-first order: the API contract documents ``color:...`` as the
first fragment of a styled cell. Only data-driven values are emitted —
every hex comes from the workbook itself, never a hardcoded table.
"""
fragments: list[str] = []
font = cell.font
if font and font.color is not None and isinstance(font.color.rgb, str):
# ARGB from the workbook itself — never a hardcoded table.
rgb = font.color.rgb
if len(rgb) == 8 and rgb != "FF000000":
fragments.append(f"color:#{rgb[2:].lower()}")
fill = cell.fill
if fill and fill.fgColor is not None and isinstance(fill.fgColor.rgb, str):
rgb = fill.fgColor.rgb
if len(rgb) == 8 and rgb not in ("00000000", "FFFFFFFF"):
fragments.append(f"background:#{rgb[2:].lower()}")
if font and font.bold:
fragments.append("font-weight:600")
if font and font.italic:
fragments.append("font-style:italic")
fmt = cell.number_format
if fmt and fmt not in ("General", "@"):
# A custom number format is signalled typographically (mono font)
# rather than rendered: the displayed value already carries the
# formatting from _fmt(). Single quotes: the fragment lands inside a
# double-quoted HTML attribute.
fragments.append("font-family:'JetBrains Mono',monospace")
alignment = cell.alignment
align = alignment.horizontal if alignment else None
return fragments, (align if align in ("left", "right", "center") else None)
def _sheet_style_maps(ws: Any) -> tuple[dict[str, str], dict[str, str]]:
"""Flat ``{ref: css}`` and ``{ref: align}`` maps of one worksheet.
The flat string is what the API serves and what the viewer applies
verbatim to ``td.style``; a plain cell is simply absent from the map.
``left`` is the table default and never included. Bounded by
``MAX_ROWS x MAX_COLS`` like the render itself.
"""
styles: dict[str, str] = {}
aligns: dict[str, str] = {}
for row in ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS):
for cell in row:
if cell.value is None and cell.number_format == "General":
continue
fragments, align = _cell_fragments(cell)
if fragments:
styles[cell.coordinate] = ";".join(fragments)
if align and align != "left":
aligns[cell.coordinate] = align
return styles, aligns
def read_sheet_styles(file_path: Path, sheet: str) -> dict[str, dict[str, Any]]:
"""Return ``{ref: {style, align}}`` for the styled cells of one sheet.
``style`` is the flat CSS fragment the viewer applies verbatim and
``align`` the horizontal text-align when it is not the table default.
Normal (non-streaming) load — styles are unavailable in read_only mode;
a failure yields ``{}`` so the viewer falls back to the plain rendering.
"""
meta = read_workbook_meta(file_path).get(sheet, {})
styles_map = meta.get("styles", {})
aligns = meta.get("aligns", {})
out: dict[str, dict[str, Any]] = {}
for ref, css in styles_map.items():
entry: dict[str, Any] = {"style": css}
if ref in aligns:
entry["align"] = aligns[ref]
out[ref] = entry
return out
def read_sheet_merges(file_path: Path, sheet: str) -> list[str]:
"""Return the merged ranges of one sheet as ``A1:C3`` strings."""
try:
# Styles and merges are only fully materialised in normal mode
# (read_only=True leaves merged_cells empty).
wb = load_workbook(str(file_path))
except Exception:
return []
try:
if sheet not in wb.sheetnames:
return []
merged = getattr(wb[sheet], "merged_cells", None)
ranges = getattr(merged, "ranges", None) or []
return [str(r) for r in ranges]
except Exception:
logger.debug("xlsx merges unavailable", exc_info=True)
return []
finally:
wb.close()
def read_sheet_freeze(file_path: Path, sheet: str) -> str:
"""Return the freeze-panes anchor of one sheet ('' when not frozen).
Normal (non-streaming) load: `freeze_panes` is NOT materialised on
ReadOnlyWorksheet in openpyxl 3.1.x — read_only=True always yields ''.
"""
try:
wb = load_workbook(str(file_path))
except Exception:
return ""
try:
if sheet not in wb.sheetnames:
return ""
return str(getattr(wb[sheet], "freeze_panes", None) or "")
except Exception:
return ""
finally:
wb.close()
def _fmt(value: Any) -> str:
if value is None:
return ""
if isinstance(value, datetime):
return value.strftime("%Y-%m-%d %H:%M")
if isinstance(value, date):
return value.isoformat()
return str(value)
def _trim(grid: list[list[str]]) -> list[list[str]]:
"""Drop trailing empty rows and columns (openpyxl pads to max_col)."""
while grid and not any(grid[-1]):
grid.pop()
if not grid:
return grid
width = 0
for row in grid:
for i in range(len(row) - 1, -1, -1):
if row[i]:
width = max(width, i + 1)
break
return [row[:width] for row in grid]
def _cell_cached(cached: list[list[str]] | None, r: int, c: int) -> str:
"""Return the cached result for a 0-based cell, or ``""``.
The shadow grid is read positionally and may be narrower than the formula
grid (``_trim`` collapses the trailing empty columns of each grid
independently), so every lookup is bounds-checked rather than assumed.
"""
if not cached or r >= len(cached):
return ""
row = cached[r]
return row[c] if c < len(row) else ""
def _table(
grid: list[list[str]],
cached: list[list[str]] | None = None,
row_offset: int = 0,
styles: dict[str, dict[str, Any]] | None = None,
) -> str:
"""Render a grid as an HTML table.
``cached`` is the same grid read with ``data_only=True`` (#153 A12): where a
formula cell still carries its last computed result, it is shown as a
discreet second line (``<span class="xlsx-cached">``) so the user sees the
number Excel last calculated instead of only the formula text. The span
carries ``data-cached-value`` and is titled client-side from
``xlsx.cached_value_title`` — the backend never emits UI text.
``row_offset`` is the number of rows skipped before this grid (#153 A9): the
row numbers and the ``data-cell`` references must stay the real A1
coordinates of the sheet, not of the window.
``styles`` maps A1 references to ``{style, align}`` fragments (#153 A15:
bold, italic, background, alignment) — the backend only reads the
workbook, the fragments are built from it and always data-driven, never
hardcoded colors. A plain ``str`` value is tolerated (legacy callers).
"""
if not grid:
return "<p><em>Feuille vide</em></p>"
n_cols = max(len(row) for row in grid)
out = [
(
'<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
'<thead><tr><th class="xlsx-corner"></th>'
)
]
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
out.append("</tr></thead><tbody>")
for r, row in enumerate(grid, start=row_offset + 1):
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
for c, val in enumerate(row, start=1):
ref = f"{get_column_letter(c)}{r}"
meta = (styles or {}).get(ref)
if meta is None:
style_attr = ""
else:
# Legacy callers may still pass a bare CSS string.
if isinstance(meta, str):
meta = {"style": meta}
fragment = meta.get("style", "")
align = meta.get("align")
if align and align not in ("left",):
# left is the table default; only non-default alignments
# need an explicit declaration.
fragment = f"{fragment};text-align:{align}" if fragment else f"text-align:{align}"
style_attr = f' style="{fragment}"' if fragment else ""
# The cached result only makes sense for a formula cell: on a plain
# value cell the two reads are identical and showing both would
# duplicate the text.
shadow = ""
if cached is not None and val.startswith("="):
# `c` is 1-based (A1 notation) and `r` too, while the grid is
# 0-based: translate both.
cval = _cell_cached(cached, r - 1, c - 1)
if cval and cval != val:
# The tooltip is translated client-side from
# `xlsx.cached_value_title`; never hardcode UI text here.
shadow = (
f'<span class="xlsx-cached" data-cached-value="1">'
f"{html.escape(cval)}</span>"
)
out.append(
f'<td data-cell="{ref}"{style_attr}>{html.escape(val)}{shadow}</td>'
)
out.append("</tr>")
out.append("</tbody></table></div>")
return "".join(out)
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
"""True when at least one formula cell still carries its computed value."""
budget = _MAX_PROBE_BYTES
for name in zf.namelist():
if not name.startswith("xl/worksheets/sheet") or not name.endswith(".xml"):
continue
try:
with zf.open(name) as fh:
while budget > 0:
chunk = fh.read(65536)
if not chunk:
break
budget -= len(chunk)
if _CACHED_FORMULA_RE.search(chunk):
return True
except (KeyError, OSError, zipfile.BadZipFile):
continue
return False
def inspect_workbook(file_path: Path) -> list[str]:
"""Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*.
Read-only inspection of the OPC package (central directory + a bounded scan
of the sheet XML). Never raises: an unreadable or encrypted workbook simply
yields ``[]`` and the save path keeps its current behaviour.
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
the cached result, so the workbook stays correct once Excel recalculates it.
"""
try:
with zipfile.ZipFile(file_path) as zf:
names = set(zf.namelist())
found = {
key
for key, prefixes in LOSSY_PARTS.items()
if any(name.startswith(prefix) for name in names for prefix in prefixes)
}
if _has_cached_formulas(zf):
found.add("cached_values")
return sorted(found)
except (OSError, zipfile.BadZipFile):
return []
def read_workbook_meta(file_path: Path) -> dict[str, dict[str, Any]]:
"""Return ``{sheet: {styles, aligns, merges, freeze}}`` for every sheet.
One normal (non-streaming) load serves the three A15 metadata maps: the
fragments are the workbook's own values, a failure yields ``{}`` per sheet
so the viewer keeps its plain rendering. Styles are read with
``data_only=False`` — the edited value is the formula, not its result.
"""
try:
wb = load_workbook(str(file_path), data_only=False)
except Exception:
return {}
out: dict[str, dict[str, Any]] = {}
try:
for ws in wb.worksheets:
styles, aligns = _sheet_style_maps(ws)
merged = getattr(ws, "merged_cells", None)
ranges = getattr(merged, "ranges", None) or []
out[ws.title] = {
"styles": styles,
"aligns": aligns,
"merges": [str(r) for r in ranges],
"freeze": str(getattr(ws, "freeze_panes", None) or ""),
}
return out
except Exception:
logger.debug("xlsx meta unavailable", exc_info=True)
for t in wb.sheetnames:
out.setdefault(t, {"styles": {}, "aligns": {}, "merges": [], "freeze": ""})
return out
finally:
wb.close()
def render_sheets(file_path: Path) -> list[dict[str, Any]]:
"""Return one dict per sheet: ``{name, html, rows, cols, total_*, truncated}``.
Reads the workbook twice: once with ``data_only=False`` for the formulas
(what the user must edit) and, when any formula carries a cached result
(#153 A12), once with ``data_only=True`` to show what Excel last computed.
The second pass is skipped entirely when the archive holds no cached value,
so the common case still costs a single load.
``total_rows``/``total_cols`` are the dimensions the sheet declares and
``truncated`` says whether the hard caps actually cut it (#153 A8) — the
viewer needs both to stop silently hiding the tail of a sheet.
"""
wb = load_workbook(str(file_path), read_only=True, data_only=False)
try:
formulas = [_sheet_grid(ws) for ws in wb.worksheets]
titles = [ws.title for ws in wb.worksheets]
extents = [_sheet_extent(ws) for ws in wb.worksheets]
finally:
wb.close()
cached: list[list[list[str]]] | None = None
if _has_cached_values(file_path):
cached = _read_cached_grids(file_path, titles)
# #153 A15 — one extra normal-mode load serves the styles/merges/freeze
# metadata of every sheet; the HTML then carries the fragments itself.
meta = read_workbook_meta(file_path)
sheets = []
for i, title in enumerate(titles):
grid = _trim(formulas[i])
# BUG-094 — a blank sheet still needs an editable grid (see constants):
# the viewer's cell editing and structure actions all hang off a cell.
if not grid:
grid = [[""] * EMPTY_SHEET_COLS for _ in range(EMPTY_SHEET_ROWS)]
# The shadow grid is NOT trimmed independently: _trim drops the
# trailing empty columns of each grid on its own width, which would
# shift every cached value left of its formula. Indexing it
# positionally against the untrimmed grid keeps the two aligned.
shadow = cached[i] if cached is not None and i < len(cached) else None
total_rows, total_cols = extents[i]
sheet_meta = meta.get(title, {})
sheets.append(
{
"name": title,
"html": _table(grid, shadow, styles=sheet_meta.get("styles")),
"rows": len(grid),
"cols": max((len(r) for r in grid), default=0),
"total_rows": total_rows,
"total_cols": total_cols,
# Coverage, not display size: `rows`/`cols` are post-trim (a
# sheet of 3 filled cells in a 500-row block renders 1x1), and
# the client must announce the cap it stopped at, not how many
# cells happen to be non-empty.
"max_rows": MAX_ROWS,
"max_cols": MAX_COLS,
# A sheet is truncated when the caps, not the trailing blanks,
# decided its shape: comparing against the *rendered* size would
# flag every sheet carrying a few empty formatted rows.
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
"styles": sheet_meta.get("styles", {}),
"aligns": sheet_meta.get("aligns", {}),
"merges": sheet_meta.get("merges", []),
"freeze": sheet_meta.get("freeze", ""),
}
)
return sheets
def read_sheet_window(
file_path: Path,
sheet: str,
offset: int = 0,
limit: int = DEFAULT_WINDOW_ROWS,
) -> dict[str, Any] | None:
"""Return a window of rows of one sheet, or ``None`` if the sheet is unknown.
Backs the lazy per-sheet loading of #153 A9: the viewer asks for the rows
it is about to display instead of shipping every sheet in the initial file
payload. ``offset`` is 0-based; the row numbers and the ``data-cell``
references in the returned ``html`` are the real A1 coordinates of the
sheet, so a window is indistinguishable from a full render.
``limit`` is clamped to :data:`MAX_WINDOW_ROWS`. Raises nothing: an unknown
sheet yields ``None`` and a broken workbook propagates the caller's usual
500.
"""
offset = max(int(offset), 0)
limit = min(max(int(limit), 1), MAX_WINDOW_ROWS)
wb = load_workbook(str(file_path), read_only=True, data_only=False)
try:
if sheet not in wb.sheetnames:
return None
ws = wb[sheet]
total_rows, total_cols = _sheet_extent(ws)
grid = _trim(
_sheet_grid(ws, min_row=offset + 1, max_row=offset + limit)
)
finally:
wb.close()
shadow: list[list[str]] | None = None
# Same A12 rule as the full render: the second read only happens when the
# archive really holds cached results.
if _has_cached_values(file_path):
shadow = _read_cached_window(file_path, sheet, offset, limit)
# #153 A15 — same metadata as the full render, so a lazy window is
# indistinguishable from it (styles in the HTML, merges/freeze for the
# client-side spanning).
meta = read_workbook_meta(file_path).get(sheet, {})
return {
"sheet": sheet,
"offset": offset,
"limit": limit,
"rows": len(grid),
"cols": max((len(r) for r in grid), default=0),
"total_rows": total_rows,
"total_cols": total_cols,
"max_rows": MAX_ROWS,
"max_cols": MAX_COLS,
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
"has_more": offset + len(grid) < total_rows,
"html": _table(grid, shadow, row_offset=offset, styles=meta.get("styles")),
"styles": meta.get("styles", {}),
"aligns": meta.get("aligns", {}),
"merges": meta.get("merges", []),
"freeze": meta.get("freeze", ""),
}
def _read_cached_window(
file_path: Path, sheet: str, offset: int, limit: int
) -> list[list[str]] | None:
"""``data_only=True`` grid for one window, or ``None`` if unavailable.
Best effort like :func:`_read_cached_grids`: a workbook Excel opens but
openpyxl cannot re-read must still display (formulas only).
"""
try:
wb = load_workbook(str(file_path), read_only=True, data_only=True)
except Exception:
return None
try:
if sheet not in wb.sheetnames:
return None
return _sheet_grid(
wb[sheet], min_row=offset + 1, max_row=offset + limit
)
except Exception:
logger.debug("xlsx cached window unavailable", exc_info=True)
return None
finally:
wb.close()
def _sheet_extent(ws: Any) -> tuple[int, int]:
"""Rows and columns the worksheet declares, never negative.
``max_row``/``max_column`` come from the sheet's dimension record; a
hand-edited file may omit it, hence the defensive coercion.
"""
try:
rows = max(int(getattr(ws, "max_row", 0) or 0), 0)
except (TypeError, ValueError):
rows = 0
try:
cols = max(int(getattr(ws, "max_column", 0) or 0), 0)
except (TypeError, ValueError):
cols = 0
return rows, cols
def _sheet_grid(
ws: Any, min_row: int = 1, max_row: int = MAX_ROWS, max_col: int = MAX_COLS
) -> list[list[str]]:
"""Read a worksheet window into a grid of formatted strings, bounded by the caps."""
return [
[_fmt(v) for v in row]
for row in ws.iter_rows(
min_row=min_row, max_row=max_row, max_col=max_col, values_only=True
)
]
def _has_cached_values(file_path: Path) -> bool:
"""True when the archive holds at least one ``<f>…</f><v>…</v>``."""
try:
with zipfile.ZipFile(file_path) as zf:
return _has_cached_formulas(zf)
except (OSError, zipfile.BadZipFile):
return False
def _read_cached_grids(
file_path: Path, titles: list[str]
) -> list[list[list[str]]] | None:
"""Read every sheet with ``data_only=True`` (what Excel last computed).
Best effort: returns ``None`` on any failure so the viewer falls back to the
formula-only rendering. A workbook Excel opens but openpyxl cannot re-read
must still display.
"""
try:
wb = load_workbook(str(file_path), read_only=True, data_only=True)
except Exception:
return None
try:
grids = [_sheet_grid(ws) for ws in wb.worksheets]
if [ws.title for ws in wb.worksheets] != titles:
return None
return grids
except Exception:
logger.debug("xlsx cached values unavailable", exc_info=True)
return None
finally:
wb.close()
def extract_indexable_text(file_path: Path) -> str:
"""Return searchable text for the TF-IDF / semantic index (#153 A5).
Sheet names plus the first :data:`_INDEX_ROWS_PER_SHEET` rows of each
sheet, capped at :data:`MAX_INDEX_CHARS`. Rows are tab-joined so a search
for a header matches the sheet it belongs to.
Never raises: a corrupt, encrypted or unsupported workbook yields ``""`` so
the file still gets indexed by name (same contract as :func:`inspect_workbook`).
"""
chunks: list[str] = []
budget = MAX_INDEX_CHARS
try:
wb = load_workbook(str(file_path), read_only=True, data_only=True)
except Exception:
# Encrypted (BadZipFile) or not a real workbook: name-only indexing.
return ""
try:
for ws in wb.worksheets[:MAX_INDEX_SHEETS]:
if budget <= 0:
break
# The sheet title alone is a strong signal ("Recettes", "Budget").
block = [ws.title]
for row in ws.iter_rows(
min_row=1, max_row=_INDEX_ROWS_PER_SHEET, max_col=MAX_COLS, values_only=True
):
cells = [_fmt(v) for v in row]
# Skip blank rows instead of emitting runs of tabs.
if not any(c.strip() for c in cells):
continue
block.append("\t".join(cells).rstrip())
text = "\n".join(block)
chunks.append(text[:budget])
budget -= len(text)
except Exception:
# Truncated but still useful: keep whatever was collected.
pass
finally:
wb.close()
return "\n".join(c for c in chunks if c).strip()
# ── #153 A17 — dashboard metadata ───────────────────────────────────
def read_workbook_dashboard(file_path: Path) -> dict[str, Any]:
"""Return the dashboard metadata of a workbook (#153 A17).
Shape::
{
"named_ranges": [{"name", "scope", "ref"}],
"objects": {"charts": int, "pivots": int},
"sheets": [{
"name": str,
"cells": int, # non-empty cells inside the caps
"rows": int, # rows carrying at least one non-empty cell
"cols": int, # columns carrying at least one non-empty cell
"formulas": int,
"numeric": int,
"kpi": [ # first 8 numeric cells as {"label", "value"}
{"label": str, "value": float}
],
}],
}
Named ranges come from the streaming load (available read-only), cell
stats from ``iter_rows(values_only=True)``. Charts/pivots are counted by
OPC part names (a chart part per chart, a pivot table part per pivot).
Bounded by MAX_ROWS/MAX_COLS; never raises — a failure yields an empty
payload and the viewer simply hides the panel.
"""
payload: dict[str, Any] = {
"named_ranges": [],
"objects": {"charts": 0, "pivots": 0},
"sheets": [],
}
try:
wb = load_workbook(str(file_path), read_only=True, data_only=False)
except Exception:
return payload
try:
dn = getattr(wb, "defined_names", None)
items: list[tuple[Any, Any]] = (
list(dn.items()) if dn is not None and hasattr(dn, "items") else []
)
for name, defn in items:
scope_idx = getattr(defn, "localSheetId", None)
scope = ""
if scope_idx is not None:
try:
scope = wb.sheetnames[int(scope_idx)]
except (IndexError, ValueError):
scope = ""
payload["named_ranges"].append(
{
"name": str(name),
"scope": scope,
"ref": str(getattr(defn, "attr_text", "") or ""),
}
)
payload["named_ranges"].sort(key=lambda d: d["name"].lower())
for ws in wb.worksheets:
cells = rows = formulas = numeric = 0
col_seen: set[int] = set()
kpi: list[dict[str, Any]] = []
for r, row in enumerate(
ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS, values_only=True),
start=1,
):
row_has_value = False
for c, value in enumerate(row, start=1):
if value is None or (isinstance(value, str) and not value.strip()):
continue
cells += 1
col_seen.add(c)
row_has_value = True
if isinstance(value, str) and value.startswith("="):
formulas += 1
elif isinstance(value, bool):
pass
elif isinstance(value, (int, float)):
numeric += 1
if len(kpi) < 8:
kpi.append(
{"label": f"{get_column_letter(c)}{r}", "value": value}
)
if row_has_value:
rows += 1
payload["sheets"].append(
{
"name": ws.title,
"cells": cells,
"rows": rows,
"cols": len(col_seen),
"formulas": formulas,
"numeric": numeric,
"kpi": kpi,
}
)
# Chart/pivot parts, counted from the archive (chart XML parts are
# one per chart; pivot parts one per pivot table/cache).
with zipfile.ZipFile(file_path) as zf:
names = zf.namelist()
payload["objects"]["charts"] = sum(1 for n in names if _CHART_PART_RE.match(n))
payload["objects"]["pivots"] = sum(1 for n in names if _PIVOT_PART_RE.match(n))
return payload
except Exception:
logger.debug("xlsx dashboard unavailable", exc_info=True)
return {
"named_ranges": [],
"objects": {"charts": 0, "pivots": 0},
"sheets": [],
}
finally:
wb.close()
# ── #153 A16 — additional spreadsheet formats ───────────────────────────────
def render_csv_table(raw: str, *, delimiter: str = ",") -> str:
"""Render CSV text as the same HTML table shape the xlsx viewer consumes.
Row numbers replace the A1 column: a CSV has no fixed column count, so
the first row is a plain data row like the others (the viewer offers the
toolbar either way). Every cell is HTML-escaped at render time.
"""
import csv as csv_mod
import io as io_mod
reader = csv_mod.reader(io_mod.StringIO(raw), delimiter=delimiter)
try:
rows = [row for row in reader]
except csv_mod.Error:
# A malformed CSV still renders: each line becomes a one-cell row.
rows = [[line] for line in raw.splitlines()]
if not rows:
return "<p><em>Feuille vide</em></p>"
n_cols = max(len(r) for r in rows)
out = [
('<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
'<thead><tr><th class="xlsx-corner"></th>')
]
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
out.append("</tr></thead><tbody>")
for r, row in enumerate(rows, start=1):
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
for c in range(1, n_cols + 1):
val = row[c - 1] if c - 1 < len(row) else ""
out.append(f'<td data-cell="{get_column_letter(c)}{r}">{html.escape(val)}</td>')
out.append("</tr>")
out.append("</tbody></table></div>")
return "".join(out)
def render_legacy_workbook(file_path: Path, ext: str) -> list[dict[str, Any]]:
"""Render ``.xls``/``.ods`` sheets with the same dict shape as xlsx.
Read-only formats (#153 A16): ``styles``/``aligns``/``merges``/``freeze``
are served empty so the client-side wiring keeps one code path. Raises
nothing to the render path: an unreadable file yields one error sheet.
"""
name = file_path.name
try:
if ext == ".xls":
import xlrd
book = xlrd.open_workbook(str(file_path))
titles = book.sheet_names()
grids = []
for si in range(book.nsheets):
sh = book.sheet_by_index(si)
grid = [
[_fmt(sh.cell_value(r, c)) for c in range(min(sh.ncols, MAX_COLS))]
for r in range(min(sh.nrows, MAX_ROWS))
]
grids.append(_trim(grid))
total = [(sh.nrows, sh.ncols) for sh in (book.sheet_by_index(i) for i in range(book.nsheets))]
elif ext == ".ods":
from odf.opendocument import load as odf_load
from odf.table import Table, TableCell, TableRow
from odf.teletype import extractText
doc = odf_load(str(file_path))
titles = []
grids = []
total = []
for table in doc.getElementsByType(Table):
title = table.getAttribute("name") or f"Feuille {len(titles) + 1}"
titles.append(title)
grid = []
for row in table.getElementsByType(TableRow)[:MAX_ROWS]:
row_cells = row.getElementsByType(TableCell)
values: list[str] = []
for tc in row_cells[:MAX_COLS]:
repeat = int(tc.getAttribute("numbercolumnsrepeated") or 1)
values.extend([extractText(tc)] * min(repeat, MAX_COLS - len(values)))
grid.append(values)
grids.append(_trim(grid))
total.append((len(grid), max((len(r) for r in grid), default=0)))
else:
raise ValueError(f"Unsupported legacy format: {ext}")
except Exception as exc:
logger.warning("legacy workbook render failed for %s: %s", name, exc)
return [
{
"name": name,
"html": (
'<p><em>Feuille vide</em></p>'
),
"rows": 0,
"cols": 0,
"total_rows": 0,
"total_cols": 0,
"max_rows": MAX_ROWS,
"max_cols": MAX_COLS,
"truncated": False,
"styles": {},
"aligns": {},
"merges": [],
"freeze": "",
}
]
sheets: list[dict[str, Any]] = []
for i, title in enumerate(titles):
grid = grids[i] if i < len(grids) else []
t_rows, t_cols = total[i] if i < len(total) else (0, 0)
sheets.append(
{
"name": title,
"html": _table(grid),
"rows": len(grid),
"cols": max((len(r) for r in grid), default=0),
"total_rows": t_rows,
"total_cols": t_cols,
"max_rows": MAX_ROWS,
"max_cols": MAX_COLS,
"truncated": t_rows > MAX_ROWS or t_cols > MAX_COLS,
"styles": {},
"aligns": {},
"merges": [],
"freeze": "",
}
)
return sheets