A17 — nouveau endpoint GET /api/file/{vault}/xlsx/dashboard (read_workbook_
dashboard : plages nommees avec portee depuis defined_names read_only,
comptage graphiques/TCD par parts OPC, stats par feuille bornées 500x40 :
cellules/lignes/colonnes/formules/numerique + 8 premieres valeurs en cartes
KPI) et panneau frontend toggled depuis la toolbar (table des plages,
cartes KPI par feuille, hint actions IA). Bouton absent pour .csv et
formats en lecture seule ; le menu structure est saute quand le bouton
n'existe pas. i18n FR/EN (xlsx.dashboard_*), 8 tests backend + 2 tests
JSDOM + contre-preuve (5 echecs sur neutralisation), ruff/mypy 0.
🤖 Generated with Codebuff
Co-Authored-By: Codebuff <[email protected]>
899 lines
35 KiB
Python
899 lines
35 KiB
Python
"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx).
|
|
|
|
Read-only: formulas are shown as their text (``data_only=False``) so a
|
|
round-trip through the viewer never depends on Excel's cached values.
|
|
Write-side lives in ``backend.services.mutations.edit_xlsx_cells``.
|
|
|
|
:func:`inspect_workbook` lists the workbook features that an openpyxl
|
|
round-trip would drop (#153 A1) so the UI can warn before saving.
|
|
|
|
#153 A16 — :func:`render_sheets` also accepts ``.xlsm`` (macros preserved on
|
|
save via ``keep_vba``), ``.xls`` (xlrd) and ``.ods`` (odfpy), both served
|
|
read-only; :func:`render_csv_table` turns a CSV into the same table shape.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import html
|
|
import logging
|
|
import re
|
|
import zipfile
|
|
from datetime import date, datetime
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from openpyxl import load_workbook
|
|
from openpyxl.utils import get_column_letter
|
|
|
|
logger = logging.getLogger("obsigate.xlsx_reader")
|
|
|
|
# ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet).
|
|
# Raise them, or paginate per sheet, if a real workbook needs more.
|
|
MAX_ROWS = 500
|
|
MAX_COLS = 40
|
|
|
|
# #153 A9 — window size served by ``read_sheet_window()`` (lazy per-sheet
|
|
# loading). The endpoint is bounded so a single request can never ask for the
|
|
# whole workbook back in one JSON payload; the UI pages through the rest.
|
|
MAX_WINDOW_ROWS = 1_000
|
|
DEFAULT_WINDOW_ROWS = 200
|
|
|
|
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
|
|
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
|
|
# DO survive the round-trip, so they are deliberately absent from this map.
|
|
LOSSY_PARTS: dict[str, tuple[str, ...]] = {
|
|
"slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"),
|
|
"form_controls": ("xl/ctrlProps/", "xl/activeX/"),
|
|
"connections": ("xl/queryTables/", "xl/connections.xml"),
|
|
"custom_xml": ("customXml/",),
|
|
"signature": ("_xmlsignatures/",),
|
|
"rich_comments": ("xl/threadedComments/", "xl/persons/"),
|
|
"macros": ("xl/vbaProject.bin",),
|
|
}
|
|
|
|
# A formula cell carrying its last computed result: ``<f>…</f><v>…</v>``.
|
|
# openpyxl writes an EMPTY ``<v></v>`` itself, hence the ``[^<]`` guard: only a
|
|
# non-empty value counts. openpyxl keeps the formula but drops the cached result,
|
|
# so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until
|
|
# Excel recalculates.
|
|
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
|
|
|
|
# Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES).
|
|
_MAX_PROBE_BYTES = 8_000_000
|
|
|
|
# #153 A17 — OPC parts of chart / pivot objects, matched against the archive
|
|
# name list (xl/charts/chart1.xml, xl/pivotTables/pivotTable1.xml, …).
|
|
_CHART_PART_RE = re.compile(r"^xl/charts/chart\d+\.xml$")
|
|
_PIVOT_PART_RE = re.compile(r"^xl/pivotTables/pivotTable\d+\.xml$")
|
|
|
|
# #153 A5 — ceiling on the text handed to the TF-IDF / semantic index. A workbook
|
|
# is a data dump, not prose: indexing every cell would flood the inverted index
|
|
# and bury the notes. Sheet names + the first rows are enough to make a
|
|
# spreadsheet findable by its headers.
|
|
MAX_INDEX_CHARS = 5_000
|
|
_INDEX_ROWS_PER_SHEET = 20
|
|
MAX_INDEX_SHEETS = 20
|
|
|
|
# #153 A15 — reading styles is a second (non-read_only) pass on the sheet XML.
|
|
# Bounded like everything else: a cell must be INSIDE the rendered window to
|
|
# deserve an inline style, so a huge workbook never triggers a huge payload.
|
|
# Only data-driven fragments are emitted: the hex values come from the file,
|
|
# never from a hardcoded color table.
|
|
|
|
|
|
def _cell_fragments(cell: Any) -> tuple[list[str], str | None]:
|
|
"""Inline CSS fragments of one cell plus its horizontal alignment.
|
|
|
|
Fixed, color-first order: the API contract documents ``color:...`` as the
|
|
first fragment of a styled cell. Only data-driven values are emitted —
|
|
every hex comes from the workbook itself, never a hardcoded table.
|
|
"""
|
|
fragments: list[str] = []
|
|
font = cell.font
|
|
if font and font.color is not None and isinstance(font.color.rgb, str):
|
|
# ARGB from the workbook itself — never a hardcoded table.
|
|
rgb = font.color.rgb
|
|
if len(rgb) == 8 and rgb != "FF000000":
|
|
fragments.append(f"color:#{rgb[2:].lower()}")
|
|
fill = cell.fill
|
|
if fill and fill.fgColor is not None and isinstance(fill.fgColor.rgb, str):
|
|
rgb = fill.fgColor.rgb
|
|
if len(rgb) == 8 and rgb not in ("00000000", "FFFFFFFF"):
|
|
fragments.append(f"background:#{rgb[2:].lower()}")
|
|
if font and font.bold:
|
|
fragments.append("font-weight:600")
|
|
if font and font.italic:
|
|
fragments.append("font-style:italic")
|
|
fmt = cell.number_format
|
|
if fmt and fmt not in ("General", "@"):
|
|
# A custom number format is signalled typographically (mono font)
|
|
# rather than rendered: the displayed value already carries the
|
|
# formatting from _fmt(). Single quotes: the fragment lands inside a
|
|
# double-quoted HTML attribute.
|
|
fragments.append("font-family:'JetBrains Mono',monospace")
|
|
alignment = cell.alignment
|
|
align = alignment.horizontal if alignment else None
|
|
return fragments, (align if align in ("left", "right", "center") else None)
|
|
|
|
|
|
def _sheet_style_maps(ws: Any) -> tuple[dict[str, str], dict[str, str]]:
|
|
"""Flat ``{ref: css}`` and ``{ref: align}`` maps of one worksheet.
|
|
|
|
The flat string is what the API serves and what the viewer applies
|
|
verbatim to ``td.style``; a plain cell is simply absent from the map.
|
|
``left`` is the table default and never included. Bounded by
|
|
``MAX_ROWS x MAX_COLS`` like the render itself.
|
|
"""
|
|
styles: dict[str, str] = {}
|
|
aligns: dict[str, str] = {}
|
|
for row in ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS):
|
|
for cell in row:
|
|
if cell.value is None and cell.number_format == "General":
|
|
continue
|
|
fragments, align = _cell_fragments(cell)
|
|
if fragments:
|
|
styles[cell.coordinate] = ";".join(fragments)
|
|
if align and align != "left":
|
|
aligns[cell.coordinate] = align
|
|
return styles, aligns
|
|
|
|
|
|
def read_sheet_styles(file_path: Path, sheet: str) -> dict[str, dict[str, Any]]:
|
|
"""Return ``{ref: {style, align}}`` for the styled cells of one sheet.
|
|
|
|
``style`` is the flat CSS fragment the viewer applies verbatim and
|
|
``align`` the horizontal text-align when it is not the table default.
|
|
Normal (non-streaming) load — styles are unavailable in read_only mode;
|
|
a failure yields ``{}`` so the viewer falls back to the plain rendering.
|
|
"""
|
|
meta = read_workbook_meta(file_path).get(sheet, {})
|
|
styles_map = meta.get("styles", {})
|
|
aligns = meta.get("aligns", {})
|
|
out: dict[str, dict[str, Any]] = {}
|
|
for ref, css in styles_map.items():
|
|
entry: dict[str, Any] = {"style": css}
|
|
if ref in aligns:
|
|
entry["align"] = aligns[ref]
|
|
out[ref] = entry
|
|
return out
|
|
|
|
|
|
def read_sheet_merges(file_path: Path, sheet: str) -> list[str]:
|
|
"""Return the merged ranges of one sheet as ``A1:C3`` strings."""
|
|
try:
|
|
# Styles and merges are only fully materialised in normal mode
|
|
# (read_only=True leaves merged_cells empty).
|
|
wb = load_workbook(str(file_path))
|
|
except Exception:
|
|
return []
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return []
|
|
merged = getattr(wb[sheet], "merged_cells", None)
|
|
ranges = getattr(merged, "ranges", None) or []
|
|
return [str(r) for r in ranges]
|
|
except Exception:
|
|
logger.debug("xlsx merges unavailable", exc_info=True)
|
|
return []
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def read_sheet_freeze(file_path: Path, sheet: str) -> str:
|
|
"""Return the freeze-panes anchor of one sheet ('' when not frozen).
|
|
|
|
Normal (non-streaming) load: `freeze_panes` is NOT materialised on
|
|
ReadOnlyWorksheet in openpyxl 3.1.x — read_only=True always yields ''.
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path))
|
|
except Exception:
|
|
return ""
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return ""
|
|
return str(getattr(wb[sheet], "freeze_panes", None) or "")
|
|
except Exception:
|
|
return ""
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def _fmt(value: Any) -> str:
|
|
if value is None:
|
|
return ""
|
|
if isinstance(value, datetime):
|
|
return value.strftime("%Y-%m-%d %H:%M")
|
|
if isinstance(value, date):
|
|
return value.isoformat()
|
|
return str(value)
|
|
|
|
|
|
def _trim(grid: list[list[str]]) -> list[list[str]]:
|
|
"""Drop trailing empty rows and columns (openpyxl pads to max_col)."""
|
|
while grid and not any(grid[-1]):
|
|
grid.pop()
|
|
if not grid:
|
|
return grid
|
|
width = 0
|
|
for row in grid:
|
|
for i in range(len(row) - 1, -1, -1):
|
|
if row[i]:
|
|
width = max(width, i + 1)
|
|
break
|
|
return [row[:width] for row in grid]
|
|
|
|
|
|
def _cell_cached(cached: list[list[str]] | None, r: int, c: int) -> str:
|
|
"""Return the cached result for a 0-based cell, or ``""``.
|
|
|
|
The shadow grid is read positionally and may be narrower than the formula
|
|
grid (``_trim`` collapses the trailing empty columns of each grid
|
|
independently), so every lookup is bounds-checked rather than assumed.
|
|
"""
|
|
if not cached or r >= len(cached):
|
|
return ""
|
|
row = cached[r]
|
|
return row[c] if c < len(row) else ""
|
|
|
|
|
|
def _table(
|
|
grid: list[list[str]],
|
|
cached: list[list[str]] | None = None,
|
|
row_offset: int = 0,
|
|
styles: dict[str, dict[str, Any]] | None = None,
|
|
) -> str:
|
|
"""Render a grid as an HTML table.
|
|
|
|
``cached`` is the same grid read with ``data_only=True`` (#153 A12): where a
|
|
formula cell still carries its last computed result, it is shown as a
|
|
discreet second line (``<span class="xlsx-cached">``) so the user sees the
|
|
number Excel last calculated instead of only the formula text. The span
|
|
carries ``data-cached-value`` and is titled client-side from
|
|
``xlsx.cached_value_title`` — the backend never emits UI text.
|
|
|
|
``row_offset`` is the number of rows skipped before this grid (#153 A9): the
|
|
row numbers and the ``data-cell`` references must stay the real A1
|
|
coordinates of the sheet, not of the window.
|
|
|
|
``styles`` maps A1 references to ``{style, align}`` fragments (#153 A15:
|
|
bold, italic, background, alignment) — the backend only reads the
|
|
workbook, the fragments are built from it and always data-driven, never
|
|
hardcoded colors. A plain ``str`` value is tolerated (legacy callers).
|
|
"""
|
|
if not grid:
|
|
return "<p><em>Feuille vide</em></p>"
|
|
n_cols = max(len(row) for row in grid)
|
|
out = [
|
|
(
|
|
'<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
|
'<thead><tr><th class="xlsx-corner"></th>'
|
|
)
|
|
]
|
|
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
|
out.append("</tr></thead><tbody>")
|
|
for r, row in enumerate(grid, start=row_offset + 1):
|
|
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
|
for c, val in enumerate(row, start=1):
|
|
ref = f"{get_column_letter(c)}{r}"
|
|
meta = (styles or {}).get(ref)
|
|
if meta is None:
|
|
style_attr = ""
|
|
else:
|
|
# Legacy callers may still pass a bare CSS string.
|
|
if isinstance(meta, str):
|
|
meta = {"style": meta}
|
|
fragment = meta.get("style", "")
|
|
align = meta.get("align")
|
|
if align and align not in ("left",):
|
|
# left is the table default; only non-default alignments
|
|
# need an explicit declaration.
|
|
fragment = f"{fragment};text-align:{align}" if fragment else f"text-align:{align}"
|
|
style_attr = f' style="{fragment}"' if fragment else ""
|
|
# The cached result only makes sense for a formula cell: on a plain
|
|
# value cell the two reads are identical and showing both would
|
|
# duplicate the text.
|
|
shadow = ""
|
|
if cached is not None and val.startswith("="):
|
|
# `c` is 1-based (A1 notation) and `r` too, while the grid is
|
|
# 0-based: translate both.
|
|
cval = _cell_cached(cached, r - 1, c - 1)
|
|
if cval and cval != val:
|
|
# The tooltip is translated client-side from
|
|
# `xlsx.cached_value_title`; never hardcode UI text here.
|
|
shadow = (
|
|
f'<span class="xlsx-cached" data-cached-value="1">'
|
|
f"{html.escape(cval)}</span>"
|
|
)
|
|
out.append(
|
|
f'<td data-cell="{ref}"{style_attr}>{html.escape(val)}{shadow}</td>'
|
|
)
|
|
out.append("</tr>")
|
|
out.append("</tbody></table></div>")
|
|
return "".join(out)
|
|
|
|
|
|
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
|
|
"""True when at least one formula cell still carries its computed value."""
|
|
budget = _MAX_PROBE_BYTES
|
|
for name in zf.namelist():
|
|
if not name.startswith("xl/worksheets/sheet") or not name.endswith(".xml"):
|
|
continue
|
|
try:
|
|
with zf.open(name) as fh:
|
|
while budget > 0:
|
|
chunk = fh.read(65536)
|
|
if not chunk:
|
|
break
|
|
budget -= len(chunk)
|
|
if _CACHED_FORMULA_RE.search(chunk):
|
|
return True
|
|
except (KeyError, OSError, zipfile.BadZipFile):
|
|
continue
|
|
return False
|
|
|
|
|
|
def inspect_workbook(file_path: Path) -> list[str]:
|
|
"""Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*.
|
|
|
|
Read-only inspection of the OPC package (central directory + a bounded scan
|
|
of the sheet XML). Never raises: an unreadable or encrypted workbook simply
|
|
yields ``[]`` and the save path keeps its current behaviour.
|
|
|
|
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
|
|
the cached result, so the workbook stays correct once Excel recalculates it.
|
|
"""
|
|
try:
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
names = set(zf.namelist())
|
|
found = {
|
|
key
|
|
for key, prefixes in LOSSY_PARTS.items()
|
|
if any(name.startswith(prefix) for name in names for prefix in prefixes)
|
|
}
|
|
if _has_cached_formulas(zf):
|
|
found.add("cached_values")
|
|
return sorted(found)
|
|
except (OSError, zipfile.BadZipFile):
|
|
return []
|
|
|
|
|
|
def read_workbook_meta(file_path: Path) -> dict[str, dict[str, Any]]:
|
|
"""Return ``{sheet: {styles, aligns, merges, freeze}}`` for every sheet.
|
|
|
|
One normal (non-streaming) load serves the three A15 metadata maps: the
|
|
fragments are the workbook's own values, a failure yields ``{}`` per sheet
|
|
so the viewer keeps its plain rendering. Styles are read with
|
|
``data_only=False`` — the edited value is the formula, not its result.
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), data_only=False)
|
|
except Exception:
|
|
return {}
|
|
out: dict[str, dict[str, Any]] = {}
|
|
try:
|
|
for ws in wb.worksheets:
|
|
styles, aligns = _sheet_style_maps(ws)
|
|
merged = getattr(ws, "merged_cells", None)
|
|
ranges = getattr(merged, "ranges", None) or []
|
|
out[ws.title] = {
|
|
"styles": styles,
|
|
"aligns": aligns,
|
|
"merges": [str(r) for r in ranges],
|
|
"freeze": str(getattr(ws, "freeze_panes", None) or ""),
|
|
}
|
|
return out
|
|
except Exception:
|
|
logger.debug("xlsx meta unavailable", exc_info=True)
|
|
for t in wb.sheetnames:
|
|
out.setdefault(t, {"styles": {}, "aligns": {}, "merges": [], "freeze": ""})
|
|
return out
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def render_sheets(file_path: Path) -> list[dict[str, Any]]:
|
|
"""Return one dict per sheet: ``{name, html, rows, cols, total_*, truncated}``.
|
|
|
|
Reads the workbook twice: once with ``data_only=False`` for the formulas
|
|
(what the user must edit) and, when any formula carries a cached result
|
|
(#153 A12), once with ``data_only=True`` to show what Excel last computed.
|
|
The second pass is skipped entirely when the archive holds no cached value,
|
|
so the common case still costs a single load.
|
|
|
|
``total_rows``/``total_cols`` are the dimensions the sheet declares and
|
|
``truncated`` says whether the hard caps actually cut it (#153 A8) — the
|
|
viewer needs both to stop silently hiding the tail of a sheet.
|
|
"""
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
try:
|
|
formulas = [_sheet_grid(ws) for ws in wb.worksheets]
|
|
titles = [ws.title for ws in wb.worksheets]
|
|
extents = [_sheet_extent(ws) for ws in wb.worksheets]
|
|
finally:
|
|
wb.close()
|
|
|
|
cached: list[list[list[str]]] | None = None
|
|
if _has_cached_values(file_path):
|
|
cached = _read_cached_grids(file_path, titles)
|
|
|
|
# #153 A15 — one extra normal-mode load serves the styles/merges/freeze
|
|
# metadata of every sheet; the HTML then carries the fragments itself.
|
|
meta = read_workbook_meta(file_path)
|
|
|
|
sheets = []
|
|
for i, title in enumerate(titles):
|
|
grid = _trim(formulas[i])
|
|
# The shadow grid is NOT trimmed independently: _trim drops the
|
|
# trailing empty columns of each grid on its own width, which would
|
|
# shift every cached value left of its formula. Indexing it
|
|
# positionally against the untrimmed grid keeps the two aligned.
|
|
shadow = cached[i] if cached is not None and i < len(cached) else None
|
|
total_rows, total_cols = extents[i]
|
|
sheet_meta = meta.get(title, {})
|
|
sheets.append(
|
|
{
|
|
"name": title,
|
|
"html": _table(grid, shadow, styles=sheet_meta.get("styles")),
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": total_rows,
|
|
"total_cols": total_cols,
|
|
# Coverage, not display size: `rows`/`cols` are post-trim (a
|
|
# sheet of 3 filled cells in a 500-row block renders 1x1), and
|
|
# the client must announce the cap it stopped at, not how many
|
|
# cells happen to be non-empty.
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
# A sheet is truncated when the caps, not the trailing blanks,
|
|
# decided its shape: comparing against the *rendered* size would
|
|
# flag every sheet carrying a few empty formatted rows.
|
|
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
|
"styles": sheet_meta.get("styles", {}),
|
|
"aligns": sheet_meta.get("aligns", {}),
|
|
"merges": sheet_meta.get("merges", []),
|
|
"freeze": sheet_meta.get("freeze", ""),
|
|
}
|
|
)
|
|
return sheets
|
|
|
|
|
|
def read_sheet_window(
|
|
file_path: Path,
|
|
sheet: str,
|
|
offset: int = 0,
|
|
limit: int = DEFAULT_WINDOW_ROWS,
|
|
) -> dict[str, Any] | None:
|
|
"""Return a window of rows of one sheet, or ``None`` if the sheet is unknown.
|
|
|
|
Backs the lazy per-sheet loading of #153 A9: the viewer asks for the rows
|
|
it is about to display instead of shipping every sheet in the initial file
|
|
payload. ``offset`` is 0-based; the row numbers and the ``data-cell``
|
|
references in the returned ``html`` are the real A1 coordinates of the
|
|
sheet, so a window is indistinguishable from a full render.
|
|
|
|
``limit`` is clamped to :data:`MAX_WINDOW_ROWS`. Raises nothing: an unknown
|
|
sheet yields ``None`` and a broken workbook propagates the caller's usual
|
|
500.
|
|
"""
|
|
offset = max(int(offset), 0)
|
|
limit = min(max(int(limit), 1), MAX_WINDOW_ROWS)
|
|
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return None
|
|
ws = wb[sheet]
|
|
total_rows, total_cols = _sheet_extent(ws)
|
|
grid = _trim(
|
|
_sheet_grid(ws, min_row=offset + 1, max_row=offset + limit)
|
|
)
|
|
finally:
|
|
wb.close()
|
|
|
|
shadow: list[list[str]] | None = None
|
|
# Same A12 rule as the full render: the second read only happens when the
|
|
# archive really holds cached results.
|
|
if _has_cached_values(file_path):
|
|
shadow = _read_cached_window(file_path, sheet, offset, limit)
|
|
# #153 A15 — same metadata as the full render, so a lazy window is
|
|
# indistinguishable from it (styles in the HTML, merges/freeze for the
|
|
# client-side spanning).
|
|
meta = read_workbook_meta(file_path).get(sheet, {})
|
|
return {
|
|
"sheet": sheet,
|
|
"offset": offset,
|
|
"limit": limit,
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": total_rows,
|
|
"total_cols": total_cols,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
|
"has_more": offset + len(grid) < total_rows,
|
|
"html": _table(grid, shadow, row_offset=offset, styles=meta.get("styles")),
|
|
"styles": meta.get("styles", {}),
|
|
"aligns": meta.get("aligns", {}),
|
|
"merges": meta.get("merges", []),
|
|
"freeze": meta.get("freeze", ""),
|
|
}
|
|
|
|
|
|
def _read_cached_window(
|
|
file_path: Path, sheet: str, offset: int, limit: int
|
|
) -> list[list[str]] | None:
|
|
"""``data_only=True`` grid for one window, or ``None`` if unavailable.
|
|
|
|
Best effort like :func:`_read_cached_grids`: a workbook Excel opens but
|
|
openpyxl cannot re-read must still display (formulas only).
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
return None
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return None
|
|
return _sheet_grid(
|
|
wb[sheet], min_row=offset + 1, max_row=offset + limit
|
|
)
|
|
except Exception:
|
|
logger.debug("xlsx cached window unavailable", exc_info=True)
|
|
return None
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def _sheet_extent(ws: Any) -> tuple[int, int]:
|
|
"""Rows and columns the worksheet declares, never negative.
|
|
|
|
``max_row``/``max_column`` come from the sheet's dimension record; a
|
|
hand-edited file may omit it, hence the defensive coercion.
|
|
"""
|
|
try:
|
|
rows = max(int(getattr(ws, "max_row", 0) or 0), 0)
|
|
except (TypeError, ValueError):
|
|
rows = 0
|
|
try:
|
|
cols = max(int(getattr(ws, "max_column", 0) or 0), 0)
|
|
except (TypeError, ValueError):
|
|
cols = 0
|
|
return rows, cols
|
|
|
|
|
|
def _sheet_grid(
|
|
ws: Any, min_row: int = 1, max_row: int = MAX_ROWS, max_col: int = MAX_COLS
|
|
) -> list[list[str]]:
|
|
"""Read a worksheet window into a grid of formatted strings, bounded by the caps."""
|
|
return [
|
|
[_fmt(v) for v in row]
|
|
for row in ws.iter_rows(
|
|
min_row=min_row, max_row=max_row, max_col=max_col, values_only=True
|
|
)
|
|
]
|
|
|
|
|
|
def _has_cached_values(file_path: Path) -> bool:
|
|
"""True when the archive holds at least one ``<f>…</f><v>…</v>``."""
|
|
try:
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
return _has_cached_formulas(zf)
|
|
except (OSError, zipfile.BadZipFile):
|
|
return False
|
|
|
|
|
|
def _read_cached_grids(
|
|
file_path: Path, titles: list[str]
|
|
) -> list[list[list[str]]] | None:
|
|
"""Read every sheet with ``data_only=True`` (what Excel last computed).
|
|
|
|
Best effort: returns ``None`` on any failure so the viewer falls back to the
|
|
formula-only rendering. A workbook Excel opens but openpyxl cannot re-read
|
|
must still display.
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
return None
|
|
try:
|
|
grids = [_sheet_grid(ws) for ws in wb.worksheets]
|
|
if [ws.title for ws in wb.worksheets] != titles:
|
|
return None
|
|
return grids
|
|
except Exception:
|
|
logger.debug("xlsx cached values unavailable", exc_info=True)
|
|
return None
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def extract_indexable_text(file_path: Path) -> str:
|
|
"""Return searchable text for the TF-IDF / semantic index (#153 A5).
|
|
|
|
Sheet names plus the first :data:`_INDEX_ROWS_PER_SHEET` rows of each
|
|
sheet, capped at :data:`MAX_INDEX_CHARS`. Rows are tab-joined so a search
|
|
for a header matches the sheet it belongs to.
|
|
|
|
Never raises: a corrupt, encrypted or unsupported workbook yields ``""`` so
|
|
the file still gets indexed by name (same contract as :func:`inspect_workbook`).
|
|
"""
|
|
chunks: list[str] = []
|
|
budget = MAX_INDEX_CHARS
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
# Encrypted (BadZipFile) or not a real workbook: name-only indexing.
|
|
return ""
|
|
try:
|
|
for ws in wb.worksheets[:MAX_INDEX_SHEETS]:
|
|
if budget <= 0:
|
|
break
|
|
# The sheet title alone is a strong signal ("Recettes", "Budget").
|
|
block = [ws.title]
|
|
for row in ws.iter_rows(
|
|
min_row=1, max_row=_INDEX_ROWS_PER_SHEET, max_col=MAX_COLS, values_only=True
|
|
):
|
|
cells = [_fmt(v) for v in row]
|
|
# Skip blank rows instead of emitting runs of tabs.
|
|
if not any(c.strip() for c in cells):
|
|
continue
|
|
block.append("\t".join(cells).rstrip())
|
|
text = "\n".join(block)
|
|
chunks.append(text[:budget])
|
|
budget -= len(text)
|
|
except Exception:
|
|
# Truncated but still useful: keep whatever was collected.
|
|
pass
|
|
finally:
|
|
wb.close()
|
|
return "\n".join(c for c in chunks if c).strip()
|
|
|
|
|
|
# ── #153 A17 — dashboard metadata ───────────────────────────────────
|
|
|
|
|
|
def read_workbook_dashboard(file_path: Path) -> dict[str, Any]:
|
|
"""Return the dashboard metadata of a workbook (#153 A17).
|
|
|
|
Shape::
|
|
|
|
{
|
|
"named_ranges": [{"name", "scope", "ref"}],
|
|
"objects": {"charts": int, "pivots": int},
|
|
"sheets": [{
|
|
"name": str,
|
|
"cells": int, # non-empty cells inside the caps
|
|
"rows": int, # rows carrying at least one non-empty cell
|
|
"cols": int, # columns carrying at least one non-empty cell
|
|
"formulas": int,
|
|
"numeric": int,
|
|
"kpi": [ # first 8 numeric cells as {"label", "value"}
|
|
{"label": str, "value": float}
|
|
],
|
|
}],
|
|
}
|
|
|
|
Named ranges come from the streaming load (available read-only), cell
|
|
stats from ``iter_rows(values_only=True)``. Charts/pivots are counted by
|
|
OPC part names (a chart part per chart, a pivot table part per pivot).
|
|
Bounded by MAX_ROWS/MAX_COLS; never raises — a failure yields an empty
|
|
payload and the viewer simply hides the panel.
|
|
"""
|
|
payload: dict[str, Any] = {
|
|
"named_ranges": [],
|
|
"objects": {"charts": 0, "pivots": 0},
|
|
"sheets": [],
|
|
}
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
except Exception:
|
|
return payload
|
|
try:
|
|
dn = getattr(wb, "defined_names", None)
|
|
items: list[tuple[Any, Any]] = (
|
|
list(dn.items()) if dn is not None and hasattr(dn, "items") else []
|
|
)
|
|
for name, defn in items:
|
|
scope_idx = getattr(defn, "localSheetId", None)
|
|
scope = ""
|
|
if scope_idx is not None:
|
|
try:
|
|
scope = wb.sheetnames[int(scope_idx)]
|
|
except (IndexError, ValueError):
|
|
scope = ""
|
|
payload["named_ranges"].append(
|
|
{
|
|
"name": str(name),
|
|
"scope": scope,
|
|
"ref": str(getattr(defn, "attr_text", "") or ""),
|
|
}
|
|
)
|
|
payload["named_ranges"].sort(key=lambda d: d["name"].lower())
|
|
|
|
for ws in wb.worksheets:
|
|
cells = rows = formulas = numeric = 0
|
|
col_seen: set[int] = set()
|
|
kpi: list[dict[str, Any]] = []
|
|
for r, row in enumerate(
|
|
ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS, values_only=True),
|
|
start=1,
|
|
):
|
|
row_has_value = False
|
|
for c, value in enumerate(row, start=1):
|
|
if value is None or (isinstance(value, str) and not value.strip()):
|
|
continue
|
|
cells += 1
|
|
col_seen.add(c)
|
|
row_has_value = True
|
|
if isinstance(value, str) and value.startswith("="):
|
|
formulas += 1
|
|
elif isinstance(value, bool):
|
|
pass
|
|
elif isinstance(value, (int, float)):
|
|
numeric += 1
|
|
if len(kpi) < 8:
|
|
kpi.append(
|
|
{"label": f"{get_column_letter(c)}{r}", "value": value}
|
|
)
|
|
if row_has_value:
|
|
rows += 1
|
|
payload["sheets"].append(
|
|
{
|
|
"name": ws.title,
|
|
"cells": cells,
|
|
"rows": rows,
|
|
"cols": len(col_seen),
|
|
"formulas": formulas,
|
|
"numeric": numeric,
|
|
"kpi": kpi,
|
|
}
|
|
)
|
|
# Chart/pivot parts, counted from the archive (chart XML parts are
|
|
# one per chart; pivot parts one per pivot table/cache).
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
names = zf.namelist()
|
|
payload["objects"]["charts"] = sum(1 for n in names if _CHART_PART_RE.match(n))
|
|
payload["objects"]["pivots"] = sum(1 for n in names if _PIVOT_PART_RE.match(n))
|
|
return payload
|
|
except Exception:
|
|
logger.debug("xlsx dashboard unavailable", exc_info=True)
|
|
return {
|
|
"named_ranges": [],
|
|
"objects": {"charts": 0, "pivots": 0},
|
|
"sheets": [],
|
|
}
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
# ── #153 A16 — additional spreadsheet formats ───────────────────────────────
|
|
|
|
|
|
def render_csv_table(raw: str, *, delimiter: str = ",") -> str:
|
|
"""Render CSV text as the same HTML table shape the xlsx viewer consumes.
|
|
|
|
Row numbers replace the A1 column: a CSV has no fixed column count, so
|
|
the first row is a plain data row like the others (the viewer offers the
|
|
toolbar either way). Every cell is HTML-escaped at render time.
|
|
"""
|
|
import csv as csv_mod
|
|
import io as io_mod
|
|
|
|
reader = csv_mod.reader(io_mod.StringIO(raw), delimiter=delimiter)
|
|
try:
|
|
rows = [row for row in reader]
|
|
except csv_mod.Error:
|
|
# A malformed CSV still renders: each line becomes a one-cell row.
|
|
rows = [[line] for line in raw.splitlines()]
|
|
if not rows:
|
|
return "<p><em>Feuille vide</em></p>"
|
|
n_cols = max(len(r) for r in rows)
|
|
out = [
|
|
('<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
|
'<thead><tr><th class="xlsx-corner"></th>')
|
|
]
|
|
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
|
out.append("</tr></thead><tbody>")
|
|
for r, row in enumerate(rows, start=1):
|
|
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
|
for c in range(1, n_cols + 1):
|
|
val = row[c - 1] if c - 1 < len(row) else ""
|
|
out.append(f'<td data-cell="{get_column_letter(c)}{r}">{html.escape(val)}</td>')
|
|
out.append("</tr>")
|
|
out.append("</tbody></table></div>")
|
|
return "".join(out)
|
|
|
|
|
|
def render_legacy_workbook(file_path: Path, ext: str) -> list[dict[str, Any]]:
|
|
"""Render ``.xls``/``.ods`` sheets with the same dict shape as xlsx.
|
|
|
|
Read-only formats (#153 A16): ``styles``/``aligns``/``merges``/``freeze``
|
|
are served empty so the client-side wiring keeps one code path. Raises
|
|
nothing to the render path: an unreadable file yields one error sheet.
|
|
"""
|
|
name = file_path.name
|
|
try:
|
|
if ext == ".xls":
|
|
import xlrd
|
|
|
|
book = xlrd.open_workbook(str(file_path))
|
|
titles = book.sheet_names()
|
|
grids = []
|
|
for si in range(book.nsheets):
|
|
sh = book.sheet_by_index(si)
|
|
grid = [
|
|
[_fmt(sh.cell_value(r, c)) for c in range(min(sh.ncols, MAX_COLS))]
|
|
for r in range(min(sh.nrows, MAX_ROWS))
|
|
]
|
|
grids.append(_trim(grid))
|
|
total = [(sh.nrows, sh.ncols) for sh in (book.sheet_by_index(i) for i in range(book.nsheets))]
|
|
elif ext == ".ods":
|
|
from odf.opendocument import load as odf_load
|
|
from odf.table import Table, TableCell, TableRow
|
|
from odf.teletype import extractText
|
|
|
|
doc = odf_load(str(file_path))
|
|
titles = []
|
|
grids = []
|
|
total = []
|
|
for table in doc.getElementsByType(Table):
|
|
title = table.getAttribute("name") or f"Feuille {len(titles) + 1}"
|
|
titles.append(title)
|
|
grid = []
|
|
for row in table.getElementsByType(TableRow)[:MAX_ROWS]:
|
|
row_cells = row.getElementsByType(TableCell)
|
|
values: list[str] = []
|
|
for tc in row_cells[:MAX_COLS]:
|
|
repeat = int(tc.getAttribute("numbercolumnsrepeated") or 1)
|
|
values.extend([extractText(tc)] * min(repeat, MAX_COLS - len(values)))
|
|
grid.append(values)
|
|
grids.append(_trim(grid))
|
|
total.append((len(grid), max((len(r) for r in grid), default=0)))
|
|
else:
|
|
raise ValueError(f"Unsupported legacy format: {ext}")
|
|
except Exception as exc:
|
|
logger.warning("legacy workbook render failed for %s: %s", name, exc)
|
|
return [
|
|
{
|
|
"name": name,
|
|
"html": (
|
|
'<p><em>Feuille vide</em></p>'
|
|
),
|
|
"rows": 0,
|
|
"cols": 0,
|
|
"total_rows": 0,
|
|
"total_cols": 0,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": False,
|
|
"styles": {},
|
|
"aligns": {},
|
|
"merges": [],
|
|
"freeze": "",
|
|
}
|
|
]
|
|
|
|
sheets: list[dict[str, Any]] = []
|
|
for i, title in enumerate(titles):
|
|
grid = grids[i] if i < len(grids) else []
|
|
t_rows, t_cols = total[i] if i < len(total) else (0, 0)
|
|
sheets.append(
|
|
{
|
|
"name": title,
|
|
"html": _table(grid),
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": t_rows,
|
|
"total_cols": t_cols,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": t_rows > MAX_ROWS or t_cols > MAX_COLS,
|
|
"styles": {},
|
|
"aligns": {},
|
|
"merges": [],
|
|
"freeze": "",
|
|
}
|
|
)
|
|
return sheets
|