A16 — la visionneuse tableur accepte quatre formats de plus : .xlsm est
servi et sauvegarde comme un .xlsx avec keep_vba=True (les macros
survivent, la porte lossy est levée pour ce format) ; .xls (xlrd) et .ods
(odfpy) sont rendus en lecture seule (xlsx_readonly, wiring d'édition
désactivé) ; .csv devient éditable via render_csv_table (grille A1
identique au viewer) et PUT /api/file/{vault}/csv/save (réécriture csv
RFC 4180, extension de grille, valeurs stockées telles quelles). 12 tests
backend + contre-preuve (4 échecs sur neutralisation du service CSV),
JSDOM 33/33, ruff/mypy 0.
🤖 Generated with Codebuff
Co-Authored-By: Codebuff <[email protected]>
777 lines
30 KiB
Python
777 lines
30 KiB
Python
"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx).
|
|
|
|
Read-only: formulas are shown as their text (``data_only=False``) so a
|
|
round-trip through the viewer never depends on Excel's cached values.
|
|
Write-side lives in ``backend.services.mutations.edit_xlsx_cells``.
|
|
|
|
:func:`inspect_workbook` lists the workbook features that an openpyxl
|
|
round-trip would drop (#153 A1) so the UI can warn before saving.
|
|
|
|
#153 A16 — :func:`render_sheets` also accepts ``.xlsm`` (macros preserved on
|
|
save via ``keep_vba``), ``.xls`` (xlrd) and ``.ods`` (odfpy), both served
|
|
read-only; :func:`render_csv_table` turns a CSV into the same table shape.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import html
|
|
import logging
|
|
import re
|
|
import zipfile
|
|
from datetime import date, datetime
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from openpyxl import load_workbook
|
|
from openpyxl.utils import get_column_letter
|
|
|
|
logger = logging.getLogger("obsigate.xlsx_reader")
|
|
|
|
# ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet).
|
|
# Raise them, or paginate per sheet, if a real workbook needs more.
|
|
MAX_ROWS = 500
|
|
MAX_COLS = 40
|
|
|
|
# #153 A9 — window size served by ``read_sheet_window()`` (lazy per-sheet
|
|
# loading). The endpoint is bounded so a single request can never ask for the
|
|
# whole workbook back in one JSON payload; the UI pages through the rest.
|
|
MAX_WINDOW_ROWS = 1_000
|
|
DEFAULT_WINDOW_ROWS = 200
|
|
|
|
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
|
|
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
|
|
# DO survive the round-trip, so they are deliberately absent from this map.
|
|
LOSSY_PARTS: dict[str, tuple[str, ...]] = {
|
|
"slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"),
|
|
"form_controls": ("xl/ctrlProps/", "xl/activeX/"),
|
|
"connections": ("xl/queryTables/", "xl/connections.xml"),
|
|
"custom_xml": ("customXml/",),
|
|
"signature": ("_xmlsignatures/",),
|
|
"rich_comments": ("xl/threadedComments/", "xl/persons/"),
|
|
"macros": ("xl/vbaProject.bin",),
|
|
}
|
|
|
|
# A formula cell carrying its last computed result: ``<f>…</f><v>…</v>``.
|
|
# openpyxl writes an EMPTY ``<v></v>`` itself, hence the ``[^<]`` guard: only a
|
|
# non-empty value counts. openpyxl keeps the formula but drops the cached result,
|
|
# so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until
|
|
# Excel recalculates.
|
|
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
|
|
|
|
# Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES).
|
|
_MAX_PROBE_BYTES = 8_000_000
|
|
|
|
# #153 A5 — ceiling on the text handed to the TF-IDF / semantic index. A workbook
|
|
# is a data dump, not prose: indexing every cell would flood the inverted index
|
|
# and bury the notes. Sheet names + the first rows are enough to make a
|
|
# spreadsheet findable by its headers.
|
|
MAX_INDEX_CHARS = 5_000
|
|
_INDEX_ROWS_PER_SHEET = 20
|
|
MAX_INDEX_SHEETS = 20
|
|
|
|
# #153 A15 — reading styles is a second (non-read_only) pass on the sheet XML.
|
|
# Bounded like everything else: a cell must be INSIDE the rendered window to
|
|
# deserve an inline style, so a huge workbook never triggers a huge payload.
|
|
# Only data-driven fragments are emitted: the hex values come from the file,
|
|
# never from a hardcoded color table.
|
|
|
|
|
|
def _cell_fragments(cell: Any) -> tuple[list[str], str | None]:
|
|
"""Inline CSS fragments of one cell plus its horizontal alignment.
|
|
|
|
Fixed, color-first order: the API contract documents ``color:...`` as the
|
|
first fragment of a styled cell. Only data-driven values are emitted —
|
|
every hex comes from the workbook itself, never a hardcoded table.
|
|
"""
|
|
fragments: list[str] = []
|
|
font = cell.font
|
|
if font and font.color is not None and isinstance(font.color.rgb, str):
|
|
# ARGB from the workbook itself — never a hardcoded table.
|
|
rgb = font.color.rgb
|
|
if len(rgb) == 8 and rgb != "FF000000":
|
|
fragments.append(f"color:#{rgb[2:].lower()}")
|
|
fill = cell.fill
|
|
if fill and fill.fgColor is not None and isinstance(fill.fgColor.rgb, str):
|
|
rgb = fill.fgColor.rgb
|
|
if len(rgb) == 8 and rgb not in ("00000000", "FFFFFFFF"):
|
|
fragments.append(f"background:#{rgb[2:].lower()}")
|
|
if font and font.bold:
|
|
fragments.append("font-weight:600")
|
|
if font and font.italic:
|
|
fragments.append("font-style:italic")
|
|
fmt = cell.number_format
|
|
if fmt and fmt not in ("General", "@"):
|
|
# A custom number format is signalled typographically (mono font)
|
|
# rather than rendered: the displayed value already carries the
|
|
# formatting from _fmt(). Single quotes: the fragment lands inside a
|
|
# double-quoted HTML attribute.
|
|
fragments.append("font-family:'JetBrains Mono',monospace")
|
|
alignment = cell.alignment
|
|
align = alignment.horizontal if alignment else None
|
|
return fragments, (align if align in ("left", "right", "center") else None)
|
|
|
|
|
|
def _sheet_style_maps(ws: Any) -> tuple[dict[str, str], dict[str, str]]:
|
|
"""Flat ``{ref: css}`` and ``{ref: align}`` maps of one worksheet.
|
|
|
|
The flat string is what the API serves and what the viewer applies
|
|
verbatim to ``td.style``; a plain cell is simply absent from the map.
|
|
``left`` is the table default and never included. Bounded by
|
|
``MAX_ROWS x MAX_COLS`` like the render itself.
|
|
"""
|
|
styles: dict[str, str] = {}
|
|
aligns: dict[str, str] = {}
|
|
for row in ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS):
|
|
for cell in row:
|
|
if cell.value is None and cell.number_format == "General":
|
|
continue
|
|
fragments, align = _cell_fragments(cell)
|
|
if fragments:
|
|
styles[cell.coordinate] = ";".join(fragments)
|
|
if align and align != "left":
|
|
aligns[cell.coordinate] = align
|
|
return styles, aligns
|
|
|
|
|
|
def read_sheet_styles(file_path: Path, sheet: str) -> dict[str, dict[str, Any]]:
|
|
"""Return ``{ref: {style, align}}`` for the styled cells of one sheet.
|
|
|
|
``style`` is the flat CSS fragment the viewer applies verbatim and
|
|
``align`` the horizontal text-align when it is not the table default.
|
|
Normal (non-streaming) load — styles are unavailable in read_only mode;
|
|
a failure yields ``{}`` so the viewer falls back to the plain rendering.
|
|
"""
|
|
meta = read_workbook_meta(file_path).get(sheet, {})
|
|
styles_map = meta.get("styles", {})
|
|
aligns = meta.get("aligns", {})
|
|
out: dict[str, dict[str, Any]] = {}
|
|
for ref, css in styles_map.items():
|
|
entry: dict[str, Any] = {"style": css}
|
|
if ref in aligns:
|
|
entry["align"] = aligns[ref]
|
|
out[ref] = entry
|
|
return out
|
|
|
|
|
|
def read_sheet_merges(file_path: Path, sheet: str) -> list[str]:
|
|
"""Return the merged ranges of one sheet as ``A1:C3`` strings."""
|
|
try:
|
|
# Styles and merges are only fully materialised in normal mode
|
|
# (read_only=True leaves merged_cells empty).
|
|
wb = load_workbook(str(file_path))
|
|
except Exception:
|
|
return []
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return []
|
|
merged = getattr(wb[sheet], "merged_cells", None)
|
|
ranges = getattr(merged, "ranges", None) or []
|
|
return [str(r) for r in ranges]
|
|
except Exception:
|
|
logger.debug("xlsx merges unavailable", exc_info=True)
|
|
return []
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def read_sheet_freeze(file_path: Path, sheet: str) -> str:
|
|
"""Return the freeze-panes anchor of one sheet ('' when not frozen).
|
|
|
|
Normal (non-streaming) load: `freeze_panes` is NOT materialised on
|
|
ReadOnlyWorksheet in openpyxl 3.1.x — read_only=True always yields ''.
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path))
|
|
except Exception:
|
|
return ""
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return ""
|
|
return str(getattr(wb[sheet], "freeze_panes", None) or "")
|
|
except Exception:
|
|
return ""
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def _fmt(value: Any) -> str:
|
|
if value is None:
|
|
return ""
|
|
if isinstance(value, datetime):
|
|
return value.strftime("%Y-%m-%d %H:%M")
|
|
if isinstance(value, date):
|
|
return value.isoformat()
|
|
return str(value)
|
|
|
|
|
|
def _trim(grid: list[list[str]]) -> list[list[str]]:
|
|
"""Drop trailing empty rows and columns (openpyxl pads to max_col)."""
|
|
while grid and not any(grid[-1]):
|
|
grid.pop()
|
|
if not grid:
|
|
return grid
|
|
width = 0
|
|
for row in grid:
|
|
for i in range(len(row) - 1, -1, -1):
|
|
if row[i]:
|
|
width = max(width, i + 1)
|
|
break
|
|
return [row[:width] for row in grid]
|
|
|
|
|
|
def _cell_cached(cached: list[list[str]] | None, r: int, c: int) -> str:
|
|
"""Return the cached result for a 0-based cell, or ``""``.
|
|
|
|
The shadow grid is read positionally and may be narrower than the formula
|
|
grid (``_trim`` collapses the trailing empty columns of each grid
|
|
independently), so every lookup is bounds-checked rather than assumed.
|
|
"""
|
|
if not cached or r >= len(cached):
|
|
return ""
|
|
row = cached[r]
|
|
return row[c] if c < len(row) else ""
|
|
|
|
|
|
def _table(
|
|
grid: list[list[str]],
|
|
cached: list[list[str]] | None = None,
|
|
row_offset: int = 0,
|
|
styles: dict[str, dict[str, Any]] | None = None,
|
|
) -> str:
|
|
"""Render a grid as an HTML table.
|
|
|
|
``cached`` is the same grid read with ``data_only=True`` (#153 A12): where a
|
|
formula cell still carries its last computed result, it is shown as a
|
|
discreet second line (``<span class="xlsx-cached">``) so the user sees the
|
|
number Excel last calculated instead of only the formula text. The span
|
|
carries ``data-cached-value`` and is titled client-side from
|
|
``xlsx.cached_value_title`` — the backend never emits UI text.
|
|
|
|
``row_offset`` is the number of rows skipped before this grid (#153 A9): the
|
|
row numbers and the ``data-cell`` references must stay the real A1
|
|
coordinates of the sheet, not of the window.
|
|
|
|
``styles`` maps A1 references to ``{style, align}`` fragments (#153 A15:
|
|
bold, italic, background, alignment) — the backend only reads the
|
|
workbook, the fragments are built from it and always data-driven, never
|
|
hardcoded colors. A plain ``str`` value is tolerated (legacy callers).
|
|
"""
|
|
if not grid:
|
|
return "<p><em>Feuille vide</em></p>"
|
|
n_cols = max(len(row) for row in grid)
|
|
out = [
|
|
(
|
|
'<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
|
'<thead><tr><th class="xlsx-corner"></th>'
|
|
)
|
|
]
|
|
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
|
out.append("</tr></thead><tbody>")
|
|
for r, row in enumerate(grid, start=row_offset + 1):
|
|
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
|
for c, val in enumerate(row, start=1):
|
|
ref = f"{get_column_letter(c)}{r}"
|
|
meta = (styles or {}).get(ref)
|
|
if meta is None:
|
|
style_attr = ""
|
|
else:
|
|
# Legacy callers may still pass a bare CSS string.
|
|
if isinstance(meta, str):
|
|
meta = {"style": meta}
|
|
fragment = meta.get("style", "")
|
|
align = meta.get("align")
|
|
if align and align not in ("left",):
|
|
# left is the table default; only non-default alignments
|
|
# need an explicit declaration.
|
|
fragment = f"{fragment};text-align:{align}" if fragment else f"text-align:{align}"
|
|
style_attr = f' style="{fragment}"' if fragment else ""
|
|
# The cached result only makes sense for a formula cell: on a plain
|
|
# value cell the two reads are identical and showing both would
|
|
# duplicate the text.
|
|
shadow = ""
|
|
if cached is not None and val.startswith("="):
|
|
# `c` is 1-based (A1 notation) and `r` too, while the grid is
|
|
# 0-based: translate both.
|
|
cval = _cell_cached(cached, r - 1, c - 1)
|
|
if cval and cval != val:
|
|
# The tooltip is translated client-side from
|
|
# `xlsx.cached_value_title`; never hardcode UI text here.
|
|
shadow = (
|
|
f'<span class="xlsx-cached" data-cached-value="1">'
|
|
f"{html.escape(cval)}</span>"
|
|
)
|
|
out.append(
|
|
f'<td data-cell="{ref}"{style_attr}>{html.escape(val)}{shadow}</td>'
|
|
)
|
|
out.append("</tr>")
|
|
out.append("</tbody></table></div>")
|
|
return "".join(out)
|
|
|
|
|
|
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
|
|
"""True when at least one formula cell still carries its computed value."""
|
|
budget = _MAX_PROBE_BYTES
|
|
for name in zf.namelist():
|
|
if not name.startswith("xl/worksheets/sheet") or not name.endswith(".xml"):
|
|
continue
|
|
try:
|
|
with zf.open(name) as fh:
|
|
while budget > 0:
|
|
chunk = fh.read(65536)
|
|
if not chunk:
|
|
break
|
|
budget -= len(chunk)
|
|
if _CACHED_FORMULA_RE.search(chunk):
|
|
return True
|
|
except (KeyError, OSError, zipfile.BadZipFile):
|
|
continue
|
|
return False
|
|
|
|
|
|
def inspect_workbook(file_path: Path) -> list[str]:
|
|
"""Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*.
|
|
|
|
Read-only inspection of the OPC package (central directory + a bounded scan
|
|
of the sheet XML). Never raises: an unreadable or encrypted workbook simply
|
|
yields ``[]`` and the save path keeps its current behaviour.
|
|
|
|
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
|
|
the cached result, so the workbook stays correct once Excel recalculates it.
|
|
"""
|
|
try:
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
names = set(zf.namelist())
|
|
found = {
|
|
key
|
|
for key, prefixes in LOSSY_PARTS.items()
|
|
if any(name.startswith(prefix) for name in names for prefix in prefixes)
|
|
}
|
|
if _has_cached_formulas(zf):
|
|
found.add("cached_values")
|
|
return sorted(found)
|
|
except (OSError, zipfile.BadZipFile):
|
|
return []
|
|
|
|
|
|
def read_workbook_meta(file_path: Path) -> dict[str, dict[str, Any]]:
|
|
"""Return ``{sheet: {styles, aligns, merges, freeze}}`` for every sheet.
|
|
|
|
One normal (non-streaming) load serves the three A15 metadata maps: the
|
|
fragments are the workbook's own values, a failure yields ``{}`` per sheet
|
|
so the viewer keeps its plain rendering. Styles are read with
|
|
``data_only=False`` — the edited value is the formula, not its result.
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), data_only=False)
|
|
except Exception:
|
|
return {}
|
|
out: dict[str, dict[str, Any]] = {}
|
|
try:
|
|
for ws in wb.worksheets:
|
|
styles, aligns = _sheet_style_maps(ws)
|
|
merged = getattr(ws, "merged_cells", None)
|
|
ranges = getattr(merged, "ranges", None) or []
|
|
out[ws.title] = {
|
|
"styles": styles,
|
|
"aligns": aligns,
|
|
"merges": [str(r) for r in ranges],
|
|
"freeze": str(getattr(ws, "freeze_panes", None) or ""),
|
|
}
|
|
return out
|
|
except Exception:
|
|
logger.debug("xlsx meta unavailable", exc_info=True)
|
|
for t in wb.sheetnames:
|
|
out.setdefault(t, {"styles": {}, "aligns": {}, "merges": [], "freeze": ""})
|
|
return out
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def render_sheets(file_path: Path) -> list[dict[str, Any]]:
|
|
"""Return one dict per sheet: ``{name, html, rows, cols, total_*, truncated}``.
|
|
|
|
Reads the workbook twice: once with ``data_only=False`` for the formulas
|
|
(what the user must edit) and, when any formula carries a cached result
|
|
(#153 A12), once with ``data_only=True`` to show what Excel last computed.
|
|
The second pass is skipped entirely when the archive holds no cached value,
|
|
so the common case still costs a single load.
|
|
|
|
``total_rows``/``total_cols`` are the dimensions the sheet declares and
|
|
``truncated`` says whether the hard caps actually cut it (#153 A8) — the
|
|
viewer needs both to stop silently hiding the tail of a sheet.
|
|
"""
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
try:
|
|
formulas = [_sheet_grid(ws) for ws in wb.worksheets]
|
|
titles = [ws.title for ws in wb.worksheets]
|
|
extents = [_sheet_extent(ws) for ws in wb.worksheets]
|
|
finally:
|
|
wb.close()
|
|
|
|
cached: list[list[list[str]]] | None = None
|
|
if _has_cached_values(file_path):
|
|
cached = _read_cached_grids(file_path, titles)
|
|
|
|
# #153 A15 — one extra normal-mode load serves the styles/merges/freeze
|
|
# metadata of every sheet; the HTML then carries the fragments itself.
|
|
meta = read_workbook_meta(file_path)
|
|
|
|
sheets = []
|
|
for i, title in enumerate(titles):
|
|
grid = _trim(formulas[i])
|
|
# The shadow grid is NOT trimmed independently: _trim drops the
|
|
# trailing empty columns of each grid on its own width, which would
|
|
# shift every cached value left of its formula. Indexing it
|
|
# positionally against the untrimmed grid keeps the two aligned.
|
|
shadow = cached[i] if cached is not None and i < len(cached) else None
|
|
total_rows, total_cols = extents[i]
|
|
sheet_meta = meta.get(title, {})
|
|
sheets.append(
|
|
{
|
|
"name": title,
|
|
"html": _table(grid, shadow, styles=sheet_meta.get("styles")),
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": total_rows,
|
|
"total_cols": total_cols,
|
|
# Coverage, not display size: `rows`/`cols` are post-trim (a
|
|
# sheet of 3 filled cells in a 500-row block renders 1x1), and
|
|
# the client must announce the cap it stopped at, not how many
|
|
# cells happen to be non-empty.
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
# A sheet is truncated when the caps, not the trailing blanks,
|
|
# decided its shape: comparing against the *rendered* size would
|
|
# flag every sheet carrying a few empty formatted rows.
|
|
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
|
"styles": sheet_meta.get("styles", {}),
|
|
"aligns": sheet_meta.get("aligns", {}),
|
|
"merges": sheet_meta.get("merges", []),
|
|
"freeze": sheet_meta.get("freeze", ""),
|
|
}
|
|
)
|
|
return sheets
|
|
|
|
|
|
def read_sheet_window(
|
|
file_path: Path,
|
|
sheet: str,
|
|
offset: int = 0,
|
|
limit: int = DEFAULT_WINDOW_ROWS,
|
|
) -> dict[str, Any] | None:
|
|
"""Return a window of rows of one sheet, or ``None`` if the sheet is unknown.
|
|
|
|
Backs the lazy per-sheet loading of #153 A9: the viewer asks for the rows
|
|
it is about to display instead of shipping every sheet in the initial file
|
|
payload. ``offset`` is 0-based; the row numbers and the ``data-cell``
|
|
references in the returned ``html`` are the real A1 coordinates of the
|
|
sheet, so a window is indistinguishable from a full render.
|
|
|
|
``limit`` is clamped to :data:`MAX_WINDOW_ROWS`. Raises nothing: an unknown
|
|
sheet yields ``None`` and a broken workbook propagates the caller's usual
|
|
500.
|
|
"""
|
|
offset = max(int(offset), 0)
|
|
limit = min(max(int(limit), 1), MAX_WINDOW_ROWS)
|
|
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return None
|
|
ws = wb[sheet]
|
|
total_rows, total_cols = _sheet_extent(ws)
|
|
grid = _trim(
|
|
_sheet_grid(ws, min_row=offset + 1, max_row=offset + limit)
|
|
)
|
|
finally:
|
|
wb.close()
|
|
|
|
shadow: list[list[str]] | None = None
|
|
# Same A12 rule as the full render: the second read only happens when the
|
|
# archive really holds cached results.
|
|
if _has_cached_values(file_path):
|
|
shadow = _read_cached_window(file_path, sheet, offset, limit)
|
|
# #153 A15 — same metadata as the full render, so a lazy window is
|
|
# indistinguishable from it (styles in the HTML, merges/freeze for the
|
|
# client-side spanning).
|
|
meta = read_workbook_meta(file_path).get(sheet, {})
|
|
return {
|
|
"sheet": sheet,
|
|
"offset": offset,
|
|
"limit": limit,
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": total_rows,
|
|
"total_cols": total_cols,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
|
"has_more": offset + len(grid) < total_rows,
|
|
"html": _table(grid, shadow, row_offset=offset, styles=meta.get("styles")),
|
|
"styles": meta.get("styles", {}),
|
|
"aligns": meta.get("aligns", {}),
|
|
"merges": meta.get("merges", []),
|
|
"freeze": meta.get("freeze", ""),
|
|
}
|
|
|
|
|
|
def _read_cached_window(
|
|
file_path: Path, sheet: str, offset: int, limit: int
|
|
) -> list[list[str]] | None:
|
|
"""``data_only=True`` grid for one window, or ``None`` if unavailable.
|
|
|
|
Best effort like :func:`_read_cached_grids`: a workbook Excel opens but
|
|
openpyxl cannot re-read must still display (formulas only).
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
return None
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return None
|
|
return _sheet_grid(
|
|
wb[sheet], min_row=offset + 1, max_row=offset + limit
|
|
)
|
|
except Exception:
|
|
logger.debug("xlsx cached window unavailable", exc_info=True)
|
|
return None
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def _sheet_extent(ws: Any) -> tuple[int, int]:
|
|
"""Rows and columns the worksheet declares, never negative.
|
|
|
|
``max_row``/``max_column`` come from the sheet's dimension record; a
|
|
hand-edited file may omit it, hence the defensive coercion.
|
|
"""
|
|
try:
|
|
rows = max(int(getattr(ws, "max_row", 0) or 0), 0)
|
|
except (TypeError, ValueError):
|
|
rows = 0
|
|
try:
|
|
cols = max(int(getattr(ws, "max_column", 0) or 0), 0)
|
|
except (TypeError, ValueError):
|
|
cols = 0
|
|
return rows, cols
|
|
|
|
|
|
def _sheet_grid(
|
|
ws: Any, min_row: int = 1, max_row: int = MAX_ROWS, max_col: int = MAX_COLS
|
|
) -> list[list[str]]:
|
|
"""Read a worksheet window into a grid of formatted strings, bounded by the caps."""
|
|
return [
|
|
[_fmt(v) for v in row]
|
|
for row in ws.iter_rows(
|
|
min_row=min_row, max_row=max_row, max_col=max_col, values_only=True
|
|
)
|
|
]
|
|
|
|
|
|
def _has_cached_values(file_path: Path) -> bool:
|
|
"""True when the archive holds at least one ``<f>…</f><v>…</v>``."""
|
|
try:
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
return _has_cached_formulas(zf)
|
|
except (OSError, zipfile.BadZipFile):
|
|
return False
|
|
|
|
|
|
def _read_cached_grids(
|
|
file_path: Path, titles: list[str]
|
|
) -> list[list[list[str]]] | None:
|
|
"""Read every sheet with ``data_only=True`` (what Excel last computed).
|
|
|
|
Best effort: returns ``None`` on any failure so the viewer falls back to the
|
|
formula-only rendering. A workbook Excel opens but openpyxl cannot re-read
|
|
must still display.
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
return None
|
|
try:
|
|
grids = [_sheet_grid(ws) for ws in wb.worksheets]
|
|
if [ws.title for ws in wb.worksheets] != titles:
|
|
return None
|
|
return grids
|
|
except Exception:
|
|
logger.debug("xlsx cached values unavailable", exc_info=True)
|
|
return None
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def extract_indexable_text(file_path: Path) -> str:
|
|
"""Return searchable text for the TF-IDF / semantic index (#153 A5).
|
|
|
|
Sheet names plus the first :data:`_INDEX_ROWS_PER_SHEET` rows of each
|
|
sheet, capped at :data:`MAX_INDEX_CHARS`. Rows are tab-joined so a search
|
|
for a header matches the sheet it belongs to.
|
|
|
|
Never raises: a corrupt, encrypted or unsupported workbook yields ``""`` so
|
|
the file still gets indexed by name (same contract as :func:`inspect_workbook`).
|
|
"""
|
|
chunks: list[str] = []
|
|
budget = MAX_INDEX_CHARS
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
# Encrypted (BadZipFile) or not a real workbook: name-only indexing.
|
|
return ""
|
|
try:
|
|
for ws in wb.worksheets[:MAX_INDEX_SHEETS]:
|
|
if budget <= 0:
|
|
break
|
|
# The sheet title alone is a strong signal ("Recettes", "Budget").
|
|
block = [ws.title]
|
|
for row in ws.iter_rows(
|
|
min_row=1, max_row=_INDEX_ROWS_PER_SHEET, max_col=MAX_COLS, values_only=True
|
|
):
|
|
cells = [_fmt(v) for v in row]
|
|
# Skip blank rows instead of emitting runs of tabs.
|
|
if not any(c.strip() for c in cells):
|
|
continue
|
|
block.append("\t".join(cells).rstrip())
|
|
text = "\n".join(block)
|
|
chunks.append(text[:budget])
|
|
budget -= len(text)
|
|
except Exception:
|
|
# Truncated but still useful: keep whatever was collected.
|
|
pass
|
|
finally:
|
|
wb.close()
|
|
return "\n".join(c for c in chunks if c).strip()
|
|
|
|
|
|
# ── #153 A16 — additional spreadsheet formats ───────────────────────────────
|
|
|
|
|
|
def render_csv_table(raw: str, *, delimiter: str = ",") -> str:
|
|
"""Render CSV text as the same HTML table shape the xlsx viewer consumes.
|
|
|
|
Row numbers replace the A1 column: a CSV has no fixed column count, so
|
|
the first row is a plain data row like the others (the viewer offers the
|
|
toolbar either way). Every cell is HTML-escaped at render time.
|
|
"""
|
|
import csv as csv_mod
|
|
import io as io_mod
|
|
|
|
reader = csv_mod.reader(io_mod.StringIO(raw), delimiter=delimiter)
|
|
try:
|
|
rows = [row for row in reader]
|
|
except csv_mod.Error:
|
|
# A malformed CSV still renders: each line becomes a one-cell row.
|
|
rows = [[line] for line in raw.splitlines()]
|
|
if not rows:
|
|
return "<p><em>Feuille vide</em></p>"
|
|
n_cols = max(len(r) for r in rows)
|
|
out = [
|
|
('<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
|
'<thead><tr><th class="xlsx-corner"></th>')
|
|
]
|
|
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
|
out.append("</tr></thead><tbody>")
|
|
for r, row in enumerate(rows, start=1):
|
|
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
|
for c in range(1, n_cols + 1):
|
|
val = row[c - 1] if c - 1 < len(row) else ""
|
|
out.append(f'<td data-cell="{get_column_letter(c)}{r}">{html.escape(val)}</td>')
|
|
out.append("</tr>")
|
|
out.append("</tbody></table></div>")
|
|
return "".join(out)
|
|
|
|
|
|
def render_legacy_workbook(file_path: Path, ext: str) -> list[dict[str, Any]]:
|
|
"""Render ``.xls``/``.ods`` sheets with the same dict shape as xlsx.
|
|
|
|
Read-only formats (#153 A16): ``styles``/``aligns``/``merges``/``freeze``
|
|
are served empty so the client-side wiring keeps one code path. Raises
|
|
nothing to the render path: an unreadable file yields one error sheet.
|
|
"""
|
|
name = file_path.name
|
|
try:
|
|
if ext == ".xls":
|
|
import xlrd
|
|
|
|
book = xlrd.open_workbook(str(file_path))
|
|
titles = book.sheet_names()
|
|
grids = []
|
|
for si in range(book.nsheets):
|
|
sh = book.sheet_by_index(si)
|
|
grid = [
|
|
[_fmt(sh.cell_value(r, c)) for c in range(min(sh.ncols, MAX_COLS))]
|
|
for r in range(min(sh.nrows, MAX_ROWS))
|
|
]
|
|
grids.append(_trim(grid))
|
|
total = [(sh.nrows, sh.ncols) for sh in (book.sheet_by_index(i) for i in range(book.nsheets))]
|
|
elif ext == ".ods":
|
|
from odf.opendocument import load as odf_load
|
|
from odf.table import Table, TableCell, TableRow
|
|
from odf.teletype import extractText
|
|
|
|
doc = odf_load(str(file_path))
|
|
titles = []
|
|
grids = []
|
|
total = []
|
|
for table in doc.getElementsByType(Table):
|
|
title = table.getAttribute("name") or f"Feuille {len(titles) + 1}"
|
|
titles.append(title)
|
|
grid = []
|
|
for row in table.getElementsByType(TableRow)[:MAX_ROWS]:
|
|
row_cells = row.getElementsByType(TableCell)
|
|
values: list[str] = []
|
|
for tc in row_cells[:MAX_COLS]:
|
|
repeat = int(tc.getAttribute("numbercolumnsrepeated") or 1)
|
|
values.extend([extractText(tc)] * min(repeat, MAX_COLS - len(values)))
|
|
grid.append(values)
|
|
grids.append(_trim(grid))
|
|
total.append((len(grid), max((len(r) for r in grid), default=0)))
|
|
else:
|
|
raise ValueError(f"Unsupported legacy format: {ext}")
|
|
except Exception as exc:
|
|
logger.warning("legacy workbook render failed for %s: %s", name, exc)
|
|
return [
|
|
{
|
|
"name": name,
|
|
"html": (
|
|
'<p><em>Feuille vide</em></p>'
|
|
),
|
|
"rows": 0,
|
|
"cols": 0,
|
|
"total_rows": 0,
|
|
"total_cols": 0,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": False,
|
|
"styles": {},
|
|
"aligns": {},
|
|
"merges": [],
|
|
"freeze": "",
|
|
}
|
|
]
|
|
|
|
sheets: list[dict[str, Any]] = []
|
|
for i, title in enumerate(titles):
|
|
grid = grids[i] if i < len(grids) else []
|
|
t_rows, t_cols = total[i] if i < len(total) else (0, 0)
|
|
sheets.append(
|
|
{
|
|
"name": title,
|
|
"html": _table(grid),
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": t_rows,
|
|
"total_cols": t_cols,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": t_rows > MAX_ROWS or t_cols > MAX_COLS,
|
|
"styles": {},
|
|
"aligns": {},
|
|
"merges": [],
|
|
"freeze": "",
|
|
}
|
|
)
|
|
return sheets
|