L'editeur lisait les styles mais ne les ecrivait pas, exportait la feuille entiere et laissait le dernier-ecrivain gagner entre processus. - A8 : bouton Mise en forme (gras/italique/souligne, alignements, couleurs via selecteur natif, formats de nombre, fusion, volets figes, largeur/hauteur) et nouvelle route PUT .../xlsx/style (verrou, backup, swap atomique, garde de perte, If-Match) ; A9 : decision "pas de moteur de formule" annoncee dans l'UI ; A10 : undo/redo unifie, piles par fichier conservees au re-rendu - A11 : export de la selection + Markdown/HTML/impression et recherche sur toutes les feuilles ; A12 : concurrence optimiste (If-Match -> 409 reparable, retry qui relit) ; A13 : cache des metadonnees par (chemin, mtime, taille) - A14 : outils IA .xlsm/.csv + search_workbook, analyze_range, edit_xlsx_structure - tests : test_xlsx_styles.py (17), test_spreadsheet_tools.py (34), TestOptimisticConcurrency/TestMetaCache, JSDOM xlsx-viewer 107/107
1042 lines
41 KiB
Python
1042 lines
41 KiB
Python
"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx).
|
|
|
|
Read-only: formulas are shown as their text (``data_only=False``) so a
|
|
round-trip through the viewer never depends on Excel's cached values.
|
|
Write-side lives in ``backend.services.mutations.edit_xlsx_cells``.
|
|
|
|
:func:`inspect_workbook` lists the workbook features that an openpyxl
|
|
round-trip would drop (#153 A1) so the UI can warn before saving.
|
|
|
|
#153 A16 — :func:`render_sheets` also accepts ``.xlsm`` (macros preserved on
|
|
save via ``keep_vba``), ``.xls`` (xlrd) and ``.ods`` (odfpy), both served
|
|
read-only; :func:`render_csv_table` turns a CSV into the same table shape.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
import html
|
|
import logging
|
|
import re
|
|
import threading
|
|
import zipfile
|
|
from datetime import date, datetime
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from openpyxl import load_workbook
|
|
from openpyxl.utils import get_column_letter
|
|
|
|
logger = logging.getLogger("obsigate.xlsx_reader")
|
|
|
|
# ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet).
|
|
# Raise them, or paginate per sheet, if a real workbook needs more.
|
|
MAX_ROWS = 500
|
|
MAX_COLS = 40
|
|
|
|
# BUG-094 — an empty sheet used to render as a bare "Feuille vide" paragraph
|
|
# with no cell at all, so a freshly added sheet had nothing to click and no way
|
|
# to insert a row/column. Render a small blank grid instead (Excel-like), with
|
|
# real A1 coordinates, so the cells are editable and the structure actions work.
|
|
EMPTY_SHEET_ROWS = 20
|
|
EMPTY_SHEET_COLS = 8
|
|
|
|
# #153 A9 — window size served by ``read_sheet_window()`` (lazy per-sheet
|
|
# loading). The endpoint is bounded so a single request can never ask for the
|
|
# whole workbook back in one JSON payload; the UI pages through the rest.
|
|
MAX_WINDOW_ROWS = 1_000
|
|
DEFAULT_WINDOW_ROWS = 200
|
|
|
|
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
|
|
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
|
|
# DO survive the round-trip, so they are deliberately absent from this map.
|
|
LOSSY_PARTS: dict[str, tuple[str, ...]] = {
|
|
"slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"),
|
|
"form_controls": ("xl/ctrlProps/", "xl/activeX/"),
|
|
"connections": ("xl/queryTables/", "xl/connections.xml"),
|
|
"custom_xml": ("customXml/",),
|
|
"signature": ("_xmlsignatures/",),
|
|
"rich_comments": ("xl/threadedComments/", "xl/persons/"),
|
|
"macros": ("xl/vbaProject.bin",),
|
|
}
|
|
|
|
# A formula cell carrying its last computed result: ``<f>…</f><v>…</v>``.
|
|
# openpyxl writes an EMPTY ``<v></v>`` itself, hence the ``[^<]`` guard: only a
|
|
# non-empty value counts. openpyxl keeps the formula but drops the cached result,
|
|
# so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until
|
|
# Excel recalculates.
|
|
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
|
|
|
|
# Sheet XML scanned by the cached-formula probe (CPU guard, like
|
|
# MAX_REPLACE_FILE_BYTES). BUG-099 — the allowance is PER SHEET: a single huge
|
|
# first sheet used to eat the whole budget and hide a cached formula sitting in
|
|
# the next one. The total ceiling still bounds the work on a many-sheet archive.
|
|
# Running out of budget is reported as *unverified* (never as "nothing to lose")
|
|
# so the write guard stays cautious instead of silently dropping the values.
|
|
_MAX_PROBE_BYTES_PER_SHEET = 4_000_000
|
|
_MAX_PROBE_BYTES_TOTAL = 32_000_000
|
|
|
|
# #156-A13 — metadata cache. `read_sheet_window()` serves one window at a time
|
|
# and used to reload the whole workbook (normal mode, data_only=False) for every
|
|
# window, just to read the style/merge/freeze maps of one sheet. The result is
|
|
# keyed by (path, mtime_ns, size): any write replaces the file, hence the key.
|
|
_META_CACHE_MAX = 8
|
|
_meta_cache: dict[str, tuple[tuple[int, int], dict[str, dict[str, Any]]]] = {}
|
|
_meta_cache_lock = threading.Lock()
|
|
|
|
# #153 A17 — OPC parts of chart / pivot objects, matched against the archive
|
|
# name list (xl/charts/chart1.xml, xl/pivotTables/pivotTable1.xml, …).
|
|
_CHART_PART_RE = re.compile(r"^xl/charts/chart\d+\.xml$")
|
|
_PIVOT_PART_RE = re.compile(r"^xl/pivotTables/pivotTable\d+\.xml$")
|
|
|
|
# #153 A5 — ceiling on the text handed to the TF-IDF / semantic index. A workbook
|
|
# is a data dump, not prose: indexing every cell would flood the inverted index
|
|
# and bury the notes. Sheet names + the first rows are enough to make a
|
|
# spreadsheet findable by its headers.
|
|
MAX_INDEX_CHARS = 5_000
|
|
_INDEX_ROWS_PER_SHEET = 20
|
|
MAX_INDEX_SHEETS = 20
|
|
|
|
# #153 A15 — reading styles is a second (non-read_only) pass on the sheet XML.
|
|
# Bounded like everything else: a cell must be INSIDE the rendered window to
|
|
# deserve an inline style, so a huge workbook never triggers a huge payload.
|
|
# Only data-driven fragments are emitted: the hex values come from the file,
|
|
# never from a hardcoded color table.
|
|
|
|
|
|
def _cell_fragments(cell: Any) -> tuple[list[str], str | None]:
|
|
"""Inline CSS fragments of one cell plus its horizontal alignment.
|
|
|
|
Fixed, color-first order: the API contract documents ``color:...`` as the
|
|
first fragment of a styled cell. Only data-driven values are emitted —
|
|
every hex comes from the workbook itself, never a hardcoded table.
|
|
"""
|
|
fragments: list[str] = []
|
|
font = cell.font
|
|
if font and font.color is not None and isinstance(font.color.rgb, str):
|
|
# ARGB from the workbook itself — never a hardcoded table.
|
|
rgb = font.color.rgb
|
|
if len(rgb) == 8 and rgb != "FF000000":
|
|
fragments.append(f"color:#{rgb[2:].lower()}")
|
|
fill = cell.fill
|
|
if fill and fill.fgColor is not None and isinstance(fill.fgColor.rgb, str):
|
|
rgb = fill.fgColor.rgb
|
|
if len(rgb) == 8 and rgb not in ("00000000", "FFFFFFFF"):
|
|
fragments.append(f"background:#{rgb[2:].lower()}")
|
|
if font and font.bold:
|
|
fragments.append("font-weight:600")
|
|
if font and font.italic:
|
|
fragments.append("font-style:italic")
|
|
# #156-A8 — underline is written from the viewer too, so it is read back
|
|
# (the toggle in the formatting menu needs to see its own effect).
|
|
if font and font.underline:
|
|
fragments.append("text-decoration:underline")
|
|
fmt = cell.number_format
|
|
if fmt and fmt not in ("General", "@"):
|
|
# A custom number format is signalled typographically (mono font)
|
|
# rather than rendered: the displayed value already carries the
|
|
# formatting from _fmt(). Single quotes: the fragment lands inside a
|
|
# double-quoted HTML attribute.
|
|
fragments.append("font-family:'JetBrains Mono',monospace")
|
|
alignment = cell.alignment
|
|
align = alignment.horizontal if alignment else None
|
|
return fragments, (align if align in ("left", "right", "center") else None)
|
|
|
|
|
|
def _sheet_style_maps(ws: Any) -> tuple[dict[str, str], dict[str, str]]:
|
|
"""Flat ``{ref: css}`` and ``{ref: align}`` maps of one worksheet.
|
|
|
|
The flat string is what the API serves and what the viewer applies
|
|
verbatim to ``td.style``; a plain cell is simply absent from the map.
|
|
``left`` is the table default and never included. Bounded by
|
|
``MAX_ROWS x MAX_COLS`` like the render itself.
|
|
"""
|
|
styles: dict[str, str] = {}
|
|
aligns: dict[str, str] = {}
|
|
for row in ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS):
|
|
for cell in row:
|
|
if cell.value is None and cell.number_format == "General":
|
|
continue
|
|
fragments, align = _cell_fragments(cell)
|
|
if fragments:
|
|
styles[cell.coordinate] = ";".join(fragments)
|
|
if align and align != "left":
|
|
aligns[cell.coordinate] = align
|
|
return styles, aligns
|
|
|
|
|
|
def read_sheet_styles(file_path: Path, sheet: str) -> dict[str, dict[str, Any]]:
|
|
"""Return ``{ref: {style, align}}`` for the styled cells of one sheet.
|
|
|
|
``style`` is the flat CSS fragment the viewer applies verbatim and
|
|
``align`` the horizontal text-align when it is not the table default.
|
|
Normal (non-streaming) load — styles are unavailable in read_only mode;
|
|
a failure yields ``{}`` so the viewer falls back to the plain rendering.
|
|
"""
|
|
meta = read_workbook_meta(file_path).get(sheet, {})
|
|
styles_map = meta.get("styles", {})
|
|
aligns = meta.get("aligns", {})
|
|
out: dict[str, dict[str, Any]] = {}
|
|
for ref, css in styles_map.items():
|
|
entry: dict[str, Any] = {"style": css}
|
|
if ref in aligns:
|
|
entry["align"] = aligns[ref]
|
|
out[ref] = entry
|
|
return out
|
|
|
|
|
|
def read_sheet_merges(file_path: Path, sheet: str) -> list[str]:
|
|
"""Return the merged ranges of one sheet as ``A1:C3`` strings."""
|
|
try:
|
|
# Styles and merges are only fully materialised in normal mode
|
|
# (read_only=True leaves merged_cells empty).
|
|
wb = load_workbook(str(file_path))
|
|
except Exception:
|
|
return []
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return []
|
|
merged = getattr(wb[sheet], "merged_cells", None)
|
|
ranges = getattr(merged, "ranges", None) or []
|
|
return [str(r) for r in ranges]
|
|
except Exception:
|
|
logger.debug("xlsx merges unavailable", exc_info=True)
|
|
return []
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def read_sheet_freeze(file_path: Path, sheet: str) -> str:
|
|
"""Return the freeze-panes anchor of one sheet ('' when not frozen).
|
|
|
|
Normal (non-streaming) load: `freeze_panes` is NOT materialised on
|
|
ReadOnlyWorksheet in openpyxl 3.1.x — read_only=True always yields ''.
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path))
|
|
except Exception:
|
|
return ""
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return ""
|
|
return str(getattr(wb[sheet], "freeze_panes", None) or "")
|
|
except Exception:
|
|
return ""
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def _fmt(value: Any) -> str:
|
|
if value is None:
|
|
return ""
|
|
if isinstance(value, datetime):
|
|
return value.strftime("%Y-%m-%d %H:%M")
|
|
if isinstance(value, date):
|
|
return value.isoformat()
|
|
return str(value)
|
|
|
|
|
|
def _trim(grid: list[list[str]]) -> list[list[str]]:
|
|
"""Drop trailing empty rows and columns (openpyxl pads to max_col)."""
|
|
while grid and not any(grid[-1]):
|
|
grid.pop()
|
|
if not grid:
|
|
return grid
|
|
width = 0
|
|
for row in grid:
|
|
for i in range(len(row) - 1, -1, -1):
|
|
if row[i]:
|
|
width = max(width, i + 1)
|
|
break
|
|
return [row[:width] for row in grid]
|
|
|
|
|
|
def _cell_cached(cached: list[list[str]] | None, r: int, c: int) -> str:
|
|
"""Return the cached result for a 0-based cell, or ``""``.
|
|
|
|
The shadow grid is read positionally and may be narrower than the formula
|
|
grid (``_trim`` collapses the trailing empty columns of each grid
|
|
independently), so every lookup is bounds-checked rather than assumed.
|
|
"""
|
|
if not cached or r >= len(cached):
|
|
return ""
|
|
row = cached[r]
|
|
return row[c] if c < len(row) else ""
|
|
|
|
|
|
def _table(
|
|
grid: list[list[str]],
|
|
cached: list[list[str]] | None = None,
|
|
row_offset: int = 0,
|
|
styles: dict[str, dict[str, Any]] | None = None,
|
|
) -> str:
|
|
"""Render a grid as an HTML table.
|
|
|
|
``cached`` is the same grid read with ``data_only=True`` (#153 A12): where a
|
|
formula cell still carries its last computed result, it is shown as a
|
|
discreet second line (``<span class="xlsx-cached">``) so the user sees the
|
|
number Excel last calculated instead of only the formula text. The span
|
|
carries ``data-cached-value`` and is titled client-side from
|
|
``xlsx.cached_value_title`` — the backend never emits UI text.
|
|
|
|
``row_offset`` is the number of rows skipped before this grid (#153 A9): the
|
|
row numbers and the ``data-cell`` references must stay the real A1
|
|
coordinates of the sheet, not of the window.
|
|
|
|
``styles`` maps A1 references to ``{style, align}`` fragments (#153 A15:
|
|
bold, italic, background, alignment) — the backend only reads the
|
|
workbook, the fragments are built from it and always data-driven, never
|
|
hardcoded colors. A plain ``str`` value is tolerated (legacy callers).
|
|
"""
|
|
if not grid:
|
|
return "<p><em>Feuille vide</em></p>"
|
|
n_cols = max(len(row) for row in grid)
|
|
out = [
|
|
(
|
|
'<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
|
'<thead><tr><th class="xlsx-corner"></th>'
|
|
)
|
|
]
|
|
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
|
out.append("</tr></thead><tbody>")
|
|
for r, row in enumerate(grid, start=row_offset + 1):
|
|
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
|
for c, val in enumerate(row, start=1):
|
|
ref = f"{get_column_letter(c)}{r}"
|
|
meta = (styles or {}).get(ref)
|
|
if meta is None:
|
|
style_attr = ""
|
|
else:
|
|
# Legacy callers may still pass a bare CSS string.
|
|
if isinstance(meta, str):
|
|
meta = {"style": meta}
|
|
fragment = meta.get("style", "")
|
|
align = meta.get("align")
|
|
if align and align not in ("left",):
|
|
# left is the table default; only non-default alignments
|
|
# need an explicit declaration.
|
|
fragment = f"{fragment};text-align:{align}" if fragment else f"text-align:{align}"
|
|
style_attr = f' style="{fragment}"' if fragment else ""
|
|
# The cached result only makes sense for a formula cell: on a plain
|
|
# value cell the two reads are identical and showing both would
|
|
# duplicate the text.
|
|
shadow = ""
|
|
if cached is not None and val.startswith("="):
|
|
# `c` is 1-based (A1 notation) and `r` too, while the grid is
|
|
# 0-based: translate both.
|
|
cval = _cell_cached(cached, r - 1, c - 1)
|
|
if cval and cval != val:
|
|
# The tooltip is translated client-side from
|
|
# `xlsx.cached_value_title`; never hardcode UI text here.
|
|
shadow = (
|
|
f'<span class="xlsx-cached" data-cached-value="1">'
|
|
f"{html.escape(cval)}</span>"
|
|
)
|
|
out.append(
|
|
f'<td data-cell="{ref}"{style_attr}>{html.escape(val)}{shadow}</td>'
|
|
)
|
|
out.append("</tr>")
|
|
out.append("</tbody></table></div>")
|
|
return "".join(out)
|
|
|
|
|
|
def _is_sheet_xml(name: str) -> bool:
|
|
"""True for the worksheet XML parts the cached-formula probe scans."""
|
|
return name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
|
|
|
|
|
|
def _scan_cached_formulas(zf: zipfile.ZipFile) -> tuple[bool, bool]:
|
|
"""Scan the sheet XML for a formula carrying a non-empty cached result.
|
|
|
|
Returns ``(found, unverified)``. ``unverified`` is True when the byte
|
|
budget stopped the scan before every sheet could be read to the end: a
|
|
negative result is then **not** proof that the workbook holds no cached
|
|
value (BUG-099), so callers must not treat it as a licence to write.
|
|
|
|
Each sheet gets its own :data:`_MAX_PROBE_BYTES_PER_SHEET` allowance (a
|
|
single huge sheet can no longer starve the others) while
|
|
:data:`_MAX_PROBE_BYTES_TOTAL` bounds the whole archive. A hit short-
|
|
circuits the scan: the answer is already known.
|
|
"""
|
|
budget = _MAX_PROBE_BYTES_TOTAL
|
|
unverified = False
|
|
for name in zf.namelist():
|
|
if not _is_sheet_xml(name):
|
|
continue
|
|
sheet_budget = min(_MAX_PROBE_BYTES_PER_SHEET, budget)
|
|
exhausted = False
|
|
try:
|
|
with zf.open(name) as fh:
|
|
while sheet_budget > 0:
|
|
# Read at most what the sheet's allowance has left, so one
|
|
# large chunk can never consume the whole total budget.
|
|
chunk = fh.read(min(65536, sheet_budget))
|
|
if not chunk:
|
|
break # read to the end: this sheet is verified clean
|
|
sheet_budget -= len(chunk)
|
|
budget -= len(chunk)
|
|
if _CACHED_FORMULA_RE.search(chunk):
|
|
return True, False
|
|
else:
|
|
# Left the loop on the budget, not on EOF.
|
|
exhausted = True
|
|
except (KeyError, OSError, zipfile.BadZipFile):
|
|
continue
|
|
if exhausted:
|
|
unverified = True
|
|
if budget <= 0:
|
|
unverified = True
|
|
break
|
|
return False, unverified
|
|
|
|
|
|
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
|
|
"""True when at least one formula cell still carries its computed value.
|
|
|
|
Boolean view of :func:`_scan_cached_formulas` for the display path (the
|
|
second ``data_only=True`` read): a truncated scan simply skips the shadow
|
|
grid, it never claims the workbook is lossless.
|
|
"""
|
|
found, _ = _scan_cached_formulas(zf)
|
|
return found
|
|
|
|
|
|
def inspect_workbook(file_path: Path) -> list[str]:
|
|
"""Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*.
|
|
|
|
Read-only inspection of the OPC package (central directory + a bounded scan
|
|
of the sheet XML). Never raises: an unreadable or encrypted workbook simply
|
|
yields ``[]`` and the save path keeps its current behaviour.
|
|
|
|
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
|
|
the cached result, so the workbook stays correct once Excel recalculates it.
|
|
``cached_values_unverified`` (BUG-099) is the other synthetic key: the
|
|
cached-value probe ran out of budget, so a negative result is not proof —
|
|
the entry keeps the write guard cautious (409 + confirmation) rather than
|
|
promising a lossless round-trip it cannot vouch for.
|
|
"""
|
|
try:
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
names = set(zf.namelist())
|
|
found = {
|
|
key
|
|
for key, prefixes in LOSSY_PARTS.items()
|
|
if any(name.startswith(prefix) for name in names for prefix in prefixes)
|
|
}
|
|
cached_found, cached_unverified = _scan_cached_formulas(zf)
|
|
if cached_found:
|
|
found.add("cached_values")
|
|
elif cached_unverified:
|
|
found.add("cached_values_unverified")
|
|
return sorted(found)
|
|
except (OSError, zipfile.BadZipFile):
|
|
return []
|
|
|
|
|
|
def invalidate_workbook_meta(file_path: Path | str) -> None:
|
|
"""Drop the cached metadata of one workbook (#156-A13).
|
|
|
|
Called by the write path right after the atomic replace: the (mtime, size)
|
|
key already changes on a rewrite, this only closes the theoretical window
|
|
where a same-size write lands on the same timestamp tick.
|
|
"""
|
|
with _meta_cache_lock:
|
|
_meta_cache.pop(str(file_path), None)
|
|
|
|
|
|
def read_workbook_meta(file_path: Path) -> dict[str, dict[str, Any]]:
|
|
"""Return ``{sheet: {styles, aligns, merges, freeze}}`` for every sheet.
|
|
|
|
One normal (non-streaming) load serves the three A15 metadata maps: the
|
|
fragments are the workbook's own values, a failure yields ``{}`` per sheet
|
|
so the viewer keeps its plain rendering. Styles are read with
|
|
``data_only=False`` — the edited value is the formula, not its result.
|
|
|
|
#156-A13 — the result is cached on ``(path, mtime_ns, size)``: a truncated
|
|
sheet is fetched window by window, and each window used to pay a full
|
|
workbook load for these maps alone. Callers only read from the mapping.
|
|
"""
|
|
try:
|
|
st = file_path.stat()
|
|
except OSError:
|
|
return {}
|
|
key = str(file_path)
|
|
stamp = (st.st_mtime_ns, st.st_size)
|
|
with _meta_cache_lock:
|
|
hit = _meta_cache.get(key)
|
|
if hit and hit[0] == stamp:
|
|
return hit[1]
|
|
out = _read_workbook_meta_uncached(file_path)
|
|
with _meta_cache_lock:
|
|
_meta_cache[key] = (stamp, out)
|
|
while len(_meta_cache) > _META_CACHE_MAX:
|
|
_meta_cache.pop(next(iter(_meta_cache)))
|
|
return out
|
|
|
|
|
|
def _read_workbook_meta_uncached(file_path: Path) -> dict[str, dict[str, Any]]:
|
|
"""Load the workbook once and build the per-sheet metadata maps."""
|
|
try:
|
|
wb = load_workbook(str(file_path), data_only=False)
|
|
except Exception:
|
|
return {}
|
|
out: dict[str, dict[str, Any]] = {}
|
|
try:
|
|
for ws in wb.worksheets:
|
|
styles, aligns = _sheet_style_maps(ws)
|
|
merged = getattr(ws, "merged_cells", None)
|
|
ranges = getattr(merged, "ranges", None) or []
|
|
out[ws.title] = {
|
|
"styles": styles,
|
|
"aligns": aligns,
|
|
"merges": [str(r) for r in ranges],
|
|
"freeze": str(getattr(ws, "freeze_panes", None) or ""),
|
|
}
|
|
return out
|
|
except Exception:
|
|
logger.debug("xlsx meta unavailable", exc_info=True)
|
|
for t in wb.sheetnames:
|
|
out.setdefault(t, {"styles": {}, "aligns": {}, "merges": [], "freeze": ""})
|
|
return out
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def render_sheets(file_path: Path) -> list[dict[str, Any]]:
|
|
"""Return one dict per sheet: ``{name, html, rows, cols, total_*, truncated}``.
|
|
|
|
Reads the workbook twice: once with ``data_only=False`` for the formulas
|
|
(what the user must edit) and, when any formula carries a cached result
|
|
(#153 A12), once with ``data_only=True`` to show what Excel last computed.
|
|
The second pass is skipped entirely when the archive holds no cached value,
|
|
so the common case still costs a single load.
|
|
|
|
``total_rows``/``total_cols`` are the dimensions the sheet declares and
|
|
``truncated`` says whether the hard caps actually cut it (#153 A8) — the
|
|
viewer needs both to stop silently hiding the tail of a sheet.
|
|
"""
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
try:
|
|
formulas = [_sheet_grid(ws) for ws in wb.worksheets]
|
|
titles = [ws.title for ws in wb.worksheets]
|
|
extents = [_sheet_extent(ws) for ws in wb.worksheets]
|
|
finally:
|
|
wb.close()
|
|
|
|
cached: list[list[list[str]]] | None = None
|
|
if _has_cached_values(file_path):
|
|
cached = _read_cached_grids(file_path, titles)
|
|
|
|
# #153 A15 — one extra normal-mode load serves the styles/merges/freeze
|
|
# metadata of every sheet; the HTML then carries the fragments itself.
|
|
meta = read_workbook_meta(file_path)
|
|
|
|
sheets = []
|
|
for i, title in enumerate(titles):
|
|
grid = _trim(formulas[i])
|
|
# BUG-094 — a blank sheet still needs an editable grid (see constants):
|
|
# the viewer's cell editing and structure actions all hang off a cell.
|
|
if not grid:
|
|
grid = [[""] * EMPTY_SHEET_COLS for _ in range(EMPTY_SHEET_ROWS)]
|
|
# The shadow grid is NOT trimmed independently: _trim drops the
|
|
# trailing empty columns of each grid on its own width, which would
|
|
# shift every cached value left of its formula. Indexing it
|
|
# positionally against the untrimmed grid keeps the two aligned.
|
|
shadow = cached[i] if cached is not None and i < len(cached) else None
|
|
total_rows, total_cols = extents[i]
|
|
sheet_meta = meta.get(title, {})
|
|
sheets.append(
|
|
{
|
|
"name": title,
|
|
"html": _table(grid, shadow, styles=sheet_meta.get("styles")),
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": total_rows,
|
|
"total_cols": total_cols,
|
|
# Coverage, not display size: `rows`/`cols` are post-trim (a
|
|
# sheet of 3 filled cells in a 500-row block renders 1x1), and
|
|
# the client must announce the cap it stopped at, not how many
|
|
# cells happen to be non-empty.
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
# A sheet is truncated when the caps, not the trailing blanks,
|
|
# decided its shape: comparing against the *rendered* size would
|
|
# flag every sheet carrying a few empty formatted rows.
|
|
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
|
"styles": sheet_meta.get("styles", {}),
|
|
"aligns": sheet_meta.get("aligns", {}),
|
|
"merges": sheet_meta.get("merges", []),
|
|
"freeze": sheet_meta.get("freeze", ""),
|
|
}
|
|
)
|
|
return sheets
|
|
|
|
|
|
def read_sheet_window(
|
|
file_path: Path,
|
|
sheet: str,
|
|
offset: int = 0,
|
|
limit: int = DEFAULT_WINDOW_ROWS,
|
|
) -> dict[str, Any] | None:
|
|
"""Return a window of rows of one sheet, or ``None`` if the sheet is unknown.
|
|
|
|
Backs the lazy per-sheet loading of #153 A9: the viewer asks for the rows
|
|
it is about to display instead of shipping every sheet in the initial file
|
|
payload. ``offset`` is 0-based; the row numbers and the ``data-cell``
|
|
references in the returned ``html`` are the real A1 coordinates of the
|
|
sheet, so a window is indistinguishable from a full render.
|
|
|
|
``limit`` is clamped to :data:`MAX_WINDOW_ROWS`. Raises nothing: an unknown
|
|
sheet yields ``None`` and a broken workbook propagates the caller's usual
|
|
500.
|
|
"""
|
|
offset = max(int(offset), 0)
|
|
limit = min(max(int(limit), 1), MAX_WINDOW_ROWS)
|
|
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return None
|
|
ws = wb[sheet]
|
|
total_rows, total_cols = _sheet_extent(ws)
|
|
grid = _trim(
|
|
_sheet_grid(ws, min_row=offset + 1, max_row=offset + limit)
|
|
)
|
|
finally:
|
|
wb.close()
|
|
|
|
shadow: list[list[str]] | None = None
|
|
# Same A12 rule as the full render: the second read only happens when the
|
|
# archive really holds cached results.
|
|
if _has_cached_values(file_path):
|
|
shadow = _read_cached_window(file_path, sheet, offset, limit)
|
|
# #153 A15 — same metadata as the full render, so a lazy window is
|
|
# indistinguishable from it (styles in the HTML, merges/freeze for the
|
|
# client-side spanning).
|
|
meta = read_workbook_meta(file_path).get(sheet, {})
|
|
return {
|
|
"sheet": sheet,
|
|
"offset": offset,
|
|
"limit": limit,
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": total_rows,
|
|
"total_cols": total_cols,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
|
"has_more": offset + len(grid) < total_rows,
|
|
"html": _table(grid, shadow, row_offset=offset, styles=meta.get("styles")),
|
|
"styles": meta.get("styles", {}),
|
|
"aligns": meta.get("aligns", {}),
|
|
"merges": meta.get("merges", []),
|
|
"freeze": meta.get("freeze", ""),
|
|
}
|
|
|
|
|
|
def _read_cached_window(
|
|
file_path: Path, sheet: str, offset: int, limit: int
|
|
) -> list[list[str]] | None:
|
|
"""``data_only=True`` grid for one window, or ``None`` if unavailable.
|
|
|
|
Best effort like :func:`_read_cached_grids`: a workbook Excel opens but
|
|
openpyxl cannot re-read must still display (formulas only).
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
return None
|
|
try:
|
|
if sheet not in wb.sheetnames:
|
|
return None
|
|
return _sheet_grid(
|
|
wb[sheet], min_row=offset + 1, max_row=offset + limit
|
|
)
|
|
except Exception:
|
|
logger.debug("xlsx cached window unavailable", exc_info=True)
|
|
return None
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def _sheet_extent(ws: Any) -> tuple[int, int]:
|
|
"""Rows and columns the worksheet declares, never negative.
|
|
|
|
``max_row``/``max_column`` come from the sheet's dimension record; a
|
|
hand-edited file may omit it, hence the defensive coercion.
|
|
"""
|
|
try:
|
|
rows = max(int(getattr(ws, "max_row", 0) or 0), 0)
|
|
except (TypeError, ValueError):
|
|
rows = 0
|
|
try:
|
|
cols = max(int(getattr(ws, "max_column", 0) or 0), 0)
|
|
except (TypeError, ValueError):
|
|
cols = 0
|
|
return rows, cols
|
|
|
|
|
|
def _sheet_grid(
|
|
ws: Any, min_row: int = 1, max_row: int = MAX_ROWS, max_col: int = MAX_COLS
|
|
) -> list[list[str]]:
|
|
"""Read a worksheet window into a grid of formatted strings, bounded by the caps."""
|
|
return [
|
|
[_fmt(v) for v in row]
|
|
for row in ws.iter_rows(
|
|
min_row=min_row, max_row=max_row, max_col=max_col, values_only=True
|
|
)
|
|
]
|
|
|
|
|
|
def _has_cached_values(file_path: Path) -> bool:
|
|
"""True when the archive holds at least one ``<f>…</f><v>…</v>``."""
|
|
try:
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
return _has_cached_formulas(zf)
|
|
except (OSError, zipfile.BadZipFile):
|
|
return False
|
|
|
|
|
|
def _read_cached_grids(
|
|
file_path: Path, titles: list[str]
|
|
) -> list[list[list[str]]] | None:
|
|
"""Read every sheet with ``data_only=True`` (what Excel last computed).
|
|
|
|
Best effort: returns ``None`` on any failure so the viewer falls back to the
|
|
formula-only rendering. A workbook Excel opens but openpyxl cannot re-read
|
|
must still display.
|
|
"""
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
return None
|
|
try:
|
|
grids = [_sheet_grid(ws) for ws in wb.worksheets]
|
|
if [ws.title for ws in wb.worksheets] != titles:
|
|
return None
|
|
return grids
|
|
except Exception:
|
|
logger.debug("xlsx cached values unavailable", exc_info=True)
|
|
return None
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
def extract_indexable_text(file_path: Path) -> str:
|
|
"""Return searchable text for the TF-IDF / semantic index (#153 A5).
|
|
|
|
Sheet names plus the first :data:`_INDEX_ROWS_PER_SHEET` rows of each
|
|
sheet, capped at :data:`MAX_INDEX_CHARS`. Rows are tab-joined so a search
|
|
for a header matches the sheet it belongs to.
|
|
|
|
Never raises: a corrupt, encrypted or unsupported workbook yields ``""`` so
|
|
the file still gets indexed by name (same contract as :func:`inspect_workbook`).
|
|
"""
|
|
chunks: list[str] = []
|
|
budget = MAX_INDEX_CHARS
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
|
except Exception:
|
|
# Encrypted (BadZipFile) or not a real workbook: name-only indexing.
|
|
return ""
|
|
try:
|
|
for ws in wb.worksheets[:MAX_INDEX_SHEETS]:
|
|
if budget <= 0:
|
|
break
|
|
# The sheet title alone is a strong signal ("Recettes", "Budget").
|
|
block = [ws.title]
|
|
for row in ws.iter_rows(
|
|
min_row=1, max_row=_INDEX_ROWS_PER_SHEET, max_col=MAX_COLS, values_only=True
|
|
):
|
|
cells = [_fmt(v) for v in row]
|
|
# Skip blank rows instead of emitting runs of tabs.
|
|
if not any(c.strip() for c in cells):
|
|
continue
|
|
block.append("\t".join(cells).rstrip())
|
|
text = "\n".join(block)
|
|
chunks.append(text[:budget])
|
|
budget -= len(text)
|
|
except Exception:
|
|
# Truncated but still useful: keep whatever was collected.
|
|
pass
|
|
finally:
|
|
wb.close()
|
|
return "\n".join(c for c in chunks if c).strip()
|
|
|
|
|
|
# ── #153 A17 — dashboard metadata ───────────────────────────────────
|
|
|
|
|
|
def read_workbook_dashboard(file_path: Path) -> dict[str, Any]:
|
|
"""Return the dashboard metadata of a workbook (#153 A17).
|
|
|
|
Shape::
|
|
|
|
{
|
|
"named_ranges": [{"name", "scope", "ref"}],
|
|
"objects": {"charts": int, "pivots": int},
|
|
"sheets": [{
|
|
"name": str,
|
|
"cells": int, # non-empty cells inside the caps
|
|
"rows": int, # rows carrying at least one non-empty cell
|
|
"cols": int, # columns carrying at least one non-empty cell
|
|
"formulas": int,
|
|
"numeric": int,
|
|
"kpi": [ # first 8 numeric cells as {"label", "value"}
|
|
{"label": str, "value": float}
|
|
],
|
|
}],
|
|
}
|
|
|
|
Named ranges come from the streaming load (available read-only), cell
|
|
stats from ``iter_rows(values_only=True)``. Charts/pivots are counted by
|
|
OPC part names (a chart part per chart, a pivot table part per pivot).
|
|
Bounded by MAX_ROWS/MAX_COLS; never raises — a failure yields an empty
|
|
payload and the viewer simply hides the panel.
|
|
"""
|
|
payload: dict[str, Any] = {
|
|
"named_ranges": [],
|
|
"objects": {"charts": 0, "pivots": 0},
|
|
"sheets": [],
|
|
}
|
|
try:
|
|
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
|
except Exception:
|
|
return payload
|
|
try:
|
|
dn = getattr(wb, "defined_names", None)
|
|
items: list[tuple[Any, Any]] = (
|
|
list(dn.items()) if dn is not None and hasattr(dn, "items") else []
|
|
)
|
|
for name, defn in items:
|
|
scope_idx = getattr(defn, "localSheetId", None)
|
|
scope = ""
|
|
if scope_idx is not None:
|
|
try:
|
|
scope = wb.sheetnames[int(scope_idx)]
|
|
except (IndexError, ValueError):
|
|
scope = ""
|
|
payload["named_ranges"].append(
|
|
{
|
|
"name": str(name),
|
|
"scope": scope,
|
|
"ref": str(getattr(defn, "attr_text", "") or ""),
|
|
}
|
|
)
|
|
payload["named_ranges"].sort(key=lambda d: d["name"].lower())
|
|
|
|
for ws in wb.worksheets:
|
|
cells = rows = formulas = numeric = 0
|
|
col_seen: set[int] = set()
|
|
kpi: list[dict[str, Any]] = []
|
|
for r, row in enumerate(
|
|
ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS, values_only=True),
|
|
start=1,
|
|
):
|
|
row_has_value = False
|
|
for c, value in enumerate(row, start=1):
|
|
if value is None or (isinstance(value, str) and not value.strip()):
|
|
continue
|
|
cells += 1
|
|
col_seen.add(c)
|
|
row_has_value = True
|
|
if isinstance(value, str) and value.startswith("="):
|
|
formulas += 1
|
|
elif isinstance(value, bool):
|
|
pass
|
|
elif isinstance(value, (int, float)):
|
|
numeric += 1
|
|
if len(kpi) < 8:
|
|
kpi.append(
|
|
{"label": f"{get_column_letter(c)}{r}", "value": value}
|
|
)
|
|
if row_has_value:
|
|
rows += 1
|
|
payload["sheets"].append(
|
|
{
|
|
"name": ws.title,
|
|
"cells": cells,
|
|
"rows": rows,
|
|
"cols": len(col_seen),
|
|
"formulas": formulas,
|
|
"numeric": numeric,
|
|
"kpi": kpi,
|
|
}
|
|
)
|
|
# Chart/pivot parts, counted from the archive (chart XML parts are
|
|
# one per chart; pivot parts one per pivot table/cache).
|
|
with zipfile.ZipFile(file_path) as zf:
|
|
names = zf.namelist()
|
|
payload["objects"]["charts"] = sum(1 for n in names if _CHART_PART_RE.match(n))
|
|
payload["objects"]["pivots"] = sum(1 for n in names if _PIVOT_PART_RE.match(n))
|
|
return payload
|
|
except Exception:
|
|
logger.debug("xlsx dashboard unavailable", exc_info=True)
|
|
return {
|
|
"named_ranges": [],
|
|
"objects": {"charts": 0, "pivots": 0},
|
|
"sheets": [],
|
|
}
|
|
finally:
|
|
wb.close()
|
|
|
|
|
|
# ── #153 A16 — additional spreadsheet formats ───────────────────────────────
|
|
|
|
|
|
# BUG-098 — a CSV written by a French office suite is `;`-separated, and the
|
|
# delimiter must be *detected* (then reused on write-back), not assumed.
|
|
CSV_SNIFF_BYTES = 4096
|
|
_CSV_DELIMITERS = (",", ";", "\t", "|")
|
|
|
|
|
|
def sniff_csv_delimiter(raw: str) -> str:
|
|
"""Return the most likely field delimiter of *raw* (``,`` as fallback).
|
|
|
|
:class:`csv.Sniffer` handles quoted fields and multi-line values; it is
|
|
unreliable on short or single-column samples, so the first non-empty line
|
|
is counted as a tie-breaker and a comma remains the last resort.
|
|
"""
|
|
sample = raw[:CSV_SNIFF_BYTES]
|
|
try:
|
|
return csv.Sniffer().sniff(sample, delimiters="".join(_CSV_DELIMITERS)).delimiter
|
|
except csv.Error:
|
|
pass
|
|
first = next((line for line in sample.splitlines() if line.strip()), "")
|
|
counts = {delim: first.count(delim) for delim in _CSV_DELIMITERS}
|
|
best = max(counts, key=lambda delim: counts[delim])
|
|
return best if counts[best] else ","
|
|
|
|
|
|
def render_csv_table(raw: str, *, delimiter: str | None = None) -> str:
|
|
"""Render CSV text as the same HTML table shape the xlsx viewer consumes.
|
|
|
|
Row numbers replace the A1 column: a CSV has no fixed column count, so
|
|
the first row is a plain data row like the others (the viewer offers the
|
|
toolbar either way). Every cell is HTML-escaped at render time.
|
|
|
|
``delimiter`` defaults to the sniffed one (:func:`sniff_csv_delimiter`,
|
|
BUG-098): a `;`-separated file used to render as a single column.
|
|
"""
|
|
import csv as csv_mod
|
|
import io as io_mod
|
|
|
|
reader = csv_mod.reader(
|
|
io_mod.StringIO(raw), delimiter=delimiter or sniff_csv_delimiter(raw)
|
|
)
|
|
try:
|
|
rows = [row for row in reader]
|
|
except csv_mod.Error:
|
|
# A malformed CSV still renders: each line becomes a one-cell row.
|
|
rows = [[line] for line in raw.splitlines()]
|
|
if not rows:
|
|
return "<p><em>Feuille vide</em></p>"
|
|
n_cols = max(len(r) for r in rows)
|
|
out = [
|
|
('<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
|
'<thead><tr><th class="xlsx-corner"></th>')
|
|
]
|
|
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
|
out.append("</tr></thead><tbody>")
|
|
for r, row in enumerate(rows, start=1):
|
|
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
|
for c in range(1, n_cols + 1):
|
|
val = row[c - 1] if c - 1 < len(row) else ""
|
|
out.append(f'<td data-cell="{get_column_letter(c)}{r}">{html.escape(val)}</td>')
|
|
out.append("</tr>")
|
|
out.append("</tbody></table></div>")
|
|
return "".join(out)
|
|
|
|
|
|
def render_legacy_workbook(file_path: Path, ext: str) -> list[dict[str, Any]]:
|
|
"""Render ``.xls``/``.ods`` sheets with the same dict shape as xlsx.
|
|
|
|
Read-only formats (#153 A16): ``styles``/``aligns``/``merges``/``freeze``
|
|
are served empty so the client-side wiring keeps one code path. Raises
|
|
nothing to the render path: an unreadable file yields one error sheet.
|
|
"""
|
|
name = file_path.name
|
|
try:
|
|
if ext == ".xls":
|
|
import xlrd
|
|
|
|
book = xlrd.open_workbook(str(file_path))
|
|
titles = book.sheet_names()
|
|
grids = []
|
|
for si in range(book.nsheets):
|
|
sh = book.sheet_by_index(si)
|
|
grid = [
|
|
[_fmt(sh.cell_value(r, c)) for c in range(min(sh.ncols, MAX_COLS))]
|
|
for r in range(min(sh.nrows, MAX_ROWS))
|
|
]
|
|
grids.append(_trim(grid))
|
|
total = [(sh.nrows, sh.ncols) for sh in (book.sheet_by_index(i) for i in range(book.nsheets))]
|
|
elif ext == ".ods":
|
|
from odf.opendocument import load as odf_load
|
|
from odf.table import Table, TableCell, TableRow
|
|
from odf.teletype import extractText
|
|
|
|
doc = odf_load(str(file_path))
|
|
titles = []
|
|
grids = []
|
|
total = []
|
|
for table in doc.getElementsByType(Table):
|
|
title = table.getAttribute("name") or f"Feuille {len(titles) + 1}"
|
|
titles.append(title)
|
|
grid = []
|
|
for row in table.getElementsByType(TableRow)[:MAX_ROWS]:
|
|
row_cells = row.getElementsByType(TableCell)
|
|
values: list[str] = []
|
|
for tc in row_cells[:MAX_COLS]:
|
|
repeat = int(tc.getAttribute("numbercolumnsrepeated") or 1)
|
|
values.extend([extractText(tc)] * min(repeat, MAX_COLS - len(values)))
|
|
grid.append(values)
|
|
grids.append(_trim(grid))
|
|
total.append((len(grid), max((len(r) for r in grid), default=0)))
|
|
else:
|
|
raise ValueError(f"Unsupported legacy format: {ext}")
|
|
except Exception as exc:
|
|
logger.warning("legacy workbook render failed for %s: %s", name, exc)
|
|
return [
|
|
{
|
|
"name": name,
|
|
"html": (
|
|
'<p><em>Feuille vide</em></p>'
|
|
),
|
|
"rows": 0,
|
|
"cols": 0,
|
|
"total_rows": 0,
|
|
"total_cols": 0,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": False,
|
|
"styles": {},
|
|
"aligns": {},
|
|
"merges": [],
|
|
"freeze": "",
|
|
}
|
|
]
|
|
|
|
sheets: list[dict[str, Any]] = []
|
|
for i, title in enumerate(titles):
|
|
grid = grids[i] if i < len(grids) else []
|
|
t_rows, t_cols = total[i] if i < len(total) else (0, 0)
|
|
sheets.append(
|
|
{
|
|
"name": title,
|
|
"html": _table(grid),
|
|
"rows": len(grid),
|
|
"cols": max((len(r) for r in grid), default=0),
|
|
"total_rows": t_rows,
|
|
"total_cols": t_cols,
|
|
"max_rows": MAX_ROWS,
|
|
"max_cols": MAX_COLS,
|
|
"truncated": t_rows > MAX_ROWS or t_cols > MAX_COLS,
|
|
"styles": {},
|
|
"aligns": {},
|
|
"merges": [],
|
|
"freeze": "",
|
|
}
|
|
)
|
|
return sheets
|