Restaure et termine le lot tableur resté non committé. Il n'existait que dans un `git stash` (24 fichiers suivis) et en fichiers non suivis (menus.js, formula.js, color-picker.js, tests, fiches) : la 2.53.3 livrée ne le contenait donc pas. Le stash, créé avant le commit BUG-108, n'a jamais été restauré. #180 Redimensionnement au curseur (en-tête de bord miroité + classe de repaint), grille thémée sur les 15 thèmes et les modes contraste élevé / sépia, menu Fichier au niveau des onglets, barre épinglée pleine largeur. #181 Modèle de saisie Google Sheets : sélection ≠ édition (caret masqué sans quitter contenteditable, double-clic / F2 / première frappe qui remplace, Entrée contextuelle, Échap qui restaure) + inventaire priorisé des écarts. #182 Désélection fiable (clic simple sans dépendre du focus), contours de plage non empilés (sélecteur de classes sans point), couleurs texte/fond sur une plage, clic droit qui préserve la sélection multiple. #183 Panneau de couleurs façon Google Sheets : palette 8 × 10, STANDARD, PERSONNALISÉ, coche selon la luminance, sortie `#rrggbb` (une valeur HSL était rejetée par normHex). #184 Peinture de format complète (toutes propriétés, source sans format = réinitialisation), sélection multi-lignes/colonnes depuis les marges, grille étendue : colonnes A → Z d'emblée, lignes ajoutées PAR BLOCS DE 100 au défilement jusqu'à 1000 — matérialiser 1000 lignes d'un coup = ~26 000 cellules câblées par feuille, ce qui épuisait le tas de la suite JSDOM ; index de cellules `ref → td`, court-circuits formule/styles, marqueur data-wired. #185 Barre de menus : les 10 menus Google Sheets (161 entrées, 125 câblées, 36 annoncées indisponibles), ruban façon Sheets, grille unie 1 px dérivée du thème. #186 « Créer un fichier » propose .xlsx et construit un vrai classeur OPC (openpyxl) au lieu d'une charge utile texte illisible. Corrections trouvées en restaurant et en exerçant le lot : - fuite mémoire : écouteur `click` anonyme posé sur #content-area à chaque rendu, jamais retiré — sa fermeture retenait la grille précédente en entier ; - sorties Markdown / HTML / Imprimer de la barre de menus inertes (`data-xlsx-export` jamais réparti, seul `data-xlsx-action` l'était) ; - collision de classe `.xlsx-structure-menu` entre la barre de menus et le menu Structure de la barre d'outils (toute requête tombait sur un nœud masqué) ; - curseur col/row-resize absent quand le pointeur est sur la table elle-même ; - l'export emportait les lignes et colonnes vides du quadrillage (CSV, Markdown, HTML, impression) ; - clic extérieur avalé par la grâce de 250 ms du menu contextuel (destinée au seul appui long tactile) ; - une entrée indisponible laissait la barre de menus ouverte. Tests : pytest 1595 passés / 2 ignorés ; ruff et mypy 0 erreur ; bandit 0 ; 33 suites frontend vertes (xlsx-viewer 163/163, xlsx-menus 19/19, xlsx-formula 14/14) ; E2E complet 132 passés / 12 ignorés ; E2E xlsx-viewer 19/19.
1171 lines
47 KiB
Python
1171 lines
47 KiB
Python
"""Render ``.xlsx`` workbooks as HTML tables for the viewer (#xlsx).
|
||
|
||
Read-only: formulas are shown as their text (``data_only=False``) so a
|
||
round-trip through the viewer never depends on Excel's cached values.
|
||
Write-side lives in ``backend.services.mutations.edit_xlsx_cells``.
|
||
|
||
:func:`inspect_workbook` lists the workbook features that an openpyxl
|
||
round-trip would drop (#153 A1) so the UI can warn before saving.
|
||
|
||
#153 A16 — :func:`render_sheets` also accepts ``.xlsm`` (macros preserved on
|
||
save via ``keep_vba``), ``.xls`` (xlrd) and ``.ods`` (odfpy), both served
|
||
read-only; :func:`render_csv_table` turns a CSV into the same table shape.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import csv
|
||
import html
|
||
import logging
|
||
import re
|
||
import threading
|
||
import zipfile
|
||
from datetime import date, datetime
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
from openpyxl import load_workbook
|
||
from openpyxl.utils import get_column_letter
|
||
|
||
logger = logging.getLogger("obsigate.xlsx_reader")
|
||
|
||
# ponytail: hard caps bound the rendered grid (500 rows x 40 cols per sheet).
|
||
# Raise them, or paginate per sheet, if a real workbook needs more.
|
||
MAX_ROWS = 500
|
||
MAX_COLS = 40
|
||
|
||
# BUG-094 — an empty sheet used to render as a bare "Feuille vide" paragraph
|
||
# with no cell at all, so a freshly added sheet had nothing to click and no way
|
||
# to insert a row/column. Render a small blank grid instead (Excel-like), with
|
||
# real A1 coordinates, so the cells are editable and the structure actions work.
|
||
EMPTY_SHEET_ROWS = 20
|
||
EMPTY_SHEET_COLS = 8
|
||
|
||
# #153 A9 — window size served by ``read_sheet_window()`` (lazy per-sheet
|
||
# loading). The endpoint is bounded so a single request can never ask for the
|
||
# whole workbook back in one JSON payload; the UI pages through the rest.
|
||
MAX_WINDOW_ROWS = 1_000
|
||
DEFAULT_WINDOW_ROWS = 200
|
||
|
||
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
|
||
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
|
||
# DO survive the round-trip, so they are deliberately absent from this map.
|
||
LOSSY_PARTS: dict[str, tuple[str, ...]] = {
|
||
"slicers": ("xl/slicers/", "xl/slicerCaches/", "xl/timelines/"),
|
||
"form_controls": ("xl/ctrlProps/", "xl/activeX/"),
|
||
"connections": ("xl/queryTables/", "xl/connections.xml"),
|
||
"custom_xml": ("customXml/",),
|
||
"signature": ("_xmlsignatures/",),
|
||
"rich_comments": ("xl/threadedComments/", "xl/persons/"),
|
||
"macros": ("xl/vbaProject.bin",),
|
||
}
|
||
|
||
# A formula cell carrying its last computed result: ``<f>…</f><v>…</v>``.
|
||
# openpyxl writes an EMPTY ``<v></v>`` itself, hence the ``[^<]`` guard: only a
|
||
# non-empty value counts. openpyxl keeps the formula but drops the cached result,
|
||
# so any reader using ``data_only=True`` (pandas, converters) sees ``None`` until
|
||
# Excel recalculates.
|
||
_CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
|
||
|
||
# Sheet XML scanned by the cached-formula probe (CPU guard, like
|
||
# MAX_REPLACE_FILE_BYTES). BUG-099 — the allowance is PER SHEET: a single huge
|
||
# first sheet used to eat the whole budget and hide a cached formula sitting in
|
||
# the next one. The total ceiling still bounds the work on a many-sheet archive.
|
||
# Running out of budget is reported as *unverified* (never as "nothing to lose")
|
||
# so the write guard stays cautious instead of silently dropping the values.
|
||
_MAX_PROBE_BYTES_PER_SHEET = 4_000_000
|
||
_MAX_PROBE_BYTES_TOTAL = 32_000_000
|
||
|
||
# #156-A13 — metadata cache. `read_sheet_window()` serves one window at a time
|
||
# and used to reload the whole workbook (normal mode, data_only=False) for every
|
||
# window, just to read the style/merge/freeze maps of one sheet. The result is
|
||
# keyed by (path, mtime_ns, size): any write replaces the file, hence the key.
|
||
_META_CACHE_MAX = 8
|
||
_meta_cache: dict[str, tuple[tuple[int, int], dict[str, dict[str, Any]]]] = {}
|
||
_meta_cache_lock = threading.Lock()
|
||
|
||
# #153 A17 — OPC parts of chart / pivot objects, matched against the archive
|
||
# name list (xl/charts/chart1.xml, xl/pivotTables/pivotTable1.xml, …).
|
||
_CHART_PART_RE = re.compile(r"^xl/charts/chart\d+\.xml$")
|
||
_PIVOT_PART_RE = re.compile(r"^xl/pivotTables/pivotTable\d+\.xml$")
|
||
|
||
# #153 A5 — ceiling on the text handed to the TF-IDF / semantic index. A workbook
|
||
# is a data dump, not prose: indexing every cell would flood the inverted index
|
||
# and bury the notes. Sheet names + the first rows are enough to make a
|
||
# spreadsheet findable by its headers.
|
||
MAX_INDEX_CHARS = 5_000
|
||
_INDEX_ROWS_PER_SHEET = 20
|
||
MAX_INDEX_SHEETS = 20
|
||
|
||
# #153 A15 — reading styles is a second (non-read_only) pass on the sheet XML.
|
||
# Bounded like everything else: a cell must be INSIDE the rendered window to
|
||
# deserve an inline style, so a huge workbook never triggers a huge payload.
|
||
# Only data-driven fragments are emitted: the hex values come from the file,
|
||
# never from a hardcoded color table.
|
||
|
||
|
||
def _cell_fragments(cell: Any) -> tuple[list[str], str | None]:
|
||
"""Inline CSS fragments of one cell plus its horizontal alignment.
|
||
|
||
Fixed, color-first order: the API contract documents ``color:...`` as the
|
||
first fragment of a styled cell. Only data-driven values are emitted —
|
||
every hex comes from the workbook itself, never a hardcoded table.
|
||
"""
|
||
fragments: list[str] = []
|
||
font = cell.font
|
||
if font and font.color is not None and isinstance(font.color.rgb, str):
|
||
# ARGB from the workbook itself — never a hardcoded table.
|
||
rgb = font.color.rgb
|
||
if len(rgb) == 8 and rgb != "FF000000":
|
||
fragments.append(f"color:#{rgb[2:].lower()}")
|
||
fill = cell.fill
|
||
if fill and fill.fgColor is not None and isinstance(fill.fgColor.rgb, str):
|
||
rgb = fill.fgColor.rgb
|
||
if len(rgb) == 8 and rgb not in ("00000000", "FFFFFFFF"):
|
||
fragments.append(f"background:#{rgb[2:].lower()}")
|
||
if font and font.bold:
|
||
fragments.append("font-weight:600")
|
||
if font and font.italic:
|
||
fragments.append("font-style:italic")
|
||
# #156-A8 — underline is written from the viewer too, so it is read back
|
||
# (the toggle in the formatting menu needs to see its own effect).
|
||
decorations = []
|
||
if font and font.underline:
|
||
decorations.append("underline")
|
||
if font and getattr(font, "strike", False):
|
||
decorations.append("line-through")
|
||
if decorations:
|
||
fragments.append("text-decoration:" + " ".join(decorations))
|
||
size = getattr(font, "size", None) if font else None
|
||
if isinstance(size, (int, float)) and size != 11:
|
||
fragments.append(f"font-size:{size:g}pt")
|
||
fmt = cell.number_format
|
||
if fmt and fmt not in ("General", "@"):
|
||
# A custom number format is signalled typographically (mono font)
|
||
# rather than rendered: the displayed value already carries the
|
||
# formatting from _fmt(). Single quotes: the fragment lands inside a
|
||
# double-quoted HTML attribute.
|
||
fragments.append("font-family:'JetBrains Mono',monospace")
|
||
alignment = cell.alignment
|
||
align = alignment.horizontal if alignment else None
|
||
if alignment is not None:
|
||
vertical = getattr(alignment, "vertical", None)
|
||
if vertical == "center":
|
||
fragments.append("vertical-align:middle")
|
||
elif vertical in ("top", "bottom"):
|
||
fragments.append(f"vertical-align:{vertical}")
|
||
if getattr(alignment, "wrap_text", None) is True:
|
||
fragments.append("white-space:normal")
|
||
elif getattr(alignment, "wrap_text", None) is False:
|
||
fragments.append("white-space:nowrap")
|
||
rotation = getattr(alignment, "text_rotation", None) or 0
|
||
if rotation == 90 or rotation == 255:
|
||
fragments.append("writing-mode:vertical-rl")
|
||
elif rotation == 180:
|
||
fragments.append("writing-mode:vertical-rl;transform:scale(-1)")
|
||
else:
|
||
# Angles diagonaux (ex. ±45° du menu Format › Rotation) : OOXML
|
||
# stocke l'anti-horaire sur 0..90 et l'horaire sur 91..180 (-45°
|
||
# → 135, cf. mutations). Le CSS tourne en sens horaire : signe
|
||
# inversé pour un rendu fidèle.
|
||
excel = rotation if rotation <= 90 else -(180 - rotation)
|
||
if excel:
|
||
fragments.append(f"transform:rotate({-excel}deg)")
|
||
border = getattr(cell, "border", None)
|
||
if border is not None:
|
||
sides = []
|
||
for side_name in ("left", "right", "top", "bottom"):
|
||
side = getattr(border, side_name, None)
|
||
style = getattr(side, "style", None)
|
||
if not style:
|
||
continue
|
||
color = getattr(getattr(side, "color", None), "rgb", None)
|
||
hexpart = ""
|
||
if isinstance(color, str) and len(color) == 8:
|
||
hexpart = f" #{color[2:].lower()}"
|
||
width = {"medium": "2px", "thick": "3px"}.get(style, "1px")
|
||
css_style = {"dashed": "dashed", "dotted": "dotted", "double": "double"}.get(
|
||
style, "solid"
|
||
)
|
||
sides.append(f"border-{side_name}:{width} {css_style}{hexpart}")
|
||
fragments.extend(sides)
|
||
return fragments, (align if align in ("left", "right", "center", "justify") else None)
|
||
|
||
|
||
def _sheet_style_maps(
|
||
ws: Any,
|
||
) -> tuple[dict[str, str], dict[str, str], dict[str, str], dict[str, str]]:
|
||
"""Flat maps of one worksheet: css, align, comment text, number format.
|
||
|
||
The flat string is what the API serves and what the viewer applies
|
||
verbatim to ``td.style``; a plain cell is simply absent from the map.
|
||
``left`` is the table default and never included. Bounded by
|
||
``MAX_ROWS x MAX_COLS`` like the render itself. Comment texts are capped
|
||
(tooltip use) and formats only list non-General number formats.
|
||
"""
|
||
styles: dict[str, str] = {}
|
||
aligns: dict[str, str] = {}
|
||
comments: dict[str, str] = {}
|
||
formats: dict[str, str] = {}
|
||
for row in ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS):
|
||
for cell in row:
|
||
comment = getattr(cell, "comment", None)
|
||
comment_text = getattr(comment, "text", None) if comment is not None else None
|
||
# Une cellule vide mais stylée (gras, diagonale, commentaire…)
|
||
# doit survivre à la relecture : `has_style` (et le commentaire)
|
||
# sont les seuls tests rapides qui la distinguent d'une cellule
|
||
# vraiment vierge — sinon le style semble « ne pas fonctionner »
|
||
# dès qu'on recharge le classeur.
|
||
if cell.value is None and cell.number_format == "General" and not cell.has_style:
|
||
if comment_text:
|
||
comments[cell.coordinate] = str(comment_text)[:500]
|
||
continue
|
||
fragments, align = _cell_fragments(cell)
|
||
if fragments:
|
||
styles[cell.coordinate] = ";".join(fragments)
|
||
if align and align != "left":
|
||
aligns[cell.coordinate] = align
|
||
if comment_text:
|
||
comments[cell.coordinate] = str(comment_text)[:500]
|
||
fmt = getattr(cell, "number_format", None)
|
||
if fmt and fmt not in ("General", "@", None):
|
||
formats[cell.coordinate] = str(fmt)
|
||
return styles, aligns, comments, formats
|
||
|
||
|
||
def _sheet_dim_maps(ws: Any) -> tuple[dict[str, float], dict[str, float]]:
|
||
"""Explicit column widths (units) and row heights (pt), by letter/number.
|
||
|
||
Only dimensions the file sets explicitly are listed (defaults are a
|
||
rendering concern of the client). Bounded like the render itself.
|
||
"""
|
||
from openpyxl.utils import column_index_from_string
|
||
|
||
colwidths: dict[str, float] = {}
|
||
try:
|
||
for key, dim in (getattr(ws, "column_dimensions", {}) or {}).items():
|
||
width = getattr(dim, "width", None)
|
||
if not isinstance(width, (int, float)):
|
||
continue
|
||
letters = str(key).split(":")
|
||
try:
|
||
cols = [column_index_from_string(a.strip()) for a in letters]
|
||
except ValueError:
|
||
continue
|
||
lo, hi = min(cols), max(cols)
|
||
for c in range(lo, min(hi, MAX_COLS) + 1):
|
||
if c >= 1:
|
||
from openpyxl.utils import get_column_letter
|
||
|
||
colwidths[get_column_letter(c)] = float(width)
|
||
except Exception:
|
||
logger.debug("xlsx column widths unavailable", exc_info=True)
|
||
rowheights: dict[str, float] = {}
|
||
try:
|
||
for r, dim in (getattr(ws, "row_dimensions", {}) or {}).items():
|
||
if not isinstance(r, int) or not 1 <= r <= MAX_ROWS:
|
||
continue
|
||
height = getattr(dim, "height", None)
|
||
if isinstance(height, (int, float)):
|
||
rowheights[str(r)] = float(height)
|
||
except Exception:
|
||
logger.debug("xlsx row heights unavailable", exc_info=True)
|
||
return colwidths, rowheights
|
||
|
||
|
||
def read_sheet_styles(file_path: Path, sheet: str) -> dict[str, dict[str, Any]]:
|
||
"""Return ``{ref: {style, align}}`` for the styled cells of one sheet.
|
||
|
||
``style`` is the flat CSS fragment the viewer applies verbatim and
|
||
``align`` the horizontal text-align when it is not the table default.
|
||
Normal (non-streaming) load — styles are unavailable in read_only mode;
|
||
a failure yields ``{}`` so the viewer falls back to the plain rendering.
|
||
"""
|
||
meta = read_workbook_meta(file_path).get(sheet, {})
|
||
styles_map = meta.get("styles", {})
|
||
aligns = meta.get("aligns", {})
|
||
out: dict[str, dict[str, Any]] = {}
|
||
for ref, css in styles_map.items():
|
||
entry: dict[str, Any] = {"style": css}
|
||
if ref in aligns:
|
||
entry["align"] = aligns[ref]
|
||
out[ref] = entry
|
||
return out
|
||
|
||
|
||
def read_sheet_merges(file_path: Path, sheet: str) -> list[str]:
|
||
"""Return the merged ranges of one sheet as ``A1:C3`` strings."""
|
||
try:
|
||
# Styles and merges are only fully materialised in normal mode
|
||
# (read_only=True leaves merged_cells empty).
|
||
wb = load_workbook(str(file_path))
|
||
except Exception:
|
||
return []
|
||
try:
|
||
if sheet not in wb.sheetnames:
|
||
return []
|
||
merged = getattr(wb[sheet], "merged_cells", None)
|
||
ranges = getattr(merged, "ranges", None) or []
|
||
return [str(r) for r in ranges]
|
||
except Exception:
|
||
logger.debug("xlsx merges unavailable", exc_info=True)
|
||
return []
|
||
finally:
|
||
wb.close()
|
||
|
||
|
||
def read_sheet_freeze(file_path: Path, sheet: str) -> str:
|
||
"""Return the freeze-panes anchor of one sheet ('' when not frozen).
|
||
|
||
Normal (non-streaming) load: `freeze_panes` is NOT materialised on
|
||
ReadOnlyWorksheet in openpyxl 3.1.x — read_only=True always yields ''.
|
||
"""
|
||
try:
|
||
wb = load_workbook(str(file_path))
|
||
except Exception:
|
||
return ""
|
||
try:
|
||
if sheet not in wb.sheetnames:
|
||
return ""
|
||
return str(getattr(wb[sheet], "freeze_panes", None) or "")
|
||
except Exception:
|
||
return ""
|
||
finally:
|
||
wb.close()
|
||
|
||
|
||
def _fmt(value: Any) -> str:
|
||
if value is None:
|
||
return ""
|
||
if isinstance(value, datetime):
|
||
return value.strftime("%Y-%m-%d %H:%M")
|
||
if isinstance(value, date):
|
||
return value.isoformat()
|
||
return str(value)
|
||
|
||
|
||
def _trim(grid: list[list[str]]) -> list[list[str]]:
|
||
"""Drop trailing empty rows and columns (openpyxl pads to max_col)."""
|
||
while grid and not any(grid[-1]):
|
||
grid.pop()
|
||
if not grid:
|
||
return grid
|
||
width = 0
|
||
for row in grid:
|
||
for i in range(len(row) - 1, -1, -1):
|
||
if row[i]:
|
||
width = max(width, i + 1)
|
||
break
|
||
return [row[:width] for row in grid]
|
||
|
||
|
||
def _cell_cached(cached: list[list[str]] | None, r: int, c: int) -> str:
|
||
"""Return the cached result for a 0-based cell, or ``""``.
|
||
|
||
The shadow grid is read positionally and may be narrower than the formula
|
||
grid (``_trim`` collapses the trailing empty columns of each grid
|
||
independently), so every lookup is bounds-checked rather than assumed.
|
||
"""
|
||
if not cached or r >= len(cached):
|
||
return ""
|
||
row = cached[r]
|
||
return row[c] if c < len(row) else ""
|
||
|
||
|
||
def _table(
|
||
grid: list[list[str]],
|
||
cached: list[list[str]] | None = None,
|
||
row_offset: int = 0,
|
||
styles: dict[str, dict[str, Any]] | None = None,
|
||
) -> str:
|
||
"""Render a grid as an HTML table.
|
||
|
||
``cached`` is the same grid read with ``data_only=True`` (#153 A12): where a
|
||
formula cell still carries its last computed result, it is shown as a
|
||
discreet second line (``<span class="xlsx-cached">``) so the user sees the
|
||
number Excel last calculated instead of only the formula text. The span
|
||
carries ``data-cached-value`` and is titled client-side from
|
||
``xlsx.cached_value_title`` — the backend never emits UI text.
|
||
|
||
``row_offset`` is the number of rows skipped before this grid (#153 A9): the
|
||
row numbers and the ``data-cell`` references must stay the real A1
|
||
coordinates of the sheet, not of the window.
|
||
|
||
``styles`` maps A1 references to ``{style, align}`` fragments (#153 A15:
|
||
bold, italic, background, alignment) — the backend only reads the
|
||
workbook, the fragments are built from it and always data-driven, never
|
||
hardcoded colors. A plain ``str`` value is tolerated (legacy callers).
|
||
"""
|
||
if not grid:
|
||
return "<p><em>Feuille vide</em></p>"
|
||
n_cols = max(len(row) for row in grid)
|
||
out = [
|
||
(
|
||
'<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
||
'<thead><tr><th class="xlsx-corner"></th>'
|
||
)
|
||
]
|
||
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
||
out.append("</tr></thead><tbody>")
|
||
for r, row in enumerate(grid, start=row_offset + 1):
|
||
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
||
for c, val in enumerate(row, start=1):
|
||
ref = f"{get_column_letter(c)}{r}"
|
||
meta = (styles or {}).get(ref)
|
||
if meta is None:
|
||
style_attr = ""
|
||
else:
|
||
# Legacy callers may still pass a bare CSS string.
|
||
if isinstance(meta, str):
|
||
meta = {"style": meta}
|
||
fragment = meta.get("style", "")
|
||
align = meta.get("align")
|
||
if align and align not in ("left",):
|
||
# left is the table default; only non-default alignments
|
||
# need an explicit declaration.
|
||
fragment = f"{fragment};text-align:{align}" if fragment else f"text-align:{align}"
|
||
style_attr = f' style="{fragment}"' if fragment else ""
|
||
# The cached result only makes sense for a formula cell: on a plain
|
||
# value cell the two reads are identical and showing both would
|
||
# duplicate the text.
|
||
shadow = ""
|
||
if cached is not None and val.startswith("="):
|
||
# `c` is 1-based (A1 notation) and `r` too, while the grid is
|
||
# 0-based: translate both.
|
||
cval = _cell_cached(cached, r - 1, c - 1)
|
||
if cval and cval != val:
|
||
# The tooltip is translated client-side from
|
||
# `xlsx.cached_value_title`; never hardcode UI text here.
|
||
shadow = (
|
||
f'<span class="xlsx-cached" data-cached-value="1">'
|
||
f"{html.escape(cval)}</span>"
|
||
)
|
||
out.append(
|
||
f'<td data-cell="{ref}"{style_attr}>{html.escape(val)}{shadow}</td>'
|
||
)
|
||
out.append("</tr>")
|
||
out.append("</tbody></table></div>")
|
||
return "".join(out)
|
||
|
||
|
||
def _is_sheet_xml(name: str) -> bool:
|
||
"""True for the worksheet XML parts the cached-formula probe scans."""
|
||
return name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
|
||
|
||
|
||
def _scan_cached_formulas(zf: zipfile.ZipFile) -> tuple[bool, bool]:
|
||
"""Scan the sheet XML for a formula carrying a non-empty cached result.
|
||
|
||
Returns ``(found, unverified)``. ``unverified`` is True when the byte
|
||
budget stopped the scan before every sheet could be read to the end: a
|
||
negative result is then **not** proof that the workbook holds no cached
|
||
value (BUG-099), so callers must not treat it as a licence to write.
|
||
|
||
Each sheet gets its own :data:`_MAX_PROBE_BYTES_PER_SHEET` allowance (a
|
||
single huge sheet can no longer starve the others) while
|
||
:data:`_MAX_PROBE_BYTES_TOTAL` bounds the whole archive. A hit short-
|
||
circuits the scan: the answer is already known.
|
||
"""
|
||
budget = _MAX_PROBE_BYTES_TOTAL
|
||
unverified = False
|
||
for name in zf.namelist():
|
||
if not _is_sheet_xml(name):
|
||
continue
|
||
sheet_budget = min(_MAX_PROBE_BYTES_PER_SHEET, budget)
|
||
exhausted = False
|
||
try:
|
||
with zf.open(name) as fh:
|
||
while sheet_budget > 0:
|
||
# Read at most what the sheet's allowance has left, so one
|
||
# large chunk can never consume the whole total budget.
|
||
chunk = fh.read(min(65536, sheet_budget))
|
||
if not chunk:
|
||
break # read to the end: this sheet is verified clean
|
||
sheet_budget -= len(chunk)
|
||
budget -= len(chunk)
|
||
if _CACHED_FORMULA_RE.search(chunk):
|
||
return True, False
|
||
else:
|
||
# Left the loop on the budget, not on EOF.
|
||
exhausted = True
|
||
except (KeyError, OSError, zipfile.BadZipFile):
|
||
continue
|
||
if exhausted:
|
||
unverified = True
|
||
if budget <= 0:
|
||
unverified = True
|
||
break
|
||
return False, unverified
|
||
|
||
|
||
def _has_cached_formulas(zf: zipfile.ZipFile) -> bool:
|
||
"""True when at least one formula cell still carries its computed value.
|
||
|
||
Boolean view of :func:`_scan_cached_formulas` for the display path (the
|
||
second ``data_only=True`` read): a truncated scan simply skips the shadow
|
||
grid, it never claims the workbook is lossless.
|
||
"""
|
||
found, _ = _scan_cached_formulas(zf)
|
||
return found
|
||
|
||
|
||
def inspect_workbook(file_path: Path) -> list[str]:
|
||
"""Return the sorted keys of :data:`LOSSY_PARTS` present in *file_path*.
|
||
|
||
Read-only inspection of the OPC package (central directory + a bounded scan
|
||
of the sheet XML). Never raises: an unreadable or encrypted workbook simply
|
||
yields ``[]`` and the save path keeps its current behaviour.
|
||
|
||
``cached_values`` is a synthetic key: openpyxl keeps the formula but drops
|
||
the cached result, so the workbook stays correct once Excel recalculates it.
|
||
``cached_values_unverified`` (BUG-099) is the other synthetic key: the
|
||
cached-value probe ran out of budget, so a negative result is not proof —
|
||
the entry keeps the write guard cautious (409 + confirmation) rather than
|
||
promising a lossless round-trip it cannot vouch for.
|
||
"""
|
||
try:
|
||
with zipfile.ZipFile(file_path) as zf:
|
||
names = set(zf.namelist())
|
||
found = {
|
||
key
|
||
for key, prefixes in LOSSY_PARTS.items()
|
||
if any(name.startswith(prefix) for name in names for prefix in prefixes)
|
||
}
|
||
cached_found, cached_unverified = _scan_cached_formulas(zf)
|
||
if cached_found:
|
||
found.add("cached_values")
|
||
elif cached_unverified:
|
||
found.add("cached_values_unverified")
|
||
return sorted(found)
|
||
except (OSError, zipfile.BadZipFile):
|
||
return []
|
||
|
||
|
||
def invalidate_workbook_meta(file_path: Path | str) -> None:
|
||
"""Drop the cached metadata of one workbook (#156-A13).
|
||
|
||
Called by the write path right after the atomic replace: the (mtime, size)
|
||
key already changes on a rewrite, this only closes the theoretical window
|
||
where a same-size write lands on the same timestamp tick.
|
||
"""
|
||
with _meta_cache_lock:
|
||
_meta_cache.pop(str(file_path), None)
|
||
|
||
|
||
def read_workbook_meta(file_path: Path) -> dict[str, dict[str, Any]]:
|
||
"""Return ``{sheet: {styles, aligns, comments, formats, colwidths,
|
||
rowheights, merges, freeze}}``.
|
||
|
||
One normal (non-streaming) load serves the three A15 metadata maps: the
|
||
fragments are the workbook's own values, a failure yields ``{}`` per sheet
|
||
so the viewer keeps its plain rendering. Styles are read with
|
||
``data_only=False`` — the edited value is the formula, not its result.
|
||
|
||
#156-A13 — the result is cached on ``(path, mtime_ns, size)``: a truncated
|
||
sheet is fetched window by window, and each window used to pay a full
|
||
workbook load for these maps alone. Callers only read from the mapping.
|
||
"""
|
||
try:
|
||
st = file_path.stat()
|
||
except OSError:
|
||
return {}
|
||
key = str(file_path)
|
||
stamp = (st.st_mtime_ns, st.st_size)
|
||
with _meta_cache_lock:
|
||
hit = _meta_cache.get(key)
|
||
if hit and hit[0] == stamp:
|
||
return hit[1]
|
||
out = _read_workbook_meta_uncached(file_path)
|
||
with _meta_cache_lock:
|
||
_meta_cache[key] = (stamp, out)
|
||
while len(_meta_cache) > _META_CACHE_MAX:
|
||
_meta_cache.pop(next(iter(_meta_cache)))
|
||
return out
|
||
|
||
|
||
def _read_workbook_meta_uncached(file_path: Path) -> dict[str, dict[str, Any]]:
|
||
"""Load the workbook once and build the per-sheet metadata maps."""
|
||
try:
|
||
wb = load_workbook(str(file_path), data_only=False)
|
||
except Exception:
|
||
return {}
|
||
out: dict[str, dict[str, Any]] = {}
|
||
try:
|
||
for ws in wb.worksheets:
|
||
styles, aligns, comments, formats = _sheet_style_maps(ws)
|
||
colwidths, rowheights = _sheet_dim_maps(ws)
|
||
merged = getattr(ws, "merged_cells", None)
|
||
ranges = getattr(merged, "ranges", None) or []
|
||
out[ws.title] = {
|
||
"styles": styles,
|
||
"aligns": aligns,
|
||
"comments": comments,
|
||
"formats": formats,
|
||
"colwidths": colwidths,
|
||
"rowheights": rowheights,
|
||
"merges": [str(r) for r in ranges],
|
||
"freeze": str(getattr(ws, "freeze_panes", None) or ""),
|
||
}
|
||
return out
|
||
except Exception:
|
||
logger.debug("xlsx meta unavailable", exc_info=True)
|
||
for t in wb.sheetnames:
|
||
out.setdefault(
|
||
t,
|
||
{
|
||
"styles": {}, "aligns": {}, "comments": {}, "formats": {},
|
||
"colwidths": {}, "rowheights": {},
|
||
"merges": [], "freeze": "",
|
||
},
|
||
)
|
||
return out
|
||
finally:
|
||
wb.close()
|
||
|
||
|
||
def render_sheets(file_path: Path) -> list[dict[str, Any]]:
|
||
"""Return one dict per sheet: ``{name, html, rows, cols, total_*, truncated}``.
|
||
|
||
Reads the workbook twice: once with ``data_only=False`` for the formulas
|
||
(what the user must edit) and, when any formula carries a cached result
|
||
(#153 A12), once with ``data_only=True`` to show what Excel last computed.
|
||
The second pass is skipped entirely when the archive holds no cached value,
|
||
so the common case still costs a single load.
|
||
|
||
``total_rows``/``total_cols`` are the dimensions the sheet declares and
|
||
``truncated`` says whether the hard caps actually cut it (#153 A8) — the
|
||
viewer needs both to stop silently hiding the tail of a sheet.
|
||
"""
|
||
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
||
try:
|
||
formulas = [_sheet_grid(ws) for ws in wb.worksheets]
|
||
titles = [ws.title for ws in wb.worksheets]
|
||
extents = [_sheet_extent(ws) for ws in wb.worksheets]
|
||
finally:
|
||
wb.close()
|
||
|
||
cached: list[list[list[str]]] | None = None
|
||
if _has_cached_values(file_path):
|
||
cached = _read_cached_grids(file_path, titles)
|
||
|
||
# #153 A15 — one extra normal-mode load serves the styles/merges/freeze
|
||
# metadata of every sheet; the HTML then carries the fragments itself.
|
||
meta = read_workbook_meta(file_path)
|
||
|
||
sheets = []
|
||
for i, title in enumerate(titles):
|
||
grid = _trim(formulas[i])
|
||
# BUG-094 — a blank sheet still needs an editable grid (see constants):
|
||
# the viewer's cell editing and structure actions all hang off a cell.
|
||
if not grid:
|
||
grid = [[""] * EMPTY_SHEET_COLS for _ in range(EMPTY_SHEET_ROWS)]
|
||
# The shadow grid is NOT trimmed independently: _trim drops the
|
||
# trailing empty columns of each grid on its own width, which would
|
||
# shift every cached value left of its formula. Indexing it
|
||
# positionally against the untrimmed grid keeps the two aligned.
|
||
shadow = cached[i] if cached is not None and i < len(cached) else None
|
||
total_rows, total_cols = extents[i]
|
||
sheet_meta = meta.get(title, {})
|
||
sheets.append(
|
||
{
|
||
"name": title,
|
||
"html": _table(grid, shadow, styles=sheet_meta.get("styles")),
|
||
"rows": len(grid),
|
||
"cols": max((len(r) for r in grid), default=0),
|
||
"total_rows": total_rows,
|
||
"total_cols": total_cols,
|
||
# Coverage, not display size: `rows`/`cols` are post-trim (a
|
||
# sheet of 3 filled cells in a 500-row block renders 1x1), and
|
||
# the client must announce the cap it stopped at, not how many
|
||
# cells happen to be non-empty.
|
||
"max_rows": MAX_ROWS,
|
||
"max_cols": MAX_COLS,
|
||
# A sheet is truncated when the caps, not the trailing blanks,
|
||
# decided its shape: comparing against the *rendered* size would
|
||
# flag every sheet carrying a few empty formatted rows.
|
||
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
||
"styles": sheet_meta.get("styles", {}),
|
||
"aligns": sheet_meta.get("aligns", {}),
|
||
"comments": sheet_meta.get("comments", {}),
|
||
"formats": sheet_meta.get("formats", {}),
|
||
"colwidths": sheet_meta.get("colwidths", {}),
|
||
"rowheights": sheet_meta.get("rowheights", {}),
|
||
"merges": sheet_meta.get("merges", []),
|
||
"freeze": sheet_meta.get("freeze", ""),
|
||
}
|
||
)
|
||
return sheets
|
||
|
||
|
||
def read_sheet_window(
|
||
file_path: Path,
|
||
sheet: str,
|
||
offset: int = 0,
|
||
limit: int = DEFAULT_WINDOW_ROWS,
|
||
) -> dict[str, Any] | None:
|
||
"""Return a window of rows of one sheet, or ``None`` if the sheet is unknown.
|
||
|
||
Backs the lazy per-sheet loading of #153 A9: the viewer asks for the rows
|
||
it is about to display instead of shipping every sheet in the initial file
|
||
payload. ``offset`` is 0-based; the row numbers and the ``data-cell``
|
||
references in the returned ``html`` are the real A1 coordinates of the
|
||
sheet, so a window is indistinguishable from a full render.
|
||
|
||
``limit`` is clamped to :data:`MAX_WINDOW_ROWS`. Raises nothing: an unknown
|
||
sheet yields ``None`` and a broken workbook propagates the caller's usual
|
||
500.
|
||
"""
|
||
offset = max(int(offset), 0)
|
||
limit = min(max(int(limit), 1), MAX_WINDOW_ROWS)
|
||
|
||
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
||
try:
|
||
if sheet not in wb.sheetnames:
|
||
return None
|
||
ws = wb[sheet]
|
||
total_rows, total_cols = _sheet_extent(ws)
|
||
grid = _trim(
|
||
_sheet_grid(ws, min_row=offset + 1, max_row=offset + limit)
|
||
)
|
||
finally:
|
||
wb.close()
|
||
|
||
shadow: list[list[str]] | None = None
|
||
# Same A12 rule as the full render: the second read only happens when the
|
||
# archive really holds cached results.
|
||
if _has_cached_values(file_path):
|
||
shadow = _read_cached_window(file_path, sheet, offset, limit)
|
||
# #153 A15 — same metadata as the full render, so a lazy window is
|
||
# indistinguishable from it (styles in the HTML, merges/freeze for the
|
||
# client-side spanning).
|
||
meta = read_workbook_meta(file_path).get(sheet, {})
|
||
return {
|
||
"sheet": sheet,
|
||
"offset": offset,
|
||
"limit": limit,
|
||
"rows": len(grid),
|
||
"cols": max((len(r) for r in grid), default=0),
|
||
"total_rows": total_rows,
|
||
"total_cols": total_cols,
|
||
"max_rows": MAX_ROWS,
|
||
"max_cols": MAX_COLS,
|
||
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
||
"has_more": offset + len(grid) < total_rows,
|
||
"html": _table(grid, shadow, row_offset=offset, styles=meta.get("styles")),
|
||
"styles": meta.get("styles", {}),
|
||
"aligns": meta.get("aligns", {}),
|
||
"comments": meta.get("comments", {}),
|
||
"formats": meta.get("formats", {}),
|
||
"colwidths": meta.get("colwidths", {}),
|
||
"rowheights": meta.get("rowheights", {}),
|
||
"merges": meta.get("merges", []),
|
||
"freeze": meta.get("freeze", ""),
|
||
}
|
||
|
||
|
||
def _read_cached_window(
|
||
file_path: Path, sheet: str, offset: int, limit: int
|
||
) -> list[list[str]] | None:
|
||
"""``data_only=True`` grid for one window, or ``None`` if unavailable.
|
||
|
||
Best effort like :func:`_read_cached_grids`: a workbook Excel opens but
|
||
openpyxl cannot re-read must still display (formulas only).
|
||
"""
|
||
try:
|
||
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
||
except Exception:
|
||
return None
|
||
try:
|
||
if sheet not in wb.sheetnames:
|
||
return None
|
||
return _sheet_grid(
|
||
wb[sheet], min_row=offset + 1, max_row=offset + limit
|
||
)
|
||
except Exception:
|
||
logger.debug("xlsx cached window unavailable", exc_info=True)
|
||
return None
|
||
finally:
|
||
wb.close()
|
||
|
||
|
||
def _sheet_extent(ws: Any) -> tuple[int, int]:
|
||
"""Rows and columns the worksheet declares, never negative.
|
||
|
||
``max_row``/``max_column`` come from the sheet's dimension record; a
|
||
hand-edited file may omit it, hence the defensive coercion.
|
||
"""
|
||
try:
|
||
rows = max(int(getattr(ws, "max_row", 0) or 0), 0)
|
||
except (TypeError, ValueError):
|
||
rows = 0
|
||
try:
|
||
cols = max(int(getattr(ws, "max_column", 0) or 0), 0)
|
||
except (TypeError, ValueError):
|
||
cols = 0
|
||
return rows, cols
|
||
|
||
|
||
def _sheet_grid(
|
||
ws: Any, min_row: int = 1, max_row: int = MAX_ROWS, max_col: int = MAX_COLS
|
||
) -> list[list[str]]:
|
||
"""Read a worksheet window into a grid of formatted strings, bounded by the caps."""
|
||
return [
|
||
[_fmt(v) for v in row]
|
||
for row in ws.iter_rows(
|
||
min_row=min_row, max_row=max_row, max_col=max_col, values_only=True
|
||
)
|
||
]
|
||
|
||
|
||
def _has_cached_values(file_path: Path) -> bool:
|
||
"""True when the archive holds at least one ``<f>…</f><v>…</v>``."""
|
||
try:
|
||
with zipfile.ZipFile(file_path) as zf:
|
||
return _has_cached_formulas(zf)
|
||
except (OSError, zipfile.BadZipFile):
|
||
return False
|
||
|
||
|
||
def _read_cached_grids(
|
||
file_path: Path, titles: list[str]
|
||
) -> list[list[list[str]]] | None:
|
||
"""Read every sheet with ``data_only=True`` (what Excel last computed).
|
||
|
||
Best effort: returns ``None`` on any failure so the viewer falls back to the
|
||
formula-only rendering. A workbook Excel opens but openpyxl cannot re-read
|
||
must still display.
|
||
"""
|
||
try:
|
||
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
||
except Exception:
|
||
return None
|
||
try:
|
||
grids = [_sheet_grid(ws) for ws in wb.worksheets]
|
||
if [ws.title for ws in wb.worksheets] != titles:
|
||
return None
|
||
return grids
|
||
except Exception:
|
||
logger.debug("xlsx cached values unavailable", exc_info=True)
|
||
return None
|
||
finally:
|
||
wb.close()
|
||
|
||
|
||
def extract_indexable_text(file_path: Path) -> str:
|
||
"""Return searchable text for the TF-IDF / semantic index (#153 A5).
|
||
|
||
Sheet names plus the first :data:`_INDEX_ROWS_PER_SHEET` rows of each
|
||
sheet, capped at :data:`MAX_INDEX_CHARS`. Rows are tab-joined so a search
|
||
for a header matches the sheet it belongs to.
|
||
|
||
Never raises: a corrupt, encrypted or unsupported workbook yields ``""`` so
|
||
the file still gets indexed by name (same contract as :func:`inspect_workbook`).
|
||
"""
|
||
chunks: list[str] = []
|
||
budget = MAX_INDEX_CHARS
|
||
try:
|
||
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
||
except Exception:
|
||
# Encrypted (BadZipFile) or not a real workbook: name-only indexing.
|
||
return ""
|
||
try:
|
||
for ws in wb.worksheets[:MAX_INDEX_SHEETS]:
|
||
if budget <= 0:
|
||
break
|
||
# The sheet title alone is a strong signal ("Recettes", "Budget").
|
||
block = [ws.title]
|
||
for row in ws.iter_rows(
|
||
min_row=1, max_row=_INDEX_ROWS_PER_SHEET, max_col=MAX_COLS, values_only=True
|
||
):
|
||
cells = [_fmt(v) for v in row]
|
||
# Skip blank rows instead of emitting runs of tabs.
|
||
if not any(c.strip() for c in cells):
|
||
continue
|
||
block.append("\t".join(cells).rstrip())
|
||
text = "\n".join(block)
|
||
chunks.append(text[:budget])
|
||
budget -= len(text)
|
||
except Exception:
|
||
# Truncated but still useful: keep whatever was collected.
|
||
pass
|
||
finally:
|
||
wb.close()
|
||
return "\n".join(c for c in chunks if c).strip()
|
||
|
||
|
||
# ── #153 A17 — dashboard metadata ───────────────────────────────────
|
||
|
||
|
||
def read_workbook_dashboard(file_path: Path) -> dict[str, Any]:
|
||
"""Return the dashboard metadata of a workbook (#153 A17).
|
||
|
||
Shape::
|
||
|
||
{
|
||
"named_ranges": [{"name", "scope", "ref"}],
|
||
"objects": {"charts": int, "pivots": int},
|
||
"sheets": [{
|
||
"name": str,
|
||
"cells": int, # non-empty cells inside the caps
|
||
"rows": int, # rows carrying at least one non-empty cell
|
||
"cols": int, # columns carrying at least one non-empty cell
|
||
"formulas": int,
|
||
"numeric": int,
|
||
"kpi": [ # first 8 numeric cells as {"label", "value"}
|
||
{"label": str, "value": float}
|
||
],
|
||
}],
|
||
}
|
||
|
||
Named ranges come from the streaming load (available read-only), cell
|
||
stats from ``iter_rows(values_only=True)``. Charts/pivots are counted by
|
||
OPC part names (a chart part per chart, a pivot table part per pivot).
|
||
Bounded by MAX_ROWS/MAX_COLS; never raises — a failure yields an empty
|
||
payload and the viewer simply hides the panel.
|
||
"""
|
||
payload: dict[str, Any] = {
|
||
"named_ranges": [],
|
||
"objects": {"charts": 0, "pivots": 0},
|
||
"sheets": [],
|
||
}
|
||
try:
|
||
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
||
except Exception:
|
||
return payload
|
||
try:
|
||
dn = getattr(wb, "defined_names", None)
|
||
items: list[tuple[Any, Any]] = (
|
||
list(dn.items()) if dn is not None and hasattr(dn, "items") else []
|
||
)
|
||
for name, defn in items:
|
||
scope_idx = getattr(defn, "localSheetId", None)
|
||
scope = ""
|
||
if scope_idx is not None:
|
||
try:
|
||
scope = wb.sheetnames[int(scope_idx)]
|
||
except (IndexError, ValueError):
|
||
scope = ""
|
||
payload["named_ranges"].append(
|
||
{
|
||
"name": str(name),
|
||
"scope": scope,
|
||
"ref": str(getattr(defn, "attr_text", "") or ""),
|
||
}
|
||
)
|
||
payload["named_ranges"].sort(key=lambda d: d["name"].lower())
|
||
|
||
for ws in wb.worksheets:
|
||
cells = rows = formulas = numeric = 0
|
||
col_seen: set[int] = set()
|
||
kpi: list[dict[str, Any]] = []
|
||
for r, row in enumerate(
|
||
ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS, values_only=True),
|
||
start=1,
|
||
):
|
||
row_has_value = False
|
||
for c, value in enumerate(row, start=1):
|
||
if value is None or (isinstance(value, str) and not value.strip()):
|
||
continue
|
||
cells += 1
|
||
col_seen.add(c)
|
||
row_has_value = True
|
||
if isinstance(value, str) and value.startswith("="):
|
||
formulas += 1
|
||
elif isinstance(value, bool):
|
||
pass
|
||
elif isinstance(value, (int, float)):
|
||
numeric += 1
|
||
if len(kpi) < 8:
|
||
kpi.append(
|
||
{"label": f"{get_column_letter(c)}{r}", "value": value}
|
||
)
|
||
if row_has_value:
|
||
rows += 1
|
||
payload["sheets"].append(
|
||
{
|
||
"name": ws.title,
|
||
"cells": cells,
|
||
"rows": rows,
|
||
"cols": len(col_seen),
|
||
"formulas": formulas,
|
||
"numeric": numeric,
|
||
"kpi": kpi,
|
||
}
|
||
)
|
||
# Chart/pivot parts, counted from the archive (chart XML parts are
|
||
# one per chart; pivot parts one per pivot table/cache).
|
||
with zipfile.ZipFile(file_path) as zf:
|
||
names = zf.namelist()
|
||
payload["objects"]["charts"] = sum(1 for n in names if _CHART_PART_RE.match(n))
|
||
payload["objects"]["pivots"] = sum(1 for n in names if _PIVOT_PART_RE.match(n))
|
||
return payload
|
||
except Exception:
|
||
logger.debug("xlsx dashboard unavailable", exc_info=True)
|
||
return {
|
||
"named_ranges": [],
|
||
"objects": {"charts": 0, "pivots": 0},
|
||
"sheets": [],
|
||
}
|
||
finally:
|
||
wb.close()
|
||
|
||
|
||
# ── #153 A16 — additional spreadsheet formats ───────────────────────────────
|
||
|
||
|
||
# BUG-098 — a CSV written by a French office suite is `;`-separated, and the
|
||
# delimiter must be *detected* (then reused on write-back), not assumed.
|
||
CSV_SNIFF_BYTES = 4096
|
||
_CSV_DELIMITERS = (",", ";", "\t", "|")
|
||
|
||
|
||
def sniff_csv_delimiter(raw: str) -> str:
|
||
"""Return the most likely field delimiter of *raw* (``,`` as fallback).
|
||
|
||
:class:`csv.Sniffer` handles quoted fields and multi-line values; it is
|
||
unreliable on short or single-column samples, so the first non-empty line
|
||
is counted as a tie-breaker and a comma remains the last resort.
|
||
"""
|
||
sample = raw[:CSV_SNIFF_BYTES]
|
||
try:
|
||
return csv.Sniffer().sniff(sample, delimiters="".join(_CSV_DELIMITERS)).delimiter
|
||
except csv.Error:
|
||
pass
|
||
first = next((line for line in sample.splitlines() if line.strip()), "")
|
||
counts = {delim: first.count(delim) for delim in _CSV_DELIMITERS}
|
||
best = max(counts, key=lambda delim: counts[delim])
|
||
return best if counts[best] else ","
|
||
|
||
|
||
def render_csv_table(raw: str, *, delimiter: str | None = None) -> str:
|
||
"""Render CSV text as the same HTML table shape the xlsx viewer consumes.
|
||
|
||
Row numbers replace the A1 column: a CSV has no fixed column count, so
|
||
the first row is a plain data row like the others (the viewer offers the
|
||
toolbar either way). Every cell is HTML-escaped at render time.
|
||
|
||
``delimiter`` defaults to the sniffed one (:func:`sniff_csv_delimiter`,
|
||
BUG-098): a `;`-separated file used to render as a single column.
|
||
"""
|
||
import csv as csv_mod
|
||
import io as io_mod
|
||
|
||
reader = csv_mod.reader(
|
||
io_mod.StringIO(raw), delimiter=delimiter or sniff_csv_delimiter(raw)
|
||
)
|
||
try:
|
||
rows = [row for row in reader]
|
||
except csv_mod.Error:
|
||
# A malformed CSV still renders: each line becomes a one-cell row.
|
||
rows = [[line] for line in raw.splitlines()]
|
||
if not rows:
|
||
return "<p><em>Feuille vide</em></p>"
|
||
n_cols = max(len(r) for r in rows)
|
||
out = [
|
||
('<div class="csv-table-wrapper"><table class="csv-table xlsx-table">'
|
||
'<thead><tr><th class="xlsx-corner"></th>')
|
||
]
|
||
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
||
out.append("</tr></thead><tbody>")
|
||
for r, row in enumerate(rows, start=1):
|
||
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
||
for c in range(1, n_cols + 1):
|
||
val = row[c - 1] if c - 1 < len(row) else ""
|
||
out.append(f'<td data-cell="{get_column_letter(c)}{r}">{html.escape(val)}</td>')
|
||
out.append("</tr>")
|
||
out.append("</tbody></table></div>")
|
||
return "".join(out)
|
||
|
||
|
||
def render_legacy_workbook(file_path: Path, ext: str) -> list[dict[str, Any]]:
|
||
"""Render ``.xls``/``.ods`` sheets with the same dict shape as xlsx.
|
||
|
||
Read-only formats (#153 A16): ``styles``/``aligns``/``merges``/``freeze``
|
||
are served empty so the client-side wiring keeps one code path. Raises
|
||
nothing to the render path: an unreadable file yields one error sheet.
|
||
"""
|
||
name = file_path.name
|
||
try:
|
||
if ext == ".xls":
|
||
import xlrd
|
||
|
||
book = xlrd.open_workbook(str(file_path))
|
||
titles = book.sheet_names()
|
||
grids = []
|
||
for si in range(book.nsheets):
|
||
sh = book.sheet_by_index(si)
|
||
grid = [
|
||
[_fmt(sh.cell_value(r, c)) for c in range(min(sh.ncols, MAX_COLS))]
|
||
for r in range(min(sh.nrows, MAX_ROWS))
|
||
]
|
||
grids.append(_trim(grid))
|
||
total = [(sh.nrows, sh.ncols) for sh in (book.sheet_by_index(i) for i in range(book.nsheets))]
|
||
elif ext == ".ods":
|
||
from odf.opendocument import load as odf_load
|
||
from odf.table import Table, TableCell, TableRow
|
||
from odf.teletype import extractText
|
||
|
||
doc = odf_load(str(file_path))
|
||
titles = []
|
||
grids = []
|
||
total = []
|
||
for table in doc.getElementsByType(Table):
|
||
title = table.getAttribute("name") or f"Feuille {len(titles) + 1}"
|
||
titles.append(title)
|
||
grid = []
|
||
for row in table.getElementsByType(TableRow)[:MAX_ROWS]:
|
||
row_cells = row.getElementsByType(TableCell)
|
||
values: list[str] = []
|
||
for tc in row_cells[:MAX_COLS]:
|
||
repeat = int(tc.getAttribute("numbercolumnsrepeated") or 1)
|
||
values.extend([extractText(tc)] * min(repeat, MAX_COLS - len(values)))
|
||
grid.append(values)
|
||
grids.append(_trim(grid))
|
||
total.append((len(grid), max((len(r) for r in grid), default=0)))
|
||
else:
|
||
raise ValueError(f"Unsupported legacy format: {ext}")
|
||
except Exception as exc:
|
||
logger.warning("legacy workbook render failed for %s: %s", name, exc)
|
||
return [
|
||
{
|
||
"name": name,
|
||
"html": (
|
||
'<p><em>Feuille vide</em></p>'
|
||
),
|
||
"rows": 0,
|
||
"cols": 0,
|
||
"total_rows": 0,
|
||
"total_cols": 0,
|
||
"max_rows": MAX_ROWS,
|
||
"max_cols": MAX_COLS,
|
||
"truncated": False,
|
||
"styles": {},
|
||
"aligns": {},
|
||
"merges": [],
|
||
"freeze": "",
|
||
}
|
||
]
|
||
|
||
sheets: list[dict[str, Any]] = []
|
||
for i, title in enumerate(titles):
|
||
grid = grids[i] if i < len(grids) else []
|
||
t_rows, t_cols = total[i] if i < len(total) else (0, 0)
|
||
sheets.append(
|
||
{
|
||
"name": title,
|
||
"html": _table(grid),
|
||
"rows": len(grid),
|
||
"cols": max((len(r) for r in grid), default=0),
|
||
"total_rows": t_rows,
|
||
"total_cols": t_cols,
|
||
"max_rows": MAX_ROWS,
|
||
"max_cols": MAX_COLS,
|
||
"truncated": t_rows > MAX_ROWS or t_cols > MAX_COLS,
|
||
"styles": {},
|
||
"aligns": {},
|
||
"merges": [],
|
||
"freeze": "",
|
||
}
|
||
)
|
||
return sheets
|