feat: troncature annoncee et lecture par fenetres des tableurs #153
- BUG-090 (#153 A8) : render_sheets() expose total_rows/total_cols, max_rows/max_cols et truncated ; la visionneuse affiche un bandeau « Feuille tronquée » (i18n FR/EN) au lieu de couper en silence, et la ligne d'en-têtes devient sticky (top:auto sur les numéros de ligne). - #153 A9 : GET /api/file/{vault}/xlsx/sheet?sheet&offset&limit sert une fenêtre de 1 à 1000 lignes avec les vraies coordonnées A1, has_more de pagination et valeurs calculées A12 ; 404 feuille inconnue, 415 non-xlsx. - Tests : TestXlsxTruncationNotice (4) + TestXlsxSheetWindow (11) avec contre-preuves, xlsx-viewer.test.mjs 14/14, E2E 7/7 (fixture sample-xlsx-large.xlsx 520 lignes), suite 1417 passed / 6 skipped, ruff/mypy 0, i18n parity. 🤖 Generated with Codebuff Co-Authored-By: Codebuff <[email protected]>
This commit is contained in:
@@ -185,6 +185,26 @@ _ENDPOINT_EXAMPLES: dict[tuple[str, str], dict[str, Any]] = {
|
||||
"request": {"sheet": "Budget", "cells": {"B1": "250"}, "allow_formula": False, "force": False},
|
||||
"response": {"status": "ok", "vault": "TestVault", "path": "data/budget.xlsx", "size": 1},
|
||||
},
|
||||
# GET : pas d'exemple de requête (un requestBody sur un GET serait un OpenAPI
|
||||
# invalide) — les paramètres sont documentés par leurs Query().
|
||||
("get", "/api/file/{vault_name}/xlsx/sheet"): {
|
||||
"response": {
|
||||
"vault": "TestVault",
|
||||
"path": "data/budget.xlsx",
|
||||
"sheet": "Budget",
|
||||
"offset": 0,
|
||||
"limit": 200,
|
||||
"rows": 2,
|
||||
"cols": 2,
|
||||
"total_rows": 640,
|
||||
"total_cols": 12,
|
||||
"max_rows": 500,
|
||||
"max_cols": 40,
|
||||
"truncated": True,
|
||||
"has_more": True,
|
||||
"html": "<table>…</table>",
|
||||
},
|
||||
},
|
||||
("post", "/api/search/replace"): {
|
||||
"request": {"query": "Python", "replacement": "Python 3", "vault": "all", "dry_run": True},
|
||||
"response": {"matches": [{"vault": "TestVault", "path": "note1.md", "title": "Python", "match_count": 3}], "total_matches": 3, "dry_run": True},
|
||||
|
||||
@@ -38,6 +38,7 @@ from backend.schemas import (
|
||||
BrowseResponse,
|
||||
FileContentResponse,
|
||||
FileRawResponse,
|
||||
XlsxSheetWindowResponse,
|
||||
)
|
||||
from backend.services.files import read_raw_file
|
||||
from backend.services.paths import resolve_safe_path
|
||||
@@ -178,6 +179,71 @@ async def api_file_backlinks(
|
||||
}
|
||||
|
||||
|
||||
@router.get(
|
||||
"/api/file/{vault_name}/xlsx/sheet", response_model=XlsxSheetWindowResponse
|
||||
)
|
||||
def api_file_xlsx_sheet(
|
||||
vault_name: str,
|
||||
path: str = Query(..., description="Relative path to the .xlsx file"),
|
||||
sheet: str = Query(..., description="Sheet name (as shown in the viewer tab)"),
|
||||
offset: int = Query(0, ge=0, description="0-based index of the first row to return"),
|
||||
limit: int = Query(
|
||||
200, ge=1, le=1000, description="Rows to return (server-capped)"
|
||||
),
|
||||
current_user=Depends(require_auth),
|
||||
):
|
||||
"""Return a window of rows of one sheet of an .xlsx workbook (#153 A9).
|
||||
|
||||
Backs the viewer's lazy loading: instead of every sheet in a single JSON
|
||||
payload, the client asks for the block it is about to display. The row
|
||||
numbers and the ``data-cell`` references are the real A1 coordinates of the
|
||||
sheet, so a window behaves like the full render (editing a cell in it
|
||||
targets the right cell).
|
||||
|
||||
The response also carries ``total_rows``/``total_cols`` and the ``truncated``
|
||||
flag, so the client can say what is hidden behind the 500x40 render caps
|
||||
instead of silently hiding it.
|
||||
|
||||
Args:
|
||||
vault_name: Name of the vault.
|
||||
path: Relative path of the .xlsx file within the vault.
|
||||
sheet: Sheet name; **404** if the workbook has no such sheet.
|
||||
offset: 0-based index of the first row to return.
|
||||
limit: Rows to return, capped server-side at 1000.
|
||||
|
||||
Returns:
|
||||
``XlsxSheetWindowResponse`` with the rendered ``html`` of the window.
|
||||
|
||||
Raises:
|
||||
HTTPException: 403 (vault access), 404 (vault, file or sheet unknown),
|
||||
415 (not an .xlsx file), 500 (unreadable workbook).
|
||||
"""
|
||||
if not check_vault_access(vault_name, current_user):
|
||||
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
|
||||
vault_data = get_vault_data(vault_name)
|
||||
if not vault_data:
|
||||
raise HTTPException(status_code=404, detail=f"Vault '{vault_name}' not found")
|
||||
|
||||
file_path = resolve_safe_path(Path(vault_data["path"]), path)
|
||||
if not file_path.is_file():
|
||||
raise HTTPException(status_code=404, detail=f"File not found: {path}")
|
||||
if file_path.suffix.lower() != ".xlsx":
|
||||
raise HTTPException(status_code=415, detail="Le fichier n'est pas un classeur .xlsx")
|
||||
|
||||
# Import tardif : openpyxl n'est chargé que si un .xlsx est réellement demandé.
|
||||
from backend.xlsx_reader import read_sheet_window
|
||||
|
||||
try:
|
||||
window = read_sheet_window(file_path, sheet, offset=offset, limit=limit)
|
||||
except Exception as e:
|
||||
logger.error(f"XLSX sheet read error for {path}: {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
|
||||
if window is None:
|
||||
raise HTTPException(status_code=404, detail=f"Feuille introuvable: {sheet}")
|
||||
|
||||
return {"vault": vault_name, "path": path, **window}
|
||||
|
||||
|
||||
@router.get("/api/file/{vault_name}", response_model=FileContentResponse)
|
||||
async def api_file(vault_name: str, path: str = Query(..., description="Relative path to file"), current_user=Depends(require_auth)):
|
||||
"""Return rendered HTML and metadata for a file.
|
||||
|
||||
+32
-1
@@ -286,7 +286,12 @@ class FileContentResponse(BaseModel):
|
||||
is_csv: bool | None = Field(default=None, description="True for CSV files")
|
||||
is_xlsx: bool | None = Field(default=None, description="True for Excel .xlsx files")
|
||||
xlsx_sheets: list[dict[str, Any]] | None = Field(
|
||||
default=None, description="Rendered xlsx sheets [{name, html}]"
|
||||
default=None,
|
||||
description=(
|
||||
"Rendered xlsx sheets [{name, html, rows, cols, total_rows, "
|
||||
"total_cols, max_rows, max_cols, truncated}] — `truncated` is true "
|
||||
"when the sheet exceeds the 500x40 render caps (#153 A8)"
|
||||
),
|
||||
)
|
||||
xlsx_lossy_features: list[str] | None = Field(
|
||||
default=None,
|
||||
@@ -305,6 +310,32 @@ class FileContentResponse(BaseModel):
|
||||
image_mime: str | None = Field(default=None, description="MIME type for image files")
|
||||
|
||||
|
||||
class XlsxSheetWindowResponse(BaseModel):
|
||||
"""One window of rows of a single .xlsx sheet (lazy loading, #153 A9).
|
||||
|
||||
Served by ``GET /api/file/{vault_name}/xlsx/sheet``; the row numbers and
|
||||
the ``data-cell`` references in ``html`` are the real A1 coordinates of the
|
||||
sheet, whatever the window.
|
||||
"""
|
||||
|
||||
vault: str = Field(description="Vault name")
|
||||
path: str = Field(description="Relative file path within the vault")
|
||||
sheet: str = Field(description="Sheet name (as shown in the tab)")
|
||||
offset: int = Field(description="0-based index of the first returned row")
|
||||
limit: int = Field(description="Maximum number of rows returned (capped server-side)")
|
||||
rows: int = Field(description="Rows actually returned in this window")
|
||||
cols: int = Field(description="Columns of the rendered window")
|
||||
total_rows: int = Field(description="Rows the sheet declares")
|
||||
total_cols: int = Field(description="Columns the sheet declares")
|
||||
max_rows: int = Field(description="Row cap of the renderer (500) — the coverage of this window")
|
||||
max_cols: int = Field(description="Column cap of the renderer (40)")
|
||||
truncated: bool = Field(
|
||||
description="True when the sheet exceeds the 500x40 render caps"
|
||||
)
|
||||
has_more: bool = Field(description="True when rows remain after this window")
|
||||
html: str = Field(description="Rendered HTML table for the window")
|
||||
|
||||
|
||||
class FileRawResponse(BaseModel):
|
||||
"""Raw text content of a file."""
|
||||
|
||||
|
||||
+145
-8
@@ -28,6 +28,12 @@ logger = logging.getLogger("obsigate.xlsx_reader")
|
||||
MAX_ROWS = 500
|
||||
MAX_COLS = 40
|
||||
|
||||
# #153 A9 — window size served by ``read_sheet_window()`` (lazy per-sheet
|
||||
# loading). The endpoint is bounded so a single request can never ask for the
|
||||
# whole workbook back in one JSON payload; the UI pages through the rest.
|
||||
MAX_WINDOW_ROWS = 1_000
|
||||
DEFAULT_WINDOW_ROWS = 200
|
||||
|
||||
# #153 A1 — workbook parts openpyxl does not re-serialize on load+save.
|
||||
# Verified against openpyxl 3.1.5: charts, images, drawings and pivot tables
|
||||
# DO survive the round-trip, so they are deliberately absent from this map.
|
||||
@@ -98,7 +104,11 @@ def _cell_cached(cached: list[list[str]] | None, r: int, c: int) -> str:
|
||||
return row[c] if c < len(row) else ""
|
||||
|
||||
|
||||
def _table(grid: list[list[str]], cached: list[list[str]] | None = None) -> str:
|
||||
def _table(
|
||||
grid: list[list[str]],
|
||||
cached: list[list[str]] | None = None,
|
||||
row_offset: int = 0,
|
||||
) -> str:
|
||||
"""Render a grid as an HTML table.
|
||||
|
||||
``cached`` is the same grid read with ``data_only=True`` (#153 A12): where a
|
||||
@@ -107,6 +117,10 @@ def _table(grid: list[list[str]], cached: list[list[str]] | None = None) -> str:
|
||||
number Excel last calculated instead of only the formula text. The span
|
||||
carries ``data-cached-value`` and is titled client-side from
|
||||
``xlsx.cached_value_title`` — the backend never emits UI text.
|
||||
|
||||
``row_offset`` is the number of rows skipped before this grid (#153 A9): the
|
||||
row numbers and the ``data-cell`` references must stay the real A1
|
||||
coordinates of the sheet, not of the window.
|
||||
"""
|
||||
if not grid:
|
||||
return "<p><em>Feuille vide</em></p>"
|
||||
@@ -119,7 +133,7 @@ def _table(grid: list[list[str]], cached: list[list[str]] | None = None) -> str:
|
||||
]
|
||||
out += [f"<th>{get_column_letter(c)}</th>" for c in range(1, n_cols + 1)]
|
||||
out.append("</tr></thead><tbody>")
|
||||
for r, row in enumerate(grid, start=1):
|
||||
for r, row in enumerate(grid, start=row_offset + 1):
|
||||
out.append(f'<tr><th class="xlsx-rownum">{r}</th>')
|
||||
for c, val in enumerate(row, start=1):
|
||||
ref = f"{get_column_letter(c)}{r}"
|
||||
@@ -191,19 +205,24 @@ def inspect_workbook(file_path: Path) -> list[str]:
|
||||
return []
|
||||
|
||||
|
||||
def render_sheets(file_path: Path) -> list[dict[str, str]]:
|
||||
"""Return ``[{"name": sheet_title, "html": table_html}, ...]``.
|
||||
def render_sheets(file_path: Path) -> list[dict[str, Any]]:
|
||||
"""Return one dict per sheet: ``{name, html, rows, cols, total_*, truncated}``.
|
||||
|
||||
Reads the workbook twice: once with ``data_only=False`` for the formulas
|
||||
(what the user must edit) and, when any formula carries a cached result
|
||||
(#153 A12), once with ``data_only=True`` to show what Excel last computed.
|
||||
The second pass is skipped entirely when the archive holds no cached value,
|
||||
so the common case still costs a single load.
|
||||
|
||||
``total_rows``/``total_cols`` are the dimensions the sheet declares and
|
||||
``truncated`` says whether the hard caps actually cut it (#153 A8) — the
|
||||
viewer needs both to stop silently hiding the tail of a sheet.
|
||||
"""
|
||||
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
||||
try:
|
||||
formulas = [_sheet_grid(ws) for ws in wb.worksheets]
|
||||
titles = [ws.title for ws in wb.worksheets]
|
||||
extents = [_sheet_extent(ws) for ws in wb.worksheets]
|
||||
finally:
|
||||
wb.close()
|
||||
|
||||
@@ -219,16 +238,134 @@ def render_sheets(file_path: Path) -> list[dict[str, str]]:
|
||||
# shift every cached value left of its formula. Indexing it
|
||||
# positionally against the untrimmed grid keeps the two aligned.
|
||||
shadow = cached[i] if cached is not None and i < len(cached) else None
|
||||
sheets.append({"name": title, "html": _table(grid, shadow)})
|
||||
total_rows, total_cols = extents[i]
|
||||
sheets.append(
|
||||
{
|
||||
"name": title,
|
||||
"html": _table(grid, shadow),
|
||||
"rows": len(grid),
|
||||
"cols": max((len(r) for r in grid), default=0),
|
||||
"total_rows": total_rows,
|
||||
"total_cols": total_cols,
|
||||
# Coverage, not display size: `rows`/`cols` are post-trim (a
|
||||
# sheet of 3 filled cells in a 500-row block renders 1x1), and
|
||||
# the client must announce the cap it stopped at, not how many
|
||||
# cells happen to be non-empty.
|
||||
"max_rows": MAX_ROWS,
|
||||
"max_cols": MAX_COLS,
|
||||
# A sheet is truncated when the caps, not the trailing blanks,
|
||||
# decided its shape: comparing against the *rendered* size would
|
||||
# flag every sheet carrying a few empty formatted rows.
|
||||
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
||||
}
|
||||
)
|
||||
return sheets
|
||||
|
||||
|
||||
def _sheet_grid(ws: Any) -> list[list[str]]:
|
||||
"""Read one worksheet into a grid of formatted strings, bounded by the caps."""
|
||||
def read_sheet_window(
|
||||
file_path: Path,
|
||||
sheet: str,
|
||||
offset: int = 0,
|
||||
limit: int = DEFAULT_WINDOW_ROWS,
|
||||
) -> dict[str, Any] | None:
|
||||
"""Return a window of rows of one sheet, or ``None`` if the sheet is unknown.
|
||||
|
||||
Backs the lazy per-sheet loading of #153 A9: the viewer asks for the rows
|
||||
it is about to display instead of shipping every sheet in the initial file
|
||||
payload. ``offset`` is 0-based; the row numbers and the ``data-cell``
|
||||
references in the returned ``html`` are the real A1 coordinates of the
|
||||
sheet, so a window is indistinguishable from a full render.
|
||||
|
||||
``limit`` is clamped to :data:`MAX_WINDOW_ROWS`. Raises nothing: an unknown
|
||||
sheet yields ``None`` and a broken workbook propagates the caller's usual
|
||||
500.
|
||||
"""
|
||||
offset = max(int(offset), 0)
|
||||
limit = min(max(int(limit), 1), MAX_WINDOW_ROWS)
|
||||
|
||||
wb = load_workbook(str(file_path), read_only=True, data_only=False)
|
||||
try:
|
||||
if sheet not in wb.sheetnames:
|
||||
return None
|
||||
ws = wb[sheet]
|
||||
total_rows, total_cols = _sheet_extent(ws)
|
||||
grid = _trim(
|
||||
_sheet_grid(ws, min_row=offset + 1, max_row=offset + limit)
|
||||
)
|
||||
finally:
|
||||
wb.close()
|
||||
|
||||
shadow: list[list[str]] | None = None
|
||||
# Same A12 rule as the full render: the second read only happens when the
|
||||
# archive really holds cached results.
|
||||
if _has_cached_values(file_path):
|
||||
shadow = _read_cached_window(file_path, sheet, offset, limit)
|
||||
return {
|
||||
"sheet": sheet,
|
||||
"offset": offset,
|
||||
"limit": limit,
|
||||
"rows": len(grid),
|
||||
"cols": max((len(r) for r in grid), default=0),
|
||||
"total_rows": total_rows,
|
||||
"total_cols": total_cols,
|
||||
"max_rows": MAX_ROWS,
|
||||
"max_cols": MAX_COLS,
|
||||
"truncated": total_rows > MAX_ROWS or total_cols > MAX_COLS,
|
||||
"has_more": offset + len(grid) < total_rows,
|
||||
"html": _table(grid, shadow, row_offset=offset),
|
||||
}
|
||||
|
||||
|
||||
def _read_cached_window(
|
||||
file_path: Path, sheet: str, offset: int, limit: int
|
||||
) -> list[list[str]] | None:
|
||||
"""``data_only=True`` grid for one window, or ``None`` if unavailable.
|
||||
|
||||
Best effort like :func:`_read_cached_grids`: a workbook Excel opens but
|
||||
openpyxl cannot re-read must still display (formulas only).
|
||||
"""
|
||||
try:
|
||||
wb = load_workbook(str(file_path), read_only=True, data_only=True)
|
||||
except Exception:
|
||||
return None
|
||||
try:
|
||||
if sheet not in wb.sheetnames:
|
||||
return None
|
||||
return _sheet_grid(
|
||||
wb[sheet], min_row=offset + 1, max_row=offset + limit
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("xlsx cached window unavailable", exc_info=True)
|
||||
return None
|
||||
finally:
|
||||
wb.close()
|
||||
|
||||
|
||||
def _sheet_extent(ws: Any) -> tuple[int, int]:
|
||||
"""Rows and columns the worksheet declares, never negative.
|
||||
|
||||
``max_row``/``max_column`` come from the sheet's dimension record; a
|
||||
hand-edited file may omit it, hence the defensive coercion.
|
||||
"""
|
||||
try:
|
||||
rows = max(int(getattr(ws, "max_row", 0) or 0), 0)
|
||||
except (TypeError, ValueError):
|
||||
rows = 0
|
||||
try:
|
||||
cols = max(int(getattr(ws, "max_column", 0) or 0), 0)
|
||||
except (TypeError, ValueError):
|
||||
cols = 0
|
||||
return rows, cols
|
||||
|
||||
|
||||
def _sheet_grid(
|
||||
ws: Any, min_row: int = 1, max_row: int = MAX_ROWS, max_col: int = MAX_COLS
|
||||
) -> list[list[str]]:
|
||||
"""Read a worksheet window into a grid of formatted strings, bounded by the caps."""
|
||||
return [
|
||||
[_fmt(v) for v in row]
|
||||
for row in ws.iter_rows(
|
||||
min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS, values_only=True
|
||||
min_row=min_row, max_row=max_row, max_col=max_col, values_only=True
|
||||
)
|
||||
]
|
||||
|
||||
|
||||
Reference in New Issue
Block a user