feat: tableau de bord classeur - plages nommees, TCD/graphiques et KPI par feuille #153

A17 — nouveau endpoint GET /api/file/{vault}/xlsx/dashboard (read_workbook_
dashboard : plages nommees avec portee depuis defined_names read_only,
comptage graphiques/TCD par parts OPC, stats par feuille bornées 500x40 :
cellules/lignes/colonnes/formules/numerique + 8 premieres valeurs en cartes
KPI) et panneau frontend toggled depuis la toolbar (table des plages,
cartes KPI par feuille, hint actions IA). Bouton absent pour .csv et
formats en lecture seule ; le menu structure est saute quand le bouton
n'existe pas. i18n FR/EN (xlsx.dashboard_*), 8 tests backend + 2 tests
JSDOM + contre-preuve (5 echecs sur neutralisation), ruff/mypy 0.

🤖 Generated with Codebuff
Co-Authored-By: Codebuff <[email protected]>
This commit is contained in:
2026-09-28 16:40:09 -04:00
parent ca6407e0c0
commit c4b8e66206
18 changed files with 562 additions and 15 deletions
+43 -1
View File
@@ -38,11 +38,12 @@ from backend.schemas import (
BrowseResponse,
FileContentResponse,
FileRawResponse,
XlsxDashboardResponse,
XlsxSheetWindowResponse,
)
from backend.services.files import read_raw_file
from backend.services.paths import resolve_safe_path
from backend.services.vaults import browse_directory
from backend.services.vaults import browse_directory, get_vault_root
logger = logging.getLogger("obsigate")
@@ -179,6 +180,47 @@ async def api_file_backlinks(
}
@router.get(
"/api/file/{vault_name}/xlsx/dashboard", response_model=XlsxDashboardResponse
)
def api_file_xlsx_dashboard(
vault_name: str,
path: str = Query(..., description="Relative path to the .xlsx workbook"),
current_user=Depends(require_auth),
):
"""Return the dashboard metadata of an .xlsx workbook (#153 A17).
Named ranges (workbook- or sheet-scoped), chart/pivot object counts and
per-sheet KPI stats (non-empty cells, rows/cols coverage, formulas,
numeric cells, first numeric values as KPI cards). Read-only, bounded by
the 500x40 render caps; never raises for an unreadable workbook — an
empty payload comes back and the viewer hides the panel.
"""
if not check_vault_access(vault_name, current_user):
raise HTTPException(status_code=403, detail=f"Accès refusé à la vault '{vault_name}'")
_vault_root = get_vault_root(vault_name)
file_path = resolve_safe_path(_vault_root, path)
if not file_path.is_file():
raise HTTPException(status_code=404, detail=f"File not found: {path}")
if file_path.suffix.lower() not in (".xlsx", ".xlsm"):
raise HTTPException(
status_code=415, detail="Le fichier n'est pas un classeur .xlsx/.xlsm"
)
from backend.xlsx_reader import read_workbook_dashboard
try:
dashboard = read_workbook_dashboard(file_path)
except Exception as e:
logger.error(f"XLSX dashboard read error for {path}: {e}")
raise HTTPException(status_code=500, detail=f"Error reading XLSX: {e!s}")
return {
"vault": vault_name,
"path": path,
**dashboard,
}
@router.get(
"/api/file/{vault_name}/xlsx/sheet", response_model=XlsxSheetWindowResponse
)
+37
View File
@@ -318,6 +318,43 @@ class FileContentResponse(BaseModel):
image_mime: str | None = Field(default=None, description="MIME type for image files")
class XlsxDashboardNamedRange(BaseModel):
"""One named range of a workbook (#153 A17)."""
name: str = Field(description="Range name as declared in the workbook")
scope: str = Field(description="Sheet name when sheet-scoped, empty when workbook-wide")
ref: str = Field(description="Formula-style reference, e.g. Data!$A$1:$B$5")
class XlsxDashboardSheetKpi(BaseModel):
"""One KPI card of a sheet dashboard (#153 A17)."""
label: str = Field(description="A1 reference of the numeric cell")
value: float = Field(description="Numeric value of the cell")
class XlsxDashboardSheet(BaseModel):
"""Per-sheet KPI stats of a workbook dashboard (#153 A17)."""
name: str = Field(description="Sheet name")
cells: int = Field(description="Non-empty cells inside the 500x40 caps")
rows: int = Field(description="Rows carrying at least one non-empty cell")
cols: int = Field(description="Columns carrying at least one non-empty cell")
formulas: int = Field(description="Cells whose value is a formula")
numeric: int = Field(description="Cells carrying a numeric value")
kpi: list[XlsxDashboardSheetKpi] = Field(description="First numeric cells as KPI cards")
class XlsxDashboardResponse(BaseModel):
"""Dashboard metadata of an .xlsx workbook (#153 A17)."""
vault: str = Field(description="Vault name")
path: str = Field(description="Relative file path within the vault")
named_ranges: list[XlsxDashboardNamedRange] = Field(description="Named ranges, sorted by name")
objects: dict[str, int] = Field(description="Object counts: {charts, pivots}")
sheets: list[XlsxDashboardSheet] = Field(description="Per-sheet KPI stats")
class XlsxSheetWindowResponse(BaseModel):
"""One window of rows of a single .xlsx sheet (lazy loading, #153 A9).
+122
View File
@@ -61,6 +61,11 @@ _CACHED_FORMULA_RE = re.compile(rb"<f[ >][^<]*</f>\s*<v>[^<]")
# Sheet XML scanned by the cached-formula probe (CPU guard, like MAX_REPLACE_FILE_BYTES).
_MAX_PROBE_BYTES = 8_000_000
# #153 A17 — OPC parts of chart / pivot objects, matched against the archive
# name list (xl/charts/chart1.xml, xl/pivotTables/pivotTable1.xml, …).
_CHART_PART_RE = re.compile(r"^xl/charts/chart\d+\.xml$")
_PIVOT_PART_RE = re.compile(r"^xl/pivotTables/pivotTable\d+\.xml$")
# #153 A5 — ceiling on the text handed to the TF-IDF / semantic index. A workbook
# is a data dump, not prose: indexing every cell would flood the inverted index
# and bury the notes. Sheet names + the first rows are enough to make a
@@ -645,6 +650,123 @@ def extract_indexable_text(file_path: Path) -> str:
return "\n".join(c for c in chunks if c).strip()
# ── #153 A17 — dashboard metadata ───────────────────────────────────
def read_workbook_dashboard(file_path: Path) -> dict[str, Any]:
"""Return the dashboard metadata of a workbook (#153 A17).
Shape::
{
"named_ranges": [{"name", "scope", "ref"}],
"objects": {"charts": int, "pivots": int},
"sheets": [{
"name": str,
"cells": int, # non-empty cells inside the caps
"rows": int, # rows carrying at least one non-empty cell
"cols": int, # columns carrying at least one non-empty cell
"formulas": int,
"numeric": int,
"kpi": [ # first 8 numeric cells as {"label", "value"}
{"label": str, "value": float}
],
}],
}
Named ranges come from the streaming load (available read-only), cell
stats from ``iter_rows(values_only=True)``. Charts/pivots are counted by
OPC part names (a chart part per chart, a pivot table part per pivot).
Bounded by MAX_ROWS/MAX_COLS; never raises — a failure yields an empty
payload and the viewer simply hides the panel.
"""
payload: dict[str, Any] = {
"named_ranges": [],
"objects": {"charts": 0, "pivots": 0},
"sheets": [],
}
try:
wb = load_workbook(str(file_path), read_only=True, data_only=False)
except Exception:
return payload
try:
dn = getattr(wb, "defined_names", None)
items: list[tuple[Any, Any]] = (
list(dn.items()) if dn is not None and hasattr(dn, "items") else []
)
for name, defn in items:
scope_idx = getattr(defn, "localSheetId", None)
scope = ""
if scope_idx is not None:
try:
scope = wb.sheetnames[int(scope_idx)]
except (IndexError, ValueError):
scope = ""
payload["named_ranges"].append(
{
"name": str(name),
"scope": scope,
"ref": str(getattr(defn, "attr_text", "") or ""),
}
)
payload["named_ranges"].sort(key=lambda d: d["name"].lower())
for ws in wb.worksheets:
cells = rows = formulas = numeric = 0
col_seen: set[int] = set()
kpi: list[dict[str, Any]] = []
for r, row in enumerate(
ws.iter_rows(min_row=1, max_row=MAX_ROWS, max_col=MAX_COLS, values_only=True),
start=1,
):
row_has_value = False
for c, value in enumerate(row, start=1):
if value is None or (isinstance(value, str) and not value.strip()):
continue
cells += 1
col_seen.add(c)
row_has_value = True
if isinstance(value, str) and value.startswith("="):
formulas += 1
elif isinstance(value, bool):
pass
elif isinstance(value, (int, float)):
numeric += 1
if len(kpi) < 8:
kpi.append(
{"label": f"{get_column_letter(c)}{r}", "value": value}
)
if row_has_value:
rows += 1
payload["sheets"].append(
{
"name": ws.title,
"cells": cells,
"rows": rows,
"cols": len(col_seen),
"formulas": formulas,
"numeric": numeric,
"kpi": kpi,
}
)
# Chart/pivot parts, counted from the archive (chart XML parts are
# one per chart; pivot parts one per pivot table/cache).
with zipfile.ZipFile(file_path) as zf:
names = zf.namelist()
payload["objects"]["charts"] = sum(1 for n in names if _CHART_PART_RE.match(n))
payload["objects"]["pivots"] = sum(1 for n in names if _PIVOT_PART_RE.match(n))
return payload
except Exception:
logger.debug("xlsx dashboard unavailable", exc_info=True)
return {
"named_ranges": [],
"objects": {"charts": 0, "pivots": 0},
"sheets": [],
}
finally:
wb.close()
# ── #153 A16 — additional spreadsheet formats ───────────────────────────────