L'editeur lisait les styles mais ne les ecrivait pas, exportait la feuille entiere et laissait le dernier-ecrivain gagner entre processus. - A8 : bouton Mise en forme (gras/italique/souligne, alignements, couleurs via selecteur natif, formats de nombre, fusion, volets figes, largeur/hauteur) et nouvelle route PUT .../xlsx/style (verrou, backup, swap atomique, garde de perte, If-Match) ; A9 : decision "pas de moteur de formule" annoncee dans l'UI ; A10 : undo/redo unifie, piles par fichier conservees au re-rendu - A11 : export de la selection + Markdown/HTML/impression et recherche sur toutes les feuilles ; A12 : concurrence optimiste (If-Match -> 409 reparable, retry qui relit) ; A13 : cache des metadonnees par (chemin, mtime, taille) - A14 : outils IA .xlsm/.csv + search_workbook, analyze_range, edit_xlsx_structure - tests : test_xlsx_styles.py (17), test_spreadsheet_tools.py (34), TestOptimisticConcurrency/TestMetaCache, JSDOM xlsx-viewer 107/107
963 lines
39 KiB
Python
963 lines
39 KiB
Python
"""Display / edit / download for .xlsx files (viewer + PUT xlsx/save).
|
|
|
|
Covers #152 (affichage / édition) and #153 P0 : A1 alerte de fidélité avant
|
|
écriture, A2 écriture atomique, A3 verrou par fichier, A4 neutralisation de
|
|
l'injection de formule.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import zipfile
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
openpyxl = pytest.importorskip("openpyxl")
|
|
|
|
VAULT = "TestVault"
|
|
|
|
|
|
@pytest.fixture
|
|
def xlsx_file(test_vault_dir: str) -> str:
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "budget.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Budget"
|
|
ws["A1"] = "Poste"
|
|
ws["B1"] = 100
|
|
ws["A2"] = "Total"
|
|
ws["B2"] = "=B1*2"
|
|
notes = wb.create_sheet("Notes")
|
|
notes["A1"] = "hello"
|
|
wb.save(path)
|
|
return str(path)
|
|
|
|
|
|
def _add_lossy_parts(path: Path, parts: dict[str, bytes]) -> None:
|
|
"""Re-pack *path* with extra OPC parts openpyxl cannot write back."""
|
|
with zipfile.ZipFile(path) as zf:
|
|
items = {n: zf.read(n) for n in zf.namelist()}
|
|
items.update(parts)
|
|
with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
for name, blob in items.items():
|
|
zf.writestr(name, blob)
|
|
|
|
|
|
@pytest.fixture
|
|
def lossy_xlsx(test_vault_dir: str) -> str:
|
|
"""Workbook with a slicer + a formula carrying its cached result."""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "risky.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Data"
|
|
ws["A1"] = 3
|
|
ws["A2"] = "=A1*3"
|
|
wb.save(path)
|
|
# <f>…</f><v>…</v> : openpyxl keeps the formula, drops the cached result.
|
|
with zipfile.ZipFile(path) as zf:
|
|
items = {n: zf.read(n) for n in zf.namelist()}
|
|
sheet = next(n for n in items if n.startswith("xl/worksheets/sheet"))
|
|
xml = items[sheet].decode("utf-8").replace(
|
|
"<f>A1*3</f>", "<f>A1*3</f><v>9</v>"
|
|
)
|
|
items[sheet] = xml.encode("utf-8")
|
|
items["xl/slicers/slicer1.xml"] = b"<slicer/>"
|
|
with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
for name, blob in items.items():
|
|
zf.writestr(name, blob)
|
|
return str(path)
|
|
|
|
|
|
# ── Display ───────────────────────────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxDisplay:
|
|
def test_renders_all_sheets(self, client, xlsx_file):
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"})
|
|
assert resp.status_code == 200
|
|
data = resp.json()
|
|
assert data["is_xlsx"] is True
|
|
assert data["unsupported"] is False
|
|
assert [s["name"] for s in data["xlsx_sheets"]] == ["Budget", "Notes"]
|
|
|
|
first = data["xlsx_sheets"][0]["html"]
|
|
assert "Poste" in first
|
|
assert 'data-cell="B1"' in first
|
|
assert "=B1*2" in first # formula kept as text (data_only=False)
|
|
assert 'data-cell="A1"' in data["xlsx_sheets"][1]["html"]
|
|
|
|
def test_corrupt_xlsx_returns_500(self, client, test_vault_dir):
|
|
bad = Path(test_vault_dir) / "corrupt.xlsx"
|
|
bad.write_bytes(b"this is not a zip archive")
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "corrupt.xlsx"})
|
|
assert resp.status_code == 500
|
|
|
|
def test_empty_sheet_renders_an_editable_blank_grid(self, client, test_vault_dir):
|
|
"""BUG-094 — a blank sheet used to render as a bare « Feuille vide »
|
|
paragraph with no cell, so a freshly added sheet could not be filled
|
|
and had no way to insert a row/column. It now exposes a small editable
|
|
grid with real A1 coordinates."""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "blank.xlsx"
|
|
wb = Workbook()
|
|
wb.active.title = "Vide"
|
|
wb.create_sheet("Vide2")
|
|
wb.save(str(path))
|
|
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "blank.xlsx"})
|
|
assert resp.status_code == 200
|
|
vide = next(s for s in resp.json()["xlsx_sheets"] if s["name"] == "Vide")
|
|
assert "Feuille vide" not in vide["html"]
|
|
assert 'data-cell="A1"' in vide["html"]
|
|
assert 'data-cell="H20"' in vide["html"] # last cell of the blank grid
|
|
assert vide["rows"] == 20
|
|
assert vide["cols"] == 8
|
|
assert vide["truncated"] is False
|
|
|
|
|
|
# ── Index parity (tree visibility) ────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxIndexing:
|
|
def test_xlsx_in_supported_extensions(self):
|
|
from backend.indexer import SUPPORTED_EXTENSIONS
|
|
|
|
assert ".xlsx" in SUPPORTED_EXTENSIONS
|
|
|
|
def test_xlsx_indexes_sheet_names_and_headers(self, test_vault_dir, xlsx_file):
|
|
"""#153 A5 — a workbook is searchable by its cell values."""
|
|
from backend.indexer import _index_single_file_sync
|
|
|
|
info = _index_single_file_sync(VAULT, test_vault_dir, xlsx_file)
|
|
assert info is not None
|
|
assert info["extension"] == ".xlsx"
|
|
# Sheet names + header rows reach TF-IDF (was metadata-only before A5).
|
|
assert "Budget" in info["content"]
|
|
assert "Poste" in info["content"]
|
|
assert info["content_preview"]
|
|
assert info["title"] # filename-derived title
|
|
|
|
|
|
class TestXlsxCachedValues:
|
|
"""#153 A12 — show what Excel last computed next to each formula."""
|
|
|
|
def test_cached_result_is_shown_beside_the_formula(self, client, lossy_xlsx):
|
|
from backend.xlsx_reader import render_sheets
|
|
|
|
html = render_sheets(Path(lossy_xlsx))[0]["html"]
|
|
# A2 is "=A1*3" with <v>9</v> in the fixture.
|
|
assert 'data-cell="A2"' in html
|
|
assert "=A1*3" in html
|
|
assert "xlsx-cached" in html
|
|
assert ">9<" in html # the cached result Excel computed
|
|
|
|
def test_no_shadow_when_no_formula_carries_a_result(self, client, xlsx_file):
|
|
from backend.xlsx_reader import render_sheets
|
|
|
|
html = render_sheets(Path(xlsx_file))[0]["html"]
|
|
assert "xlsx-cached" not in html # budget.xlsx has no <v> at all
|
|
|
|
def test_plain_cells_are_never_duplicated(self, client, lossy_xlsx):
|
|
from backend.xlsx_reader import render_sheets
|
|
|
|
html = render_sheets(Path(lossy_xlsx))[0]["html"]
|
|
# A1 is the literal 3: the two reads agree, so only one value shows.
|
|
assert 'data-cell="A1"' in html
|
|
assert html.count("xlsx-cached") == 1 # only the formula cell
|
|
|
|
|
|
class TestXlsxValueCoercion:
|
|
"""#153 A10 — a typed value comes back with the type Excel would infer."""
|
|
|
|
def _write(self, client, xlsx_file, ref, value):
|
|
return client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {ref: value}, "force": True},
|
|
)
|
|
|
|
def test_number_and_bool_are_stored_as_typed(self, client, xlsx_file):
|
|
resp = self._write(
|
|
client, xlsx_file, "D1", "42"
|
|
)
|
|
assert resp.status_code == 200
|
|
resp = self._write(client, xlsx_file, "D2", "VRAI")
|
|
assert resp.status_code == 200
|
|
resp = self._write(client, xlsx_file, "D3", "12/03/2026")
|
|
assert resp.status_code == 200
|
|
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
ws = wb["Budget"]
|
|
assert ws["D1"].value == 42 and isinstance(ws["D1"].value, int)
|
|
assert ws["D2"].value is True
|
|
assert ws["D3"].value.year == 2026 and ws["D3"].value.month == 3
|
|
assert ws["D3"].value.day == 12 # FR day-first, not 3 December
|
|
wb.close()
|
|
|
|
def test_day_first_date_is_not_read_as_us(self, client, xlsx_file):
|
|
"""'01/02/2026' is 1 February in French, not 2 January."""
|
|
assert self._write(client, xlsx_file, "E1", "01/02/2026").status_code == 200
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
d = wb["Budget"]["E1"].value
|
|
wb.close()
|
|
assert (d.month, d.day) == (2, 1)
|
|
|
|
def test_ambiguous_text_is_left_alone(self, client, xlsx_file):
|
|
"""A version string or a partial date stays text, never a date."""
|
|
assert self._write(client, xlsx_file, "F1", "3.14.2").status_code == 200
|
|
assert self._write(client, xlsx_file, "F2", "Ref 12/34").status_code == 200
|
|
assert self._write(client, xlsx_file, "F3", "12/2026").status_code == 200
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
ws = wb["Budget"]
|
|
assert ws["F1"].value == "3.14.2"
|
|
assert ws["F2"].value == "Ref 12/34"
|
|
assert ws["F3"].value == "12/2026"
|
|
wb.close()
|
|
|
|
def test_plain_integer_text_becomes_a_number(self, client, xlsx_file):
|
|
"""Typing a bare number yields a number, as it did before #153 A10."""
|
|
assert self._write(client, xlsx_file, "H1", "75001").status_code == 200
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
assert wb["Budget"]["H1"].value == 75001
|
|
wb.close()
|
|
|
|
def test_formula_looking_date_stays_text(self, client, xlsx_file):
|
|
"""A formula is never mistaken for a date (BUG-088 must not regress)."""
|
|
assert self._write(client, xlsx_file, "G1", "=12/03/2026").status_code == 200
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
c = wb["Budget"]["G1"]
|
|
wb.close()
|
|
assert c.data_type == "s"
|
|
assert c.value == "=12/03/2026"
|
|
|
|
|
|
class TestXlsxSearchable:
|
|
"""#153 A5 — a keyword living in a CELL must make the file findable."""
|
|
|
|
def test_search_finds_a_word_stored_in_a_cell(self, test_vault_dir):
|
|
"""A word typed in a CELL must make the workbook findable.
|
|
|
|
Deliberately hermetic: it drives the indexer and the inverted index
|
|
directly instead of going through the HTTP reload, because both are
|
|
process-wide singletons that other test modules mutate (some reload the
|
|
module outright), which would make this test order-dependent.
|
|
"""
|
|
from openpyxl import Workbook
|
|
|
|
import backend.indexer as indexer
|
|
import backend.search as search_mod
|
|
|
|
path = Path(test_vault_dir) / "fournisseurs.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Contacts"
|
|
ws["A1"] = "Fournisseur"
|
|
ws["A2"] = "Menuiserie Beaulieu"
|
|
wb.save(path)
|
|
|
|
# Index exactly this vault, from scratch.
|
|
indexer.index[VAULT] = {
|
|
"files": [indexer._index_single_file_sync(VAULT, test_vault_dir, str(path))],
|
|
"tags": {},
|
|
"paths": [],
|
|
"config": {},
|
|
}
|
|
entries = [f for f in indexer.index[VAULT]["files"] if f]
|
|
assert entries, "le tableur n'a pas ete indexe"
|
|
assert "Menuiserie" in entries[0]["content"], (
|
|
f"contenu indexe : {entries[0]['content'][:80]!r}"
|
|
)
|
|
|
|
search_mod.init_inverted_index()
|
|
hits = [r["path"] for r in search_mod.search("Menuiserie", "all")]
|
|
assert "fournisseurs.xlsx" in hits
|
|
|
|
def test_indexable_text_is_capped(self, tmp_path):
|
|
"""A data dump must not flood the index."""
|
|
from openpyxl import Workbook
|
|
|
|
from backend.xlsx_reader import MAX_INDEX_CHARS, extract_indexable_text
|
|
|
|
path = tmp_path / "huge.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Big"
|
|
for r in range(1, 400):
|
|
ws.cell(row=r, column=1, value=f"ligne {r} " + "x" * 60)
|
|
wb.save(path)
|
|
|
|
text = extract_indexable_text(path)
|
|
assert len(text) <= MAX_INDEX_CHARS
|
|
assert "Big" in text # the sheet name survives
|
|
|
|
def test_corrupt_workbook_indexes_as_empty_not_a_crash(self, tmp_path):
|
|
from backend.xlsx_reader import extract_indexable_text
|
|
|
|
path = tmp_path / "broken.xlsx"
|
|
path.write_bytes(b"not a zip at all")
|
|
assert extract_indexable_text(path) == ""
|
|
|
|
def test_blank_rows_are_skipped(self, tmp_path):
|
|
from openpyxl import Workbook
|
|
|
|
from backend.xlsx_reader import extract_indexable_text
|
|
|
|
path = tmp_path / "sparse.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "S"
|
|
ws["A1"] = "Alpha"
|
|
ws["A50"] = "Omega" # beyond the indexed prefix -> ignored on purpose
|
|
wb.save(path)
|
|
|
|
text = extract_indexable_text(path)
|
|
assert "Alpha" in text
|
|
assert "Omega" not in text
|
|
assert "\t\t" not in text # no run of empty columns
|
|
|
|
|
|
# ── Edit ──────────────────────────────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxSave:
|
|
def _save(self, client, body, path="budget.xlsx"):
|
|
return client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": path},
|
|
json=body,
|
|
)
|
|
|
|
def test_save_updates_cell_with_number_coercion(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"B1": "250"}})
|
|
assert resp.status_code == 200
|
|
assert resp.json()["status"] == "ok"
|
|
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
assert wb["Budget"]["B1"].value == 250 # int, not "250"
|
|
|
|
def test_save_leaves_other_sheets_and_formulas(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "Titre", "B1": 42}})
|
|
assert resp.status_code == 200
|
|
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
assert wb["Budget"]["B2"].value == "=B1*2"
|
|
assert wb["Notes"]["A1"].value == "hello"
|
|
|
|
def test_save_empty_string_clears_cell(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": ""}})
|
|
assert resp.status_code == 200
|
|
|
|
from openpyxl import load_workbook
|
|
|
|
assert load_workbook(xlsx_file)["Budget"]["A1"].value is None
|
|
|
|
def test_save_creates_backup(self, client, xlsx_file):
|
|
from backend.services.backups import get_backup_dir
|
|
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "backup-me"}})
|
|
assert resp.status_code == 200
|
|
backup_dir = Path(get_backup_dir(VAULT, "budget.xlsx"))
|
|
assert backup_dir.is_dir()
|
|
assert list(backup_dir.glob("*.bak"))
|
|
|
|
def test_unknown_sheet_400(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Nope", "cells": {"A1": "x"}})
|
|
assert resp.status_code == 400
|
|
|
|
def test_invalid_cell_ref_400(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A-1": "x"}})
|
|
assert resp.status_code == 400
|
|
|
|
def test_wrong_extension_400(self, client, test_vault_dir):
|
|
(Path(test_vault_dir) / "note.md").write_text("# hi\n", encoding="utf-8")
|
|
resp = self._save(client, {"sheet": "Sheet", "cells": {"A1": "x"}}, path="note.md")
|
|
assert resp.status_code == 400
|
|
|
|
def test_missing_file_404(self, client):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "x"}}, path="absent.xlsx")
|
|
assert resp.status_code == 404
|
|
|
|
def test_too_many_cells_400(self, client, xlsx_file):
|
|
cells = {f"A{i}": i for i in range(1, 502)}
|
|
resp = self._save(client, {"sheet": "Budget", "cells": cells})
|
|
assert resp.status_code == 400
|
|
|
|
def test_nested_value_400(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": {"nested": 1}}})
|
|
assert resp.status_code == 400
|
|
|
|
def test_missing_sheet_field_400(self, client, xlsx_file):
|
|
resp = self._save(client, {"cells": {"A1": "x"}})
|
|
assert resp.status_code == 400
|
|
|
|
def test_non_boolean_flag_400(self, client, xlsx_file):
|
|
for flag in ("force", "allow_formula"):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "x"}, flag: "yes"})
|
|
assert resp.status_code == 400, flag
|
|
|
|
|
|
# ── #153 A1 — lossy-write guard ──────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxLossyGuard:
|
|
def _save(self, client, body, path):
|
|
return client.put(
|
|
f"/api/file/{VAULT}/xlsx/save", params={"path": path}, json=body,
|
|
)
|
|
|
|
def test_read_reports_lossy_features(self, client, lossy_xlsx):
|
|
data = client.get(f"/api/file/{VAULT}", params={"path": "risky.xlsx"}).json()
|
|
assert data["is_xlsx"] is True
|
|
assert "slicers" in data["xlsx_lossy_features"]
|
|
assert "cached_values" in data["xlsx_lossy_features"]
|
|
|
|
def test_read_reports_nothing_for_a_plain_workbook(self, client, xlsx_file):
|
|
data = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"}).json()
|
|
# B2 holds "=B1*2" but openpyxl wrote no cached <v> for it.
|
|
assert data["xlsx_lossy_features"] == []
|
|
|
|
def test_save_refuses_without_force(self, client, lossy_xlsx):
|
|
resp = self._save(client, {"sheet": "Data", "cells": {"B1": "hello"}}, "risky.xlsx")
|
|
assert resp.status_code == 409
|
|
body = resp.json()
|
|
assert body["code"] == "xlsx_lossy_content"
|
|
assert "slicers" in body["details"]["features"]
|
|
|
|
def test_refused_save_leaves_the_file_untouched(self, client, lossy_xlsx):
|
|
before = Path(lossy_xlsx).read_bytes()
|
|
self._save(client, {"sheet": "Data", "cells": {"B1": "hello"}}, "risky.xlsx")
|
|
assert Path(lossy_xlsx).read_bytes() == before
|
|
|
|
def test_save_with_force_succeeds(self, client, lossy_xlsx):
|
|
resp = self._save(
|
|
client,
|
|
{"sheet": "Data", "cells": {"B1": "hello"}, "force": True},
|
|
"risky.xlsx",
|
|
)
|
|
assert resp.status_code == 200
|
|
assert openpyxl.load_workbook(lossy_xlsx)["Data"]["B1"].value == "hello"
|
|
|
|
def test_inspect_flags_every_known_family(self, lossy_xlsx):
|
|
from backend.xlsx_reader import LOSSY_PARTS, inspect_workbook
|
|
|
|
for key, prefixes in LOSSY_PARTS.items():
|
|
_add_lossy_parts(
|
|
Path(lossy_xlsx),
|
|
{f"{prefixes[0]}probe.xml": b"<x/>" for _ in [0]},
|
|
)
|
|
assert key in inspect_workbook(Path(lossy_xlsx)), key
|
|
|
|
def test_inspect_is_quiet_on_a_corrupt_archive(self, test_vault_dir):
|
|
from backend.xlsx_reader import inspect_workbook
|
|
|
|
bad = Path(test_vault_dir) / "broken.xlsx"
|
|
bad.write_bytes(b"not a zip at all")
|
|
assert inspect_workbook(bad) == []
|
|
|
|
|
|
# ── BUG-099 — the cached-value probe must never fail silently ──────────────
|
|
|
|
|
|
def _probe_zip(tmp_path: Path, sheets: list[bytes]) -> Path:
|
|
"""Build a minimal archive holding the given worksheet XML parts."""
|
|
path = tmp_path / "probe.xlsx"
|
|
with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
for i, blob in enumerate(sheets, start=1):
|
|
zf.writestr(f"xl/worksheets/sheet{i}.xml", blob)
|
|
return path
|
|
|
|
|
|
class TestXlsxCachedValueProbe:
|
|
"""BUG-099 — an exhausted probe is reported, never mistaken for clean."""
|
|
|
|
def test_exhausted_budget_is_reported_as_unverified(
|
|
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
):
|
|
from backend import xlsx_reader
|
|
|
|
# Budget smaller than the only sheet: the scan cannot reach its end.
|
|
monkeypatch.setattr(xlsx_reader, "_MAX_PROBE_BYTES_PER_SHEET", 64)
|
|
monkeypatch.setattr(xlsx_reader, "_MAX_PROBE_BYTES_TOTAL", 64)
|
|
book = _probe_zip(tmp_path, [b"<x/>" + b"filler" * 1024])
|
|
|
|
# Contre-preuve: the old code returned [] here — an interrupted scan was
|
|
# indistinguishable from a completed one, so the write guard stayed
|
|
# silent and the cached values could be dropped without warning.
|
|
assert xlsx_reader.inspect_workbook(book) == ["cached_values_unverified"]
|
|
|
|
def test_a_huge_first_sheet_does_not_starve_the_next_one(
|
|
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
):
|
|
from backend import xlsx_reader
|
|
|
|
# Per-sheet cap (1 KiB) well under the total allowance (4 KiB): a single
|
|
# shared budget would spend everything on sheet 1 and never look at
|
|
# sheet 2, where the only cached formula lives.
|
|
monkeypatch.setattr(xlsx_reader, "_MAX_PROBE_BYTES_PER_SHEET", 1024)
|
|
monkeypatch.setattr(xlsx_reader, "_MAX_PROBE_BYTES_TOTAL", 4096)
|
|
cached = b'<sheetData><c r="A1"><f>1+1</f><v>2</v></c></sheetData>'
|
|
book = _probe_zip(tmp_path, [b"filler" * 2048, cached])
|
|
|
|
assert "cached_values" in xlsx_reader.inspect_workbook(book)
|
|
|
|
def test_a_completed_clean_scan_reports_nothing(
|
|
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
):
|
|
from backend import xlsx_reader
|
|
|
|
monkeypatch.setattr(xlsx_reader, "_MAX_PROBE_BYTES_PER_SHEET", 4096)
|
|
monkeypatch.setattr(xlsx_reader, "_MAX_PROBE_BYTES_TOTAL", 8192)
|
|
book = _probe_zip(tmp_path, [b"<x/>", b"<x/>"])
|
|
|
|
# No cached value and no budget issue: no false positive either.
|
|
assert xlsx_reader.inspect_workbook(book) == []
|
|
|
|
|
|
# ── #153 A2 — atomic write ───────────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxAtomicWrite:
|
|
def test_failed_save_keeps_the_original(self, client, xlsx_file, monkeypatch):
|
|
from openpyxl.workbook.workbook import Workbook
|
|
|
|
before = Path(xlsx_file).read_bytes()
|
|
|
|
def boom(self, *args, **kwargs):
|
|
raise OSError("disk full")
|
|
|
|
monkeypatch.setattr(Workbook, "save", boom)
|
|
# TestClient re-raises the server exception (in production: 500).
|
|
with pytest.raises(OSError):
|
|
client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {"A1": "perdu"}},
|
|
)
|
|
# The workbook on disk is byte-identical : the write never reached it.
|
|
assert Path(xlsx_file).read_bytes() == before
|
|
# No temporary file left behind in the vault.
|
|
assert list(Path(xlsx_file).parent.glob("*.tmp")) == []
|
|
|
|
def test_no_tmp_left_after_a_successful_save(self, client, xlsx_file):
|
|
client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {"A1": "ok"}},
|
|
)
|
|
assert list(Path(xlsx_file).parent.glob("*.tmp")) == []
|
|
|
|
|
|
# ── #153 A3 — per-file write lock ────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxWriteLock:
|
|
def test_concurrent_write_returns_409(self, client, xlsx_file):
|
|
from backend.services import mutations
|
|
|
|
key = str(Path(xlsx_file).resolve())
|
|
with mutations._xlsx_write_lock(key):
|
|
resp = client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {"A1": "concurrent"}},
|
|
)
|
|
assert resp.status_code == 409
|
|
assert resp.json()["code"] == "conflict"
|
|
# The blocked call wrote nothing.
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A1"].value == "Poste"
|
|
|
|
def test_lock_is_released_after_a_normal_save(self, client, xlsx_file):
|
|
from backend.services import mutations
|
|
|
|
key = str(Path(xlsx_file).resolve())
|
|
client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {"A1": "premier"}},
|
|
)
|
|
# The lock must be free again once the request returned.
|
|
with mutations._xlsx_write_lock(key):
|
|
pass
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A1"].value == "premier"
|
|
|
|
|
|
@pytest.fixture
|
|
def wide_xlsx(test_vault_dir: str) -> str:
|
|
"""Workbook whose sheet exceeds BOTH render caps (501 rows x 45 cols).
|
|
|
|
Sparse on purpose: a cell in A501 and one in AS1 are enough for openpyxl
|
|
to declare those dimensions, without writing 20 000 cells to disk.
|
|
"""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "grand.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Data"
|
|
ws["A1"] = "tête"
|
|
ws["A501"] = "dernière ligne"
|
|
ws["AS1"] = "colonne 45"
|
|
wb.save(path)
|
|
return str(path)
|
|
|
|
|
|
@pytest.fixture
|
|
def edge_xlsx(test_vault_dir: str) -> str:
|
|
"""Sheet exactly on the caps (500 rows x 40 cols) — must NOT be truncated."""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "limite.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Data"
|
|
ws["A1"] = "bord"
|
|
ws["A500"] = "ligne 500"
|
|
ws["AN1"] = "colonne 40"
|
|
wb.save(path)
|
|
return str(path)
|
|
|
|
|
|
# ── #153 A8 — silent truncation made visible ─────────────────────────────
|
|
|
|
|
|
class TestXlsxTruncationNotice:
|
|
"""A sheet bigger than the caps must SAY so instead of looking complete."""
|
|
|
|
def test_render_reports_the_real_dimensions(self, client, wide_xlsx):
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "grand.xlsx"})
|
|
sheet = resp.json()["xlsx_sheets"][0]
|
|
assert (sheet["total_rows"], sheet["total_cols"]) == (501, 45)
|
|
assert sheet["truncated"] is True
|
|
# The caps are the coverage the banner announces — `rows`/`cols` are
|
|
# post-trim and would understate it on a sparse sheet.
|
|
assert (sheet["max_rows"], sheet["max_cols"]) == (500, 40)
|
|
assert (sheet["rows"], sheet["cols"]) == (1, 1) # only 3 filled cells
|
|
|
|
def test_a_sheet_on_the_caps_is_not_flagged(self, client, edge_xlsx):
|
|
"""Boundary: 500x40 is exactly what the renderer supports."""
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "limite.xlsx"})
|
|
sheet = resp.json()["xlsx_sheets"][0]
|
|
assert sheet["truncated"] is False
|
|
assert (sheet["total_rows"], sheet["total_cols"]) == (500, 40)
|
|
|
|
def test_a_normal_sheet_is_not_flagged(self, client, xlsx_file):
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"})
|
|
assert all(not s["truncated"] for s in resp.json()["xlsx_sheets"])
|
|
|
|
def test_blank_tail_is_not_reported_as_truncation(self, client, test_vault_dir):
|
|
"""A sheet with empty rows below its data fits in the caps."""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "blanc.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Data"
|
|
ws["A1"] = "seule ligne"
|
|
ws["A300"] = None # formatted-but-empty row inside the caps
|
|
wb.save(path)
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "blanc.xlsx"})
|
|
sheet = resp.json()["xlsx_sheets"][0]
|
|
assert sheet["truncated"] is False
|
|
assert sheet["rows"] == 1 # trailing blanks dropped by _trim
|
|
|
|
|
|
# ── #153 A9 — lazy per-sheet loading ─────────────────────────────────────
|
|
|
|
|
|
class TestXlsxSheetWindow:
|
|
"""GET /api/file/{vault}/xlsx/sheet — one window of one sheet."""
|
|
|
|
def _get(self, client, path="budget.xlsx", **params):
|
|
return client.get(
|
|
f"/api/file/{VAULT}/xlsx/sheet", params={"path": path, **params}
|
|
)
|
|
|
|
def test_window_returns_rows_and_totals(self, client, xlsx_file):
|
|
resp = self._get(client, sheet="Budget", offset=0, limit=10)
|
|
assert resp.status_code == 200
|
|
data = resp.json()
|
|
assert data["sheet"] == "Budget"
|
|
assert (data["offset"], data["limit"]) == (0, 10)
|
|
assert data["rows"] == 2 and data["cols"] == 2
|
|
assert data["total_rows"] == 2 and data["truncated"] is False
|
|
assert (data["max_rows"], data["max_cols"]) == (500, 40)
|
|
assert data["has_more"] is False
|
|
assert 'data-cell="A1"' in data["html"]
|
|
|
|
def test_window_keeps_the_real_a1_coordinates(self, client, xlsx_file):
|
|
"""A window must be indistinguishable from a full render: the A1
|
|
references and the row numbers have to be the sheet's, not the
|
|
window's, or an edit would land on the wrong cell. The CONTENT matters
|
|
as much as the label — row 2 of "Budget" is "Total", not "Poste"."""
|
|
data = self._get(client, sheet="Budget", offset=1, limit=1).json()
|
|
assert data["rows"] == 1
|
|
assert 'data-cell="A2"' in data["html"]
|
|
assert 'data-cell="A1"' not in data["html"]
|
|
assert "<th class=\"xlsx-rownum\">2</th>" in data["html"]
|
|
assert "Total" in data["html"] and "Poste" not in data["html"]
|
|
|
|
def test_window_offsets_walk_the_whole_sheet(self, client, wide_xlsx):
|
|
first = self._get(client, path="grand.xlsx", sheet="Data", offset=0, limit=10).json()
|
|
last = self._get(client, path="grand.xlsx", sheet="Data", offset=500, limit=10).json()
|
|
assert first["truncated"] is True and first["has_more"] is True
|
|
assert 'data-cell="A501"' in last["html"] # the row the caps used to hide
|
|
assert "dernière ligne" in last["html"]
|
|
assert "dernière ligne" not in first["html"]
|
|
assert last["rows"] == 1 and last["has_more"] is False
|
|
|
|
def test_offset_past_the_end_is_empty_not_an_error(self, client, xlsx_file):
|
|
resp = self._get(client, sheet="Budget", offset=9999, limit=10)
|
|
assert resp.status_code == 200
|
|
data = resp.json()
|
|
assert data["rows"] == 0
|
|
assert data["has_more"] is False
|
|
|
|
def test_cached_result_survives_the_window(self, client, lossy_xlsx):
|
|
"""A12 must not be lost on the lazy path."""
|
|
data = self._get(client, path="risky.xlsx", sheet="Data", offset=0, limit=10).json()
|
|
assert "xlsx-cached" in data["html"]
|
|
|
|
def test_unknown_sheet_is_404(self, client, xlsx_file):
|
|
resp = self._get(client, sheet="Nope")
|
|
assert resp.status_code == 404
|
|
assert "Feuille introuvable" in resp.json()["detail"]
|
|
|
|
def test_missing_file_is_404(self, client, test_vault_dir):
|
|
assert self._get(client, path="absent.xlsx", sheet="Budget").status_code == 404
|
|
|
|
def test_non_xlsx_file_is_415(self, client, test_vault_dir):
|
|
(Path(test_vault_dir) / "note.md").write_text("# hi", encoding="utf-8")
|
|
resp = self._get(client, path="note.md", sheet="Budget")
|
|
assert resp.status_code == 415
|
|
|
|
def test_limit_above_the_server_cap_is_rejected(self, client, xlsx_file):
|
|
"""The cap is a contract, not a silent truncation of the request."""
|
|
assert self._get(client, sheet="Budget", limit=100_000).status_code == 422
|
|
|
|
def test_reader_clamps_a_hostile_limit(self, xlsx_file):
|
|
"""Belt and braces: the reader caps too, whoever calls it."""
|
|
from backend.xlsx_reader import MAX_WINDOW_ROWS, read_sheet_window
|
|
|
|
window = read_sheet_window(Path(xlsx_file), "Budget", offset=0, limit=10**9)
|
|
assert window["limit"] == MAX_WINDOW_ROWS
|
|
|
|
def test_reader_rejects_a_negative_offset(self, xlsx_file):
|
|
from backend.xlsx_reader import read_sheet_window
|
|
|
|
window = read_sheet_window(Path(xlsx_file), "Budget", offset=-5, limit=10)
|
|
assert window["offset"] == 0
|
|
|
|
|
|
# ── #153 A4 — formula injection ──────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxFormulaGuard:
|
|
def _save(self, client, cells, **extra):
|
|
return client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": cells, **extra},
|
|
)
|
|
|
|
def test_formula_like_value_is_stored_as_text(self, client, xlsx_file):
|
|
resp = self._save(client, {"A3": "=cmd|'/c calc'!A1"})
|
|
assert resp.status_code == 200
|
|
cell = openpyxl.load_workbook(xlsx_file)["Budget"]["A3"]
|
|
assert cell.value == "=cmd|'/c calc'!A1"
|
|
assert cell.data_type == "s" # pas de <f> dans l'archive
|
|
|
|
def test_at_prefix_is_stored_as_text(self, client, xlsx_file):
|
|
self._save(client, {"A4": "@SUM(A1:A2)"})
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A4"].data_type == "s"
|
|
|
|
def test_allow_formula_keeps_a_real_formula(self, client, xlsx_file):
|
|
resp = self._save(client, {"A3": "=B1+5"}, allow_formula=True)
|
|
assert resp.status_code == 200
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A3"].data_type == "f"
|
|
|
|
def test_numbers_are_unaffected(self, client, xlsx_file):
|
|
self._save(client, {"B3": "42", "B4": "-3.5"})
|
|
ws = openpyxl.load_workbook(xlsx_file)["Budget"]
|
|
assert ws["B3"].value == 42 and isinstance(ws["B3"].value, int)
|
|
assert ws["B4"].value == -3.5
|
|
|
|
|
|
class TestMetaCache:
|
|
"""#156-A13 — ``read_workbook_meta`` is cached on (path, mtime_ns, size).
|
|
|
|
A truncated sheet is fetched window by window: without the cache every
|
|
window paid a full normal-mode workbook load just for the style / merge /
|
|
freeze maps of one sheet.
|
|
"""
|
|
|
|
def test_second_call_is_served_from_the_cache(self, xlsx_file, monkeypatch):
|
|
from backend import xlsx_reader
|
|
|
|
path = Path(xlsx_file)
|
|
xlsx_reader.invalidate_workbook_meta(path)
|
|
real_load = xlsx_reader.load_workbook
|
|
loads = []
|
|
|
|
def counting(*args, **kwargs):
|
|
loads.append(1)
|
|
return real_load(*args, **kwargs)
|
|
|
|
monkeypatch.setattr(xlsx_reader, "load_workbook", counting)
|
|
first = xlsx_reader.read_workbook_meta(path)
|
|
assert len(loads) == 1, "la première lecture charge le classeur"
|
|
second = xlsx_reader.read_workbook_meta(path)
|
|
assert len(loads) == 1, "la seconde lecture ne recharge pas le classeur"
|
|
assert second is first, "la même carte est servie (lecture seule côté appelant)"
|
|
assert "Budget" in first
|
|
|
|
def test_invalidation_forces_a_reload(self, xlsx_file, monkeypatch):
|
|
from backend import xlsx_reader
|
|
|
|
path = Path(xlsx_file)
|
|
xlsx_reader.invalidate_workbook_meta(path)
|
|
real_load = xlsx_reader.load_workbook
|
|
loads = []
|
|
|
|
def counting(*args, **kwargs):
|
|
loads.append(1)
|
|
return real_load(*args, **kwargs)
|
|
|
|
monkeypatch.setattr(xlsx_reader, "load_workbook", counting)
|
|
xlsx_reader.read_workbook_meta(path)
|
|
xlsx_reader.invalidate_workbook_meta(path)
|
|
xlsx_reader.read_workbook_meta(path)
|
|
assert len(loads) == 2, "l'invalidation explicite force un rechargement"
|
|
|
|
def test_rewriting_the_file_invalidates_the_cache(self, xlsx_file):
|
|
from backend import xlsx_reader
|
|
|
|
path = Path(xlsx_file)
|
|
xlsx_reader.invalidate_workbook_meta(path)
|
|
assert xlsx_reader.read_workbook_meta(path)["Budget"]["merges"] == []
|
|
wb = openpyxl.load_workbook(xlsx_file)
|
|
wb["Budget"].merge_cells("A1:B1")
|
|
wb.save(xlsx_file)
|
|
wb.close()
|
|
# (mtime_ns, size) changed: the cached maps cannot be served any more.
|
|
assert xlsx_reader.read_workbook_meta(path)["Budget"]["merges"] == ["A1:B1"]
|
|
|
|
def test_two_windows_share_one_metadata_load(self, xlsx_file, monkeypatch):
|
|
from backend import xlsx_reader
|
|
|
|
path = Path(xlsx_file)
|
|
xlsx_reader.invalidate_workbook_meta(path)
|
|
real_load = xlsx_reader.load_workbook
|
|
normal = []
|
|
|
|
def counting(*args, **kwargs):
|
|
if not kwargs.get("read_only"):
|
|
normal.append(1)
|
|
return real_load(*args, **kwargs)
|
|
|
|
monkeypatch.setattr(xlsx_reader, "load_workbook", counting)
|
|
assert xlsx_reader.read_sheet_window(path, "Budget", 0, 1) is not None
|
|
assert xlsx_reader.read_sheet_window(path, "Budget", 1, 1) is not None
|
|
assert len(normal) == 1, "les deux fenêtres partagent une seule lecture de métadonnées"
|
|
|
|
|
|
class TestOptimisticConcurrency:
|
|
"""#156-A12 — the read exposes a revision, a write may require it.
|
|
|
|
A stale ``if_match`` (the file changed since it was read: Excel, the
|
|
watcher, another worker) is refused with 409 ``conflict`` instead of
|
|
silently overwriting the other writer.
|
|
"""
|
|
|
|
def _save(self, client, body, path="budget.xlsx"):
|
|
return client.put(
|
|
f"/api/file/{VAULT}/xlsx/save", params={"path": path}, json=body
|
|
)
|
|
|
|
def _revision(self, client, path="budget.xlsx"):
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": path})
|
|
assert resp.status_code == 200
|
|
return resp.json()["xlsx_revision"]
|
|
|
|
def test_the_read_exposes_a_revision_and_the_write_refreshes_it(self, client, xlsx_file):
|
|
rev = self._revision(client)
|
|
assert isinstance(rev, str) and rev
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A3": "un"}, "if_match": rev})
|
|
assert resp.status_code == 200
|
|
assert resp.json()["revision"] and resp.json()["revision"] != rev
|
|
|
|
def test_a_stale_if_match_is_refused_and_writes_nothing(self, client, xlsx_file):
|
|
rev = self._revision(client)
|
|
assert self._save(
|
|
client, {"sheet": "Budget", "cells": {"A3": "un"}, "if_match": rev}
|
|
).status_code == 200
|
|
# The token is stale now: the client must re-read before writing again.
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A4": "deux"}, "if_match": rev})
|
|
assert resp.status_code == 409
|
|
assert resp.json()["code"] == "conflict"
|
|
assert resp.json()["details"]["reason"] == "stale_revision"
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A4"].value is None
|
|
|
|
def test_omitting_if_match_keeps_last_writer_wins(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A4": "sans-jeton"}})
|
|
assert resp.status_code == 200
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A4"].value == "sans-jeton"
|
|
|
|
def test_an_invalid_if_match_is_rejected(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A3": "x"}, "if_match": 42})
|
|
assert resp.status_code == 400
|
|
|
|
def test_the_structure_endpoint_honours_if_match(self, client, xlsx_file):
|
|
rev = self._revision(client)
|
|
actions = [{"op": "row_insert", "sheet": "Budget", "at": 1}]
|
|
ok = client.put(
|
|
f"/api/file/{VAULT}/xlsx/structure",
|
|
params={"path": "budget.xlsx"},
|
|
json={"actions": actions, "if_match": rev},
|
|
)
|
|
assert ok.status_code == 200
|
|
stale = client.put(
|
|
f"/api/file/{VAULT}/xlsx/structure",
|
|
params={"path": "budget.xlsx"},
|
|
json={"actions": actions, "if_match": rev},
|
|
)
|
|
assert stale.status_code == 409
|
|
assert stale.json()["code"] == "conflict"
|
|
|
|
def test_the_csv_endpoint_honours_if_match(self, client, test_vault_dir):
|
|
csv_path = Path(test_vault_dir) / "notes.csv"
|
|
csv_path.write_text("a,b\n1,2\n", encoding="utf-8")
|
|
rev = self._revision(client, "notes.csv")
|
|
ok = client.put(
|
|
f"/api/file/{VAULT}/csv/save",
|
|
params={"path": "notes.csv"},
|
|
json={"cells": {"A1": "x"}, "if_match": rev},
|
|
)
|
|
assert ok.status_code == 200
|
|
assert ok.json()["revision"] != rev
|
|
stale = client.put(
|
|
f"/api/file/{VAULT}/csv/save",
|
|
params={"path": "notes.csv"},
|
|
json={"cells": {"A1": "y"}, "if_match": rev},
|
|
)
|
|
assert stale.status_code == 409
|
|
assert csv_path.read_text(encoding="utf-8").startswith("x")
|