"""Display / edit / download for .xlsx files (viewer + PUT xlsx/save). Covers #152 (affichage / édition) and #153 P0 : A1 alerte de fidélité avant écriture, A2 écriture atomique, A3 verrou par fichier, A4 neutralisation de l'injection de formule. """ from __future__ import annotations import zipfile from pathlib import Path import pytest openpyxl = pytest.importorskip("openpyxl") VAULT = "TestVault" @pytest.fixture def xlsx_file(test_vault_dir: str) -> str: from openpyxl import Workbook path = Path(test_vault_dir) / "budget.xlsx" wb = Workbook() ws = wb.active ws.title = "Budget" ws["A1"] = "Poste" ws["B1"] = 100 ws["A2"] = "Total" ws["B2"] = "=B1*2" notes = wb.create_sheet("Notes") notes["A1"] = "hello" wb.save(path) return str(path) def _add_lossy_parts(path: Path, parts: dict[str, bytes]) -> None: """Re-pack *path* with extra OPC parts openpyxl cannot write back.""" with zipfile.ZipFile(path) as zf: items = {n: zf.read(n) for n in zf.namelist()} items.update(parts) with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf: for name, blob in items.items(): zf.writestr(name, blob) @pytest.fixture def lossy_xlsx(test_vault_dir: str) -> str: """Workbook with a slicer + a formula carrying its cached result.""" from openpyxl import Workbook path = Path(test_vault_dir) / "risky.xlsx" wb = Workbook() ws = wb.active ws.title = "Data" ws["A1"] = 3 ws["A2"] = "=A1*3" wb.save(path) # …… : openpyxl keeps the formula, drops the cached result. with zipfile.ZipFile(path) as zf: items = {n: zf.read(n) for n in zf.namelist()} sheet = next(n for n in items if n.startswith("xl/worksheets/sheet")) xml = items[sheet].decode("utf-8").replace( "A1*3", "A1*39" ) items[sheet] = xml.encode("utf-8") items["xl/slicers/slicer1.xml"] = b"" with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf: for name, blob in items.items(): zf.writestr(name, blob) return str(path) # ── Display ─────────────────────────────────────────────────────────────── class TestXlsxDisplay: def test_renders_all_sheets(self, client, xlsx_file): resp = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"}) assert resp.status_code == 200 data = resp.json() assert data["is_xlsx"] is True assert data["unsupported"] is False assert [s["name"] for s in data["xlsx_sheets"]] == ["Budget", "Notes"] first = data["xlsx_sheets"][0]["html"] assert "Poste" in first assert 'data-cell="B1"' in first assert "=B1*2" in first # formula kept as text (data_only=False) assert 'data-cell="A1"' in data["xlsx_sheets"][1]["html"] def test_corrupt_xlsx_returns_500(self, client, test_vault_dir): bad = Path(test_vault_dir) / "corrupt.xlsx" bad.write_bytes(b"this is not a zip archive") resp = client.get(f"/api/file/{VAULT}", params={"path": "corrupt.xlsx"}) assert resp.status_code == 500 # ── Index parity (tree visibility) ──────────────────────────────────────── class TestXlsxIndexing: def test_xlsx_in_supported_extensions(self): from backend.indexer import SUPPORTED_EXTENSIONS assert ".xlsx" in SUPPORTED_EXTENSIONS def test_xlsx_indexes_sheet_names_and_headers(self, test_vault_dir, xlsx_file): """#153 A5 — a workbook is searchable by its cell values.""" from backend.indexer import _index_single_file_sync info = _index_single_file_sync(VAULT, test_vault_dir, xlsx_file) assert info is not None assert info["extension"] == ".xlsx" # Sheet names + header rows reach TF-IDF (was metadata-only before A5). assert "Budget" in info["content"] assert "Poste" in info["content"] assert info["content_preview"] assert info["title"] # filename-derived title class TestXlsxCachedValues: """#153 A12 — show what Excel last computed next to each formula.""" def test_cached_result_is_shown_beside_the_formula(self, client, lossy_xlsx): from backend.xlsx_reader import render_sheets html = render_sheets(Path(lossy_xlsx))[0]["html"] # A2 is "=A1*3" with 9 in the fixture. assert 'data-cell="A2"' in html assert "=A1*3" in html assert "xlsx-cached" in html assert ">9<" in html # the cached result Excel computed def test_no_shadow_when_no_formula_carries_a_result(self, client, xlsx_file): from backend.xlsx_reader import render_sheets html = render_sheets(Path(xlsx_file))[0]["html"] assert "xlsx-cached" not in html # budget.xlsx has no at all def test_plain_cells_are_never_duplicated(self, client, lossy_xlsx): from backend.xlsx_reader import render_sheets html = render_sheets(Path(lossy_xlsx))[0]["html"] # A1 is the literal 3: the two reads agree, so only one value shows. assert 'data-cell="A1"' in html assert html.count("xlsx-cached") == 1 # only the formula cell class TestXlsxValueCoercion: """#153 A10 — a typed value comes back with the type Excel would infer.""" def _write(self, client, xlsx_file, ref, value): return client.put( f"/api/file/{VAULT}/xlsx/save", params={"path": "budget.xlsx"}, json={"sheet": "Budget", "cells": {ref: value}, "force": True}, ) def test_number_and_bool_are_stored_as_typed(self, client, xlsx_file): resp = self._write( client, xlsx_file, "D1", "42" ) assert resp.status_code == 200 resp = self._write(client, xlsx_file, "D2", "VRAI") assert resp.status_code == 200 resp = self._write(client, xlsx_file, "D3", "12/03/2026") assert resp.status_code == 200 from openpyxl import load_workbook wb = load_workbook(xlsx_file) ws = wb["Budget"] assert ws["D1"].value == 42 and isinstance(ws["D1"].value, int) assert ws["D2"].value is True assert ws["D3"].value.year == 2026 and ws["D3"].value.month == 3 assert ws["D3"].value.day == 12 # FR day-first, not 3 December wb.close() def test_day_first_date_is_not_read_as_us(self, client, xlsx_file): """'01/02/2026' is 1 February in French, not 2 January.""" assert self._write(client, xlsx_file, "E1", "01/02/2026").status_code == 200 from openpyxl import load_workbook wb = load_workbook(xlsx_file) d = wb["Budget"]["E1"].value wb.close() assert (d.month, d.day) == (2, 1) def test_ambiguous_text_is_left_alone(self, client, xlsx_file): """A version string or a partial date stays text, never a date.""" assert self._write(client, xlsx_file, "F1", "3.14.2").status_code == 200 assert self._write(client, xlsx_file, "F2", "Ref 12/34").status_code == 200 assert self._write(client, xlsx_file, "F3", "12/2026").status_code == 200 from openpyxl import load_workbook wb = load_workbook(xlsx_file) ws = wb["Budget"] assert ws["F1"].value == "3.14.2" assert ws["F2"].value == "Ref 12/34" assert ws["F3"].value == "12/2026" wb.close() def test_plain_integer_text_becomes_a_number(self, client, xlsx_file): """Typing a bare number yields a number, as it did before #153 A10.""" assert self._write(client, xlsx_file, "H1", "75001").status_code == 200 from openpyxl import load_workbook wb = load_workbook(xlsx_file) assert wb["Budget"]["H1"].value == 75001 wb.close() def test_formula_looking_date_stays_text(self, client, xlsx_file): """A formula is never mistaken for a date (BUG-088 must not regress).""" assert self._write(client, xlsx_file, "G1", "=12/03/2026").status_code == 200 from openpyxl import load_workbook wb = load_workbook(xlsx_file) c = wb["Budget"]["G1"] wb.close() assert c.data_type == "s" assert c.value == "=12/03/2026" class TestXlsxSearchable: """#153 A5 — a keyword living in a CELL must make the file findable.""" def test_search_finds_a_word_stored_in_a_cell(self, test_vault_dir): """A word typed in a CELL must make the workbook findable. Deliberately hermetic: it drives the indexer and the inverted index directly instead of going through the HTTP reload, because both are process-wide singletons that other test modules mutate (some reload the module outright), which would make this test order-dependent. """ from openpyxl import Workbook import backend.indexer as indexer import backend.search as search_mod path = Path(test_vault_dir) / "fournisseurs.xlsx" wb = Workbook() ws = wb.active ws.title = "Contacts" ws["A1"] = "Fournisseur" ws["A2"] = "Menuiserie Beaulieu" wb.save(path) # Index exactly this vault, from scratch. indexer.index[VAULT] = { "files": [indexer._index_single_file_sync(VAULT, test_vault_dir, str(path))], "tags": {}, "paths": [], "config": {}, } entries = [f for f in indexer.index[VAULT]["files"] if f] assert entries, "le tableur n'a pas ete indexe" assert "Menuiserie" in entries[0]["content"], ( f"contenu indexe : {entries[0]['content'][:80]!r}" ) search_mod.init_inverted_index() hits = [r["path"] for r in search_mod.search("Menuiserie", "all")] assert "fournisseurs.xlsx" in hits def test_indexable_text_is_capped(self, tmp_path): """A data dump must not flood the index.""" from openpyxl import Workbook from backend.xlsx_reader import MAX_INDEX_CHARS, extract_indexable_text path = tmp_path / "huge.xlsx" wb = Workbook() ws = wb.active ws.title = "Big" for r in range(1, 400): ws.cell(row=r, column=1, value=f"ligne {r} " + "x" * 60) wb.save(path) text = extract_indexable_text(path) assert len(text) <= MAX_INDEX_CHARS assert "Big" in text # the sheet name survives def test_corrupt_workbook_indexes_as_empty_not_a_crash(self, tmp_path): from backend.xlsx_reader import extract_indexable_text path = tmp_path / "broken.xlsx" path.write_bytes(b"not a zip at all") assert extract_indexable_text(path) == "" def test_blank_rows_are_skipped(self, tmp_path): from openpyxl import Workbook from backend.xlsx_reader import extract_indexable_text path = tmp_path / "sparse.xlsx" wb = Workbook() ws = wb.active ws.title = "S" ws["A1"] = "Alpha" ws["A50"] = "Omega" # beyond the indexed prefix -> ignored on purpose wb.save(path) text = extract_indexable_text(path) assert "Alpha" in text assert "Omega" not in text assert "\t\t" not in text # no run of empty columns # ── Edit ────────────────────────────────────────────────────────────────── class TestXlsxSave: def _save(self, client, body, path="budget.xlsx"): return client.put( f"/api/file/{VAULT}/xlsx/save", params={"path": path}, json=body, ) def test_save_updates_cell_with_number_coercion(self, client, xlsx_file): resp = self._save(client, {"sheet": "Budget", "cells": {"B1": "250"}}) assert resp.status_code == 200 assert resp.json()["status"] == "ok" from openpyxl import load_workbook wb = load_workbook(xlsx_file) assert wb["Budget"]["B1"].value == 250 # int, not "250" def test_save_leaves_other_sheets_and_formulas(self, client, xlsx_file): resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "Titre", "B1": 42}}) assert resp.status_code == 200 from openpyxl import load_workbook wb = load_workbook(xlsx_file) assert wb["Budget"]["B2"].value == "=B1*2" assert wb["Notes"]["A1"].value == "hello" def test_save_empty_string_clears_cell(self, client, xlsx_file): resp = self._save(client, {"sheet": "Budget", "cells": {"A1": ""}}) assert resp.status_code == 200 from openpyxl import load_workbook assert load_workbook(xlsx_file)["Budget"]["A1"].value is None def test_save_creates_backup(self, client, xlsx_file): from backend.services.backups import get_backup_dir resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "backup-me"}}) assert resp.status_code == 200 backup_dir = Path(get_backup_dir(VAULT, "budget.xlsx")) assert backup_dir.is_dir() assert list(backup_dir.glob("*.bak")) def test_unknown_sheet_400(self, client, xlsx_file): resp = self._save(client, {"sheet": "Nope", "cells": {"A1": "x"}}) assert resp.status_code == 400 def test_invalid_cell_ref_400(self, client, xlsx_file): resp = self._save(client, {"sheet": "Budget", "cells": {"A-1": "x"}}) assert resp.status_code == 400 def test_wrong_extension_400(self, client, test_vault_dir): (Path(test_vault_dir) / "note.md").write_text("# hi\n", encoding="utf-8") resp = self._save(client, {"sheet": "Sheet", "cells": {"A1": "x"}}, path="note.md") assert resp.status_code == 400 def test_missing_file_404(self, client): resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "x"}}, path="absent.xlsx") assert resp.status_code == 404 def test_too_many_cells_400(self, client, xlsx_file): cells = {f"A{i}": i for i in range(1, 502)} resp = self._save(client, {"sheet": "Budget", "cells": cells}) assert resp.status_code == 400 def test_nested_value_400(self, client, xlsx_file): resp = self._save(client, {"sheet": "Budget", "cells": {"A1": {"nested": 1}}}) assert resp.status_code == 400 def test_missing_sheet_field_400(self, client, xlsx_file): resp = self._save(client, {"cells": {"A1": "x"}}) assert resp.status_code == 400 def test_non_boolean_flag_400(self, client, xlsx_file): for flag in ("force", "allow_formula"): resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "x"}, flag: "yes"}) assert resp.status_code == 400, flag # ── #153 A1 — lossy-write guard ────────────────────────────────────────── class TestXlsxLossyGuard: def _save(self, client, body, path): return client.put( f"/api/file/{VAULT}/xlsx/save", params={"path": path}, json=body, ) def test_read_reports_lossy_features(self, client, lossy_xlsx): data = client.get(f"/api/file/{VAULT}", params={"path": "risky.xlsx"}).json() assert data["is_xlsx"] is True assert "slicers" in data["xlsx_lossy_features"] assert "cached_values" in data["xlsx_lossy_features"] def test_read_reports_nothing_for_a_plain_workbook(self, client, xlsx_file): data = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"}).json() # B2 holds "=B1*2" but openpyxl wrote no cached for it. assert data["xlsx_lossy_features"] == [] def test_save_refuses_without_force(self, client, lossy_xlsx): resp = self._save(client, {"sheet": "Data", "cells": {"B1": "hello"}}, "risky.xlsx") assert resp.status_code == 409 body = resp.json() assert body["code"] == "xlsx_lossy_content" assert "slicers" in body["details"]["features"] def test_refused_save_leaves_the_file_untouched(self, client, lossy_xlsx): before = Path(lossy_xlsx).read_bytes() self._save(client, {"sheet": "Data", "cells": {"B1": "hello"}}, "risky.xlsx") assert Path(lossy_xlsx).read_bytes() == before def test_save_with_force_succeeds(self, client, lossy_xlsx): resp = self._save( client, {"sheet": "Data", "cells": {"B1": "hello"}, "force": True}, "risky.xlsx", ) assert resp.status_code == 200 assert openpyxl.load_workbook(lossy_xlsx)["Data"]["B1"].value == "hello" def test_inspect_flags_every_known_family(self, lossy_xlsx): from backend.xlsx_reader import LOSSY_PARTS, inspect_workbook for key, prefixes in LOSSY_PARTS.items(): _add_lossy_parts( Path(lossy_xlsx), {f"{prefixes[0]}probe.xml": b"" for _ in [0]}, ) assert key in inspect_workbook(Path(lossy_xlsx)), key def test_inspect_is_quiet_on_a_corrupt_archive(self, test_vault_dir): from backend.xlsx_reader import inspect_workbook bad = Path(test_vault_dir) / "broken.xlsx" bad.write_bytes(b"not a zip at all") assert inspect_workbook(bad) == [] # ── #153 A2 — atomic write ─────────────────────────────────────────────── class TestXlsxAtomicWrite: def test_failed_save_keeps_the_original(self, client, xlsx_file, monkeypatch): from openpyxl.workbook.workbook import Workbook before = Path(xlsx_file).read_bytes() def boom(self, *args, **kwargs): raise OSError("disk full") monkeypatch.setattr(Workbook, "save", boom) # TestClient re-raises the server exception (in production: 500). with pytest.raises(OSError): client.put( f"/api/file/{VAULT}/xlsx/save", params={"path": "budget.xlsx"}, json={"sheet": "Budget", "cells": {"A1": "perdu"}}, ) # The workbook on disk is byte-identical : the write never reached it. assert Path(xlsx_file).read_bytes() == before # No temporary file left behind in the vault. assert list(Path(xlsx_file).parent.glob("*.tmp")) == [] def test_no_tmp_left_after_a_successful_save(self, client, xlsx_file): client.put( f"/api/file/{VAULT}/xlsx/save", params={"path": "budget.xlsx"}, json={"sheet": "Budget", "cells": {"A1": "ok"}}, ) assert list(Path(xlsx_file).parent.glob("*.tmp")) == [] # ── #153 A3 — per-file write lock ──────────────────────────────────────── class TestXlsxWriteLock: def test_concurrent_write_returns_409(self, client, xlsx_file): from backend.services import mutations key = str(Path(xlsx_file).resolve()) with mutations._xlsx_write_lock(key): resp = client.put( f"/api/file/{VAULT}/xlsx/save", params={"path": "budget.xlsx"}, json={"sheet": "Budget", "cells": {"A1": "concurrent"}}, ) assert resp.status_code == 409 assert resp.json()["code"] == "conflict" # The blocked call wrote nothing. assert openpyxl.load_workbook(xlsx_file)["Budget"]["A1"].value == "Poste" def test_lock_is_released_after_a_normal_save(self, client, xlsx_file): from backend.services import mutations key = str(Path(xlsx_file).resolve()) client.put( f"/api/file/{VAULT}/xlsx/save", params={"path": "budget.xlsx"}, json={"sheet": "Budget", "cells": {"A1": "premier"}}, ) # The lock must be free again once the request returned. with mutations._xlsx_write_lock(key): pass assert openpyxl.load_workbook(xlsx_file)["Budget"]["A1"].value == "premier" @pytest.fixture def wide_xlsx(test_vault_dir: str) -> str: """Workbook whose sheet exceeds BOTH render caps (501 rows x 45 cols). Sparse on purpose: a cell in A501 and one in AS1 are enough for openpyxl to declare those dimensions, without writing 20 000 cells to disk. """ from openpyxl import Workbook path = Path(test_vault_dir) / "grand.xlsx" wb = Workbook() ws = wb.active ws.title = "Data" ws["A1"] = "tête" ws["A501"] = "dernière ligne" ws["AS1"] = "colonne 45" wb.save(path) return str(path) @pytest.fixture def edge_xlsx(test_vault_dir: str) -> str: """Sheet exactly on the caps (500 rows x 40 cols) — must NOT be truncated.""" from openpyxl import Workbook path = Path(test_vault_dir) / "limite.xlsx" wb = Workbook() ws = wb.active ws.title = "Data" ws["A1"] = "bord" ws["A500"] = "ligne 500" ws["AN1"] = "colonne 40" wb.save(path) return str(path) # ── #153 A8 — silent truncation made visible ───────────────────────────── class TestXlsxTruncationNotice: """A sheet bigger than the caps must SAY so instead of looking complete.""" def test_render_reports_the_real_dimensions(self, client, wide_xlsx): resp = client.get(f"/api/file/{VAULT}", params={"path": "grand.xlsx"}) sheet = resp.json()["xlsx_sheets"][0] assert (sheet["total_rows"], sheet["total_cols"]) == (501, 45) assert sheet["truncated"] is True # The caps are the coverage the banner announces — `rows`/`cols` are # post-trim and would understate it on a sparse sheet. assert (sheet["max_rows"], sheet["max_cols"]) == (500, 40) assert (sheet["rows"], sheet["cols"]) == (1, 1) # only 3 filled cells def test_a_sheet_on_the_caps_is_not_flagged(self, client, edge_xlsx): """Boundary: 500x40 is exactly what the renderer supports.""" resp = client.get(f"/api/file/{VAULT}", params={"path": "limite.xlsx"}) sheet = resp.json()["xlsx_sheets"][0] assert sheet["truncated"] is False assert (sheet["total_rows"], sheet["total_cols"]) == (500, 40) def test_a_normal_sheet_is_not_flagged(self, client, xlsx_file): resp = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"}) assert all(not s["truncated"] for s in resp.json()["xlsx_sheets"]) def test_blank_tail_is_not_reported_as_truncation(self, client, test_vault_dir): """A sheet with empty rows below its data fits in the caps.""" from openpyxl import Workbook path = Path(test_vault_dir) / "blanc.xlsx" wb = Workbook() ws = wb.active ws.title = "Data" ws["A1"] = "seule ligne" ws["A300"] = None # formatted-but-empty row inside the caps wb.save(path) resp = client.get(f"/api/file/{VAULT}", params={"path": "blanc.xlsx"}) sheet = resp.json()["xlsx_sheets"][0] assert sheet["truncated"] is False assert sheet["rows"] == 1 # trailing blanks dropped by _trim # ── #153 A9 — lazy per-sheet loading ───────────────────────────────────── class TestXlsxSheetWindow: """GET /api/file/{vault}/xlsx/sheet — one window of one sheet.""" def _get(self, client, path="budget.xlsx", **params): return client.get( f"/api/file/{VAULT}/xlsx/sheet", params={"path": path, **params} ) def test_window_returns_rows_and_totals(self, client, xlsx_file): resp = self._get(client, sheet="Budget", offset=0, limit=10) assert resp.status_code == 200 data = resp.json() assert data["sheet"] == "Budget" assert (data["offset"], data["limit"]) == (0, 10) assert data["rows"] == 2 and data["cols"] == 2 assert data["total_rows"] == 2 and data["truncated"] is False assert (data["max_rows"], data["max_cols"]) == (500, 40) assert data["has_more"] is False assert 'data-cell="A1"' in data["html"] def test_window_keeps_the_real_a1_coordinates(self, client, xlsx_file): """A window must be indistinguishable from a full render: the A1 references and the row numbers have to be the sheet's, not the window's, or an edit would land on the wrong cell. The CONTENT matters as much as the label — row 2 of "Budget" is "Total", not "Poste".""" data = self._get(client, sheet="Budget", offset=1, limit=1).json() assert data["rows"] == 1 assert 'data-cell="A2"' in data["html"] assert 'data-cell="A1"' not in data["html"] assert "2" in data["html"] assert "Total" in data["html"] and "Poste" not in data["html"] def test_window_offsets_walk_the_whole_sheet(self, client, wide_xlsx): first = self._get(client, path="grand.xlsx", sheet="Data", offset=0, limit=10).json() last = self._get(client, path="grand.xlsx", sheet="Data", offset=500, limit=10).json() assert first["truncated"] is True and first["has_more"] is True assert 'data-cell="A501"' in last["html"] # the row the caps used to hide assert "dernière ligne" in last["html"] assert "dernière ligne" not in first["html"] assert last["rows"] == 1 and last["has_more"] is False def test_offset_past_the_end_is_empty_not_an_error(self, client, xlsx_file): resp = self._get(client, sheet="Budget", offset=9999, limit=10) assert resp.status_code == 200 data = resp.json() assert data["rows"] == 0 assert data["has_more"] is False def test_cached_result_survives_the_window(self, client, lossy_xlsx): """A12 must not be lost on the lazy path.""" data = self._get(client, path="risky.xlsx", sheet="Data", offset=0, limit=10).json() assert "xlsx-cached" in data["html"] def test_unknown_sheet_is_404(self, client, xlsx_file): resp = self._get(client, sheet="Nope") assert resp.status_code == 404 assert "Feuille introuvable" in resp.json()["detail"] def test_missing_file_is_404(self, client, test_vault_dir): assert self._get(client, path="absent.xlsx", sheet="Budget").status_code == 404 def test_non_xlsx_file_is_415(self, client, test_vault_dir): (Path(test_vault_dir) / "note.md").write_text("# hi", encoding="utf-8") resp = self._get(client, path="note.md", sheet="Budget") assert resp.status_code == 415 def test_limit_above_the_server_cap_is_rejected(self, client, xlsx_file): """The cap is a contract, not a silent truncation of the request.""" assert self._get(client, sheet="Budget", limit=100_000).status_code == 422 def test_reader_clamps_a_hostile_limit(self, xlsx_file): """Belt and braces: the reader caps too, whoever calls it.""" from backend.xlsx_reader import MAX_WINDOW_ROWS, read_sheet_window window = read_sheet_window(Path(xlsx_file), "Budget", offset=0, limit=10**9) assert window["limit"] == MAX_WINDOW_ROWS def test_reader_rejects_a_negative_offset(self, xlsx_file): from backend.xlsx_reader import read_sheet_window window = read_sheet_window(Path(xlsx_file), "Budget", offset=-5, limit=10) assert window["offset"] == 0 # ── #153 A4 — formula injection ────────────────────────────────────────── class TestXlsxFormulaGuard: def _save(self, client, cells, **extra): return client.put( f"/api/file/{VAULT}/xlsx/save", params={"path": "budget.xlsx"}, json={"sheet": "Budget", "cells": cells, **extra}, ) def test_formula_like_value_is_stored_as_text(self, client, xlsx_file): resp = self._save(client, {"A3": "=cmd|'/c calc'!A1"}) assert resp.status_code == 200 cell = openpyxl.load_workbook(xlsx_file)["Budget"]["A3"] assert cell.value == "=cmd|'/c calc'!A1" assert cell.data_type == "s" # pas de dans l'archive def test_at_prefix_is_stored_as_text(self, client, xlsx_file): self._save(client, {"A4": "@SUM(A1:A2)"}) assert openpyxl.load_workbook(xlsx_file)["Budget"]["A4"].data_type == "s" def test_allow_formula_keeps_a_real_formula(self, client, xlsx_file): resp = self._save(client, {"A3": "=B1+5"}, allow_formula=True) assert resp.status_code == 200 assert openpyxl.load_workbook(xlsx_file)["Budget"]["A3"].data_type == "f" def test_numbers_are_unaffected(self, client, xlsx_file): self._save(client, {"B3": "42", "B4": "-3.5"}) ws = openpyxl.load_workbook(xlsx_file)["Budget"] assert ws["B3"].value == 42 and isinstance(ws["B3"].value, int) assert ws["B4"].value == -3.5