744 lines
30 KiB
Python
744 lines
30 KiB
Python
"""Display / edit / download for .xlsx files (viewer + PUT xlsx/save).
|
|
|
|
Covers #152 (affichage / édition) and #153 P0 : A1 alerte de fidélité avant
|
|
écriture, A2 écriture atomique, A3 verrou par fichier, A4 neutralisation de
|
|
l'injection de formule.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import zipfile
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
openpyxl = pytest.importorskip("openpyxl")
|
|
|
|
VAULT = "TestVault"
|
|
|
|
|
|
@pytest.fixture
|
|
def xlsx_file(test_vault_dir: str) -> str:
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "budget.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Budget"
|
|
ws["A1"] = "Poste"
|
|
ws["B1"] = 100
|
|
ws["A2"] = "Total"
|
|
ws["B2"] = "=B1*2"
|
|
notes = wb.create_sheet("Notes")
|
|
notes["A1"] = "hello"
|
|
wb.save(path)
|
|
return str(path)
|
|
|
|
|
|
def _add_lossy_parts(path: Path, parts: dict[str, bytes]) -> None:
|
|
"""Re-pack *path* with extra OPC parts openpyxl cannot write back."""
|
|
with zipfile.ZipFile(path) as zf:
|
|
items = {n: zf.read(n) for n in zf.namelist()}
|
|
items.update(parts)
|
|
with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
for name, blob in items.items():
|
|
zf.writestr(name, blob)
|
|
|
|
|
|
@pytest.fixture
|
|
def lossy_xlsx(test_vault_dir: str) -> str:
|
|
"""Workbook with a slicer + a formula carrying its cached result."""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "risky.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Data"
|
|
ws["A1"] = 3
|
|
ws["A2"] = "=A1*3"
|
|
wb.save(path)
|
|
# <f>…</f><v>…</v> : openpyxl keeps the formula, drops the cached result.
|
|
with zipfile.ZipFile(path) as zf:
|
|
items = {n: zf.read(n) for n in zf.namelist()}
|
|
sheet = next(n for n in items if n.startswith("xl/worksheets/sheet"))
|
|
xml = items[sheet].decode("utf-8").replace(
|
|
"<f>A1*3</f>", "<f>A1*3</f><v>9</v>"
|
|
)
|
|
items[sheet] = xml.encode("utf-8")
|
|
items["xl/slicers/slicer1.xml"] = b"<slicer/>"
|
|
with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
for name, blob in items.items():
|
|
zf.writestr(name, blob)
|
|
return str(path)
|
|
|
|
|
|
# ── Display ───────────────────────────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxDisplay:
|
|
def test_renders_all_sheets(self, client, xlsx_file):
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"})
|
|
assert resp.status_code == 200
|
|
data = resp.json()
|
|
assert data["is_xlsx"] is True
|
|
assert data["unsupported"] is False
|
|
assert [s["name"] for s in data["xlsx_sheets"]] == ["Budget", "Notes"]
|
|
|
|
first = data["xlsx_sheets"][0]["html"]
|
|
assert "Poste" in first
|
|
assert 'data-cell="B1"' in first
|
|
assert "=B1*2" in first # formula kept as text (data_only=False)
|
|
assert 'data-cell="A1"' in data["xlsx_sheets"][1]["html"]
|
|
|
|
def test_corrupt_xlsx_returns_500(self, client, test_vault_dir):
|
|
bad = Path(test_vault_dir) / "corrupt.xlsx"
|
|
bad.write_bytes(b"this is not a zip archive")
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "corrupt.xlsx"})
|
|
assert resp.status_code == 500
|
|
|
|
def test_empty_sheet_renders_an_editable_blank_grid(self, client, test_vault_dir):
|
|
"""BUG-094 — a blank sheet used to render as a bare « Feuille vide »
|
|
paragraph with no cell, so a freshly added sheet could not be filled
|
|
and had no way to insert a row/column. It now exposes a small editable
|
|
grid with real A1 coordinates."""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "blank.xlsx"
|
|
wb = Workbook()
|
|
wb.active.title = "Vide"
|
|
wb.create_sheet("Vide2")
|
|
wb.save(str(path))
|
|
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "blank.xlsx"})
|
|
assert resp.status_code == 200
|
|
vide = next(s for s in resp.json()["xlsx_sheets"] if s["name"] == "Vide")
|
|
assert "Feuille vide" not in vide["html"]
|
|
assert 'data-cell="A1"' in vide["html"]
|
|
assert 'data-cell="H20"' in vide["html"] # last cell of the blank grid
|
|
assert vide["rows"] == 20
|
|
assert vide["cols"] == 8
|
|
assert vide["truncated"] is False
|
|
|
|
|
|
# ── Index parity (tree visibility) ────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxIndexing:
|
|
def test_xlsx_in_supported_extensions(self):
|
|
from backend.indexer import SUPPORTED_EXTENSIONS
|
|
|
|
assert ".xlsx" in SUPPORTED_EXTENSIONS
|
|
|
|
def test_xlsx_indexes_sheet_names_and_headers(self, test_vault_dir, xlsx_file):
|
|
"""#153 A5 — a workbook is searchable by its cell values."""
|
|
from backend.indexer import _index_single_file_sync
|
|
|
|
info = _index_single_file_sync(VAULT, test_vault_dir, xlsx_file)
|
|
assert info is not None
|
|
assert info["extension"] == ".xlsx"
|
|
# Sheet names + header rows reach TF-IDF (was metadata-only before A5).
|
|
assert "Budget" in info["content"]
|
|
assert "Poste" in info["content"]
|
|
assert info["content_preview"]
|
|
assert info["title"] # filename-derived title
|
|
|
|
|
|
class TestXlsxCachedValues:
|
|
"""#153 A12 — show what Excel last computed next to each formula."""
|
|
|
|
def test_cached_result_is_shown_beside_the_formula(self, client, lossy_xlsx):
|
|
from backend.xlsx_reader import render_sheets
|
|
|
|
html = render_sheets(Path(lossy_xlsx))[0]["html"]
|
|
# A2 is "=A1*3" with <v>9</v> in the fixture.
|
|
assert 'data-cell="A2"' in html
|
|
assert "=A1*3" in html
|
|
assert "xlsx-cached" in html
|
|
assert ">9<" in html # the cached result Excel computed
|
|
|
|
def test_no_shadow_when_no_formula_carries_a_result(self, client, xlsx_file):
|
|
from backend.xlsx_reader import render_sheets
|
|
|
|
html = render_sheets(Path(xlsx_file))[0]["html"]
|
|
assert "xlsx-cached" not in html # budget.xlsx has no <v> at all
|
|
|
|
def test_plain_cells_are_never_duplicated(self, client, lossy_xlsx):
|
|
from backend.xlsx_reader import render_sheets
|
|
|
|
html = render_sheets(Path(lossy_xlsx))[0]["html"]
|
|
# A1 is the literal 3: the two reads agree, so only one value shows.
|
|
assert 'data-cell="A1"' in html
|
|
assert html.count("xlsx-cached") == 1 # only the formula cell
|
|
|
|
|
|
class TestXlsxValueCoercion:
|
|
"""#153 A10 — a typed value comes back with the type Excel would infer."""
|
|
|
|
def _write(self, client, xlsx_file, ref, value):
|
|
return client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {ref: value}, "force": True},
|
|
)
|
|
|
|
def test_number_and_bool_are_stored_as_typed(self, client, xlsx_file):
|
|
resp = self._write(
|
|
client, xlsx_file, "D1", "42"
|
|
)
|
|
assert resp.status_code == 200
|
|
resp = self._write(client, xlsx_file, "D2", "VRAI")
|
|
assert resp.status_code == 200
|
|
resp = self._write(client, xlsx_file, "D3", "12/03/2026")
|
|
assert resp.status_code == 200
|
|
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
ws = wb["Budget"]
|
|
assert ws["D1"].value == 42 and isinstance(ws["D1"].value, int)
|
|
assert ws["D2"].value is True
|
|
assert ws["D3"].value.year == 2026 and ws["D3"].value.month == 3
|
|
assert ws["D3"].value.day == 12 # FR day-first, not 3 December
|
|
wb.close()
|
|
|
|
def test_day_first_date_is_not_read_as_us(self, client, xlsx_file):
|
|
"""'01/02/2026' is 1 February in French, not 2 January."""
|
|
assert self._write(client, xlsx_file, "E1", "01/02/2026").status_code == 200
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
d = wb["Budget"]["E1"].value
|
|
wb.close()
|
|
assert (d.month, d.day) == (2, 1)
|
|
|
|
def test_ambiguous_text_is_left_alone(self, client, xlsx_file):
|
|
"""A version string or a partial date stays text, never a date."""
|
|
assert self._write(client, xlsx_file, "F1", "3.14.2").status_code == 200
|
|
assert self._write(client, xlsx_file, "F2", "Ref 12/34").status_code == 200
|
|
assert self._write(client, xlsx_file, "F3", "12/2026").status_code == 200
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
ws = wb["Budget"]
|
|
assert ws["F1"].value == "3.14.2"
|
|
assert ws["F2"].value == "Ref 12/34"
|
|
assert ws["F3"].value == "12/2026"
|
|
wb.close()
|
|
|
|
def test_plain_integer_text_becomes_a_number(self, client, xlsx_file):
|
|
"""Typing a bare number yields a number, as it did before #153 A10."""
|
|
assert self._write(client, xlsx_file, "H1", "75001").status_code == 200
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
assert wb["Budget"]["H1"].value == 75001
|
|
wb.close()
|
|
|
|
def test_formula_looking_date_stays_text(self, client, xlsx_file):
|
|
"""A formula is never mistaken for a date (BUG-088 must not regress)."""
|
|
assert self._write(client, xlsx_file, "G1", "=12/03/2026").status_code == 200
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
c = wb["Budget"]["G1"]
|
|
wb.close()
|
|
assert c.data_type == "s"
|
|
assert c.value == "=12/03/2026"
|
|
|
|
|
|
class TestXlsxSearchable:
|
|
"""#153 A5 — a keyword living in a CELL must make the file findable."""
|
|
|
|
def test_search_finds_a_word_stored_in_a_cell(self, test_vault_dir):
|
|
"""A word typed in a CELL must make the workbook findable.
|
|
|
|
Deliberately hermetic: it drives the indexer and the inverted index
|
|
directly instead of going through the HTTP reload, because both are
|
|
process-wide singletons that other test modules mutate (some reload the
|
|
module outright), which would make this test order-dependent.
|
|
"""
|
|
from openpyxl import Workbook
|
|
|
|
import backend.indexer as indexer
|
|
import backend.search as search_mod
|
|
|
|
path = Path(test_vault_dir) / "fournisseurs.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Contacts"
|
|
ws["A1"] = "Fournisseur"
|
|
ws["A2"] = "Menuiserie Beaulieu"
|
|
wb.save(path)
|
|
|
|
# Index exactly this vault, from scratch.
|
|
indexer.index[VAULT] = {
|
|
"files": [indexer._index_single_file_sync(VAULT, test_vault_dir, str(path))],
|
|
"tags": {},
|
|
"paths": [],
|
|
"config": {},
|
|
}
|
|
entries = [f for f in indexer.index[VAULT]["files"] if f]
|
|
assert entries, "le tableur n'a pas ete indexe"
|
|
assert "Menuiserie" in entries[0]["content"], (
|
|
f"contenu indexe : {entries[0]['content'][:80]!r}"
|
|
)
|
|
|
|
search_mod.init_inverted_index()
|
|
hits = [r["path"] for r in search_mod.search("Menuiserie", "all")]
|
|
assert "fournisseurs.xlsx" in hits
|
|
|
|
def test_indexable_text_is_capped(self, tmp_path):
|
|
"""A data dump must not flood the index."""
|
|
from openpyxl import Workbook
|
|
|
|
from backend.xlsx_reader import MAX_INDEX_CHARS, extract_indexable_text
|
|
|
|
path = tmp_path / "huge.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Big"
|
|
for r in range(1, 400):
|
|
ws.cell(row=r, column=1, value=f"ligne {r} " + "x" * 60)
|
|
wb.save(path)
|
|
|
|
text = extract_indexable_text(path)
|
|
assert len(text) <= MAX_INDEX_CHARS
|
|
assert "Big" in text # the sheet name survives
|
|
|
|
def test_corrupt_workbook_indexes_as_empty_not_a_crash(self, tmp_path):
|
|
from backend.xlsx_reader import extract_indexable_text
|
|
|
|
path = tmp_path / "broken.xlsx"
|
|
path.write_bytes(b"not a zip at all")
|
|
assert extract_indexable_text(path) == ""
|
|
|
|
def test_blank_rows_are_skipped(self, tmp_path):
|
|
from openpyxl import Workbook
|
|
|
|
from backend.xlsx_reader import extract_indexable_text
|
|
|
|
path = tmp_path / "sparse.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "S"
|
|
ws["A1"] = "Alpha"
|
|
ws["A50"] = "Omega" # beyond the indexed prefix -> ignored on purpose
|
|
wb.save(path)
|
|
|
|
text = extract_indexable_text(path)
|
|
assert "Alpha" in text
|
|
assert "Omega" not in text
|
|
assert "\t\t" not in text # no run of empty columns
|
|
|
|
|
|
# ── Edit ──────────────────────────────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxSave:
|
|
def _save(self, client, body, path="budget.xlsx"):
|
|
return client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": path},
|
|
json=body,
|
|
)
|
|
|
|
def test_save_updates_cell_with_number_coercion(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"B1": "250"}})
|
|
assert resp.status_code == 200
|
|
assert resp.json()["status"] == "ok"
|
|
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
assert wb["Budget"]["B1"].value == 250 # int, not "250"
|
|
|
|
def test_save_leaves_other_sheets_and_formulas(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "Titre", "B1": 42}})
|
|
assert resp.status_code == 200
|
|
|
|
from openpyxl import load_workbook
|
|
|
|
wb = load_workbook(xlsx_file)
|
|
assert wb["Budget"]["B2"].value == "=B1*2"
|
|
assert wb["Notes"]["A1"].value == "hello"
|
|
|
|
def test_save_empty_string_clears_cell(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": ""}})
|
|
assert resp.status_code == 200
|
|
|
|
from openpyxl import load_workbook
|
|
|
|
assert load_workbook(xlsx_file)["Budget"]["A1"].value is None
|
|
|
|
def test_save_creates_backup(self, client, xlsx_file):
|
|
from backend.services.backups import get_backup_dir
|
|
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "backup-me"}})
|
|
assert resp.status_code == 200
|
|
backup_dir = Path(get_backup_dir(VAULT, "budget.xlsx"))
|
|
assert backup_dir.is_dir()
|
|
assert list(backup_dir.glob("*.bak"))
|
|
|
|
def test_unknown_sheet_400(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Nope", "cells": {"A1": "x"}})
|
|
assert resp.status_code == 400
|
|
|
|
def test_invalid_cell_ref_400(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A-1": "x"}})
|
|
assert resp.status_code == 400
|
|
|
|
def test_wrong_extension_400(self, client, test_vault_dir):
|
|
(Path(test_vault_dir) / "note.md").write_text("# hi\n", encoding="utf-8")
|
|
resp = self._save(client, {"sheet": "Sheet", "cells": {"A1": "x"}}, path="note.md")
|
|
assert resp.status_code == 400
|
|
|
|
def test_missing_file_404(self, client):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "x"}}, path="absent.xlsx")
|
|
assert resp.status_code == 404
|
|
|
|
def test_too_many_cells_400(self, client, xlsx_file):
|
|
cells = {f"A{i}": i for i in range(1, 502)}
|
|
resp = self._save(client, {"sheet": "Budget", "cells": cells})
|
|
assert resp.status_code == 400
|
|
|
|
def test_nested_value_400(self, client, xlsx_file):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": {"nested": 1}}})
|
|
assert resp.status_code == 400
|
|
|
|
def test_missing_sheet_field_400(self, client, xlsx_file):
|
|
resp = self._save(client, {"cells": {"A1": "x"}})
|
|
assert resp.status_code == 400
|
|
|
|
def test_non_boolean_flag_400(self, client, xlsx_file):
|
|
for flag in ("force", "allow_formula"):
|
|
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "x"}, flag: "yes"})
|
|
assert resp.status_code == 400, flag
|
|
|
|
|
|
# ── #153 A1 — lossy-write guard ──────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxLossyGuard:
|
|
def _save(self, client, body, path):
|
|
return client.put(
|
|
f"/api/file/{VAULT}/xlsx/save", params={"path": path}, json=body,
|
|
)
|
|
|
|
def test_read_reports_lossy_features(self, client, lossy_xlsx):
|
|
data = client.get(f"/api/file/{VAULT}", params={"path": "risky.xlsx"}).json()
|
|
assert data["is_xlsx"] is True
|
|
assert "slicers" in data["xlsx_lossy_features"]
|
|
assert "cached_values" in data["xlsx_lossy_features"]
|
|
|
|
def test_read_reports_nothing_for_a_plain_workbook(self, client, xlsx_file):
|
|
data = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"}).json()
|
|
# B2 holds "=B1*2" but openpyxl wrote no cached <v> for it.
|
|
assert data["xlsx_lossy_features"] == []
|
|
|
|
def test_save_refuses_without_force(self, client, lossy_xlsx):
|
|
resp = self._save(client, {"sheet": "Data", "cells": {"B1": "hello"}}, "risky.xlsx")
|
|
assert resp.status_code == 409
|
|
body = resp.json()
|
|
assert body["code"] == "xlsx_lossy_content"
|
|
assert "slicers" in body["details"]["features"]
|
|
|
|
def test_refused_save_leaves_the_file_untouched(self, client, lossy_xlsx):
|
|
before = Path(lossy_xlsx).read_bytes()
|
|
self._save(client, {"sheet": "Data", "cells": {"B1": "hello"}}, "risky.xlsx")
|
|
assert Path(lossy_xlsx).read_bytes() == before
|
|
|
|
def test_save_with_force_succeeds(self, client, lossy_xlsx):
|
|
resp = self._save(
|
|
client,
|
|
{"sheet": "Data", "cells": {"B1": "hello"}, "force": True},
|
|
"risky.xlsx",
|
|
)
|
|
assert resp.status_code == 200
|
|
assert openpyxl.load_workbook(lossy_xlsx)["Data"]["B1"].value == "hello"
|
|
|
|
def test_inspect_flags_every_known_family(self, lossy_xlsx):
|
|
from backend.xlsx_reader import LOSSY_PARTS, inspect_workbook
|
|
|
|
for key, prefixes in LOSSY_PARTS.items():
|
|
_add_lossy_parts(
|
|
Path(lossy_xlsx),
|
|
{f"{prefixes[0]}probe.xml": b"<x/>" for _ in [0]},
|
|
)
|
|
assert key in inspect_workbook(Path(lossy_xlsx)), key
|
|
|
|
def test_inspect_is_quiet_on_a_corrupt_archive(self, test_vault_dir):
|
|
from backend.xlsx_reader import inspect_workbook
|
|
|
|
bad = Path(test_vault_dir) / "broken.xlsx"
|
|
bad.write_bytes(b"not a zip at all")
|
|
assert inspect_workbook(bad) == []
|
|
|
|
|
|
# ── #153 A2 — atomic write ───────────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxAtomicWrite:
|
|
def test_failed_save_keeps_the_original(self, client, xlsx_file, monkeypatch):
|
|
from openpyxl.workbook.workbook import Workbook
|
|
|
|
before = Path(xlsx_file).read_bytes()
|
|
|
|
def boom(self, *args, **kwargs):
|
|
raise OSError("disk full")
|
|
|
|
monkeypatch.setattr(Workbook, "save", boom)
|
|
# TestClient re-raises the server exception (in production: 500).
|
|
with pytest.raises(OSError):
|
|
client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {"A1": "perdu"}},
|
|
)
|
|
# The workbook on disk is byte-identical : the write never reached it.
|
|
assert Path(xlsx_file).read_bytes() == before
|
|
# No temporary file left behind in the vault.
|
|
assert list(Path(xlsx_file).parent.glob("*.tmp")) == []
|
|
|
|
def test_no_tmp_left_after_a_successful_save(self, client, xlsx_file):
|
|
client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {"A1": "ok"}},
|
|
)
|
|
assert list(Path(xlsx_file).parent.glob("*.tmp")) == []
|
|
|
|
|
|
# ── #153 A3 — per-file write lock ────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxWriteLock:
|
|
def test_concurrent_write_returns_409(self, client, xlsx_file):
|
|
from backend.services import mutations
|
|
|
|
key = str(Path(xlsx_file).resolve())
|
|
with mutations._xlsx_write_lock(key):
|
|
resp = client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {"A1": "concurrent"}},
|
|
)
|
|
assert resp.status_code == 409
|
|
assert resp.json()["code"] == "conflict"
|
|
# The blocked call wrote nothing.
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A1"].value == "Poste"
|
|
|
|
def test_lock_is_released_after_a_normal_save(self, client, xlsx_file):
|
|
from backend.services import mutations
|
|
|
|
key = str(Path(xlsx_file).resolve())
|
|
client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": {"A1": "premier"}},
|
|
)
|
|
# The lock must be free again once the request returned.
|
|
with mutations._xlsx_write_lock(key):
|
|
pass
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A1"].value == "premier"
|
|
|
|
|
|
@pytest.fixture
|
|
def wide_xlsx(test_vault_dir: str) -> str:
|
|
"""Workbook whose sheet exceeds BOTH render caps (501 rows x 45 cols).
|
|
|
|
Sparse on purpose: a cell in A501 and one in AS1 are enough for openpyxl
|
|
to declare those dimensions, without writing 20 000 cells to disk.
|
|
"""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "grand.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Data"
|
|
ws["A1"] = "tête"
|
|
ws["A501"] = "dernière ligne"
|
|
ws["AS1"] = "colonne 45"
|
|
wb.save(path)
|
|
return str(path)
|
|
|
|
|
|
@pytest.fixture
|
|
def edge_xlsx(test_vault_dir: str) -> str:
|
|
"""Sheet exactly on the caps (500 rows x 40 cols) — must NOT be truncated."""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "limite.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Data"
|
|
ws["A1"] = "bord"
|
|
ws["A500"] = "ligne 500"
|
|
ws["AN1"] = "colonne 40"
|
|
wb.save(path)
|
|
return str(path)
|
|
|
|
|
|
# ── #153 A8 — silent truncation made visible ─────────────────────────────
|
|
|
|
|
|
class TestXlsxTruncationNotice:
|
|
"""A sheet bigger than the caps must SAY so instead of looking complete."""
|
|
|
|
def test_render_reports_the_real_dimensions(self, client, wide_xlsx):
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "grand.xlsx"})
|
|
sheet = resp.json()["xlsx_sheets"][0]
|
|
assert (sheet["total_rows"], sheet["total_cols"]) == (501, 45)
|
|
assert sheet["truncated"] is True
|
|
# The caps are the coverage the banner announces — `rows`/`cols` are
|
|
# post-trim and would understate it on a sparse sheet.
|
|
assert (sheet["max_rows"], sheet["max_cols"]) == (500, 40)
|
|
assert (sheet["rows"], sheet["cols"]) == (1, 1) # only 3 filled cells
|
|
|
|
def test_a_sheet_on_the_caps_is_not_flagged(self, client, edge_xlsx):
|
|
"""Boundary: 500x40 is exactly what the renderer supports."""
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "limite.xlsx"})
|
|
sheet = resp.json()["xlsx_sheets"][0]
|
|
assert sheet["truncated"] is False
|
|
assert (sheet["total_rows"], sheet["total_cols"]) == (500, 40)
|
|
|
|
def test_a_normal_sheet_is_not_flagged(self, client, xlsx_file):
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"})
|
|
assert all(not s["truncated"] for s in resp.json()["xlsx_sheets"])
|
|
|
|
def test_blank_tail_is_not_reported_as_truncation(self, client, test_vault_dir):
|
|
"""A sheet with empty rows below its data fits in the caps."""
|
|
from openpyxl import Workbook
|
|
|
|
path = Path(test_vault_dir) / "blanc.xlsx"
|
|
wb = Workbook()
|
|
ws = wb.active
|
|
ws.title = "Data"
|
|
ws["A1"] = "seule ligne"
|
|
ws["A300"] = None # formatted-but-empty row inside the caps
|
|
wb.save(path)
|
|
resp = client.get(f"/api/file/{VAULT}", params={"path": "blanc.xlsx"})
|
|
sheet = resp.json()["xlsx_sheets"][0]
|
|
assert sheet["truncated"] is False
|
|
assert sheet["rows"] == 1 # trailing blanks dropped by _trim
|
|
|
|
|
|
# ── #153 A9 — lazy per-sheet loading ─────────────────────────────────────
|
|
|
|
|
|
class TestXlsxSheetWindow:
|
|
"""GET /api/file/{vault}/xlsx/sheet — one window of one sheet."""
|
|
|
|
def _get(self, client, path="budget.xlsx", **params):
|
|
return client.get(
|
|
f"/api/file/{VAULT}/xlsx/sheet", params={"path": path, **params}
|
|
)
|
|
|
|
def test_window_returns_rows_and_totals(self, client, xlsx_file):
|
|
resp = self._get(client, sheet="Budget", offset=0, limit=10)
|
|
assert resp.status_code == 200
|
|
data = resp.json()
|
|
assert data["sheet"] == "Budget"
|
|
assert (data["offset"], data["limit"]) == (0, 10)
|
|
assert data["rows"] == 2 and data["cols"] == 2
|
|
assert data["total_rows"] == 2 and data["truncated"] is False
|
|
assert (data["max_rows"], data["max_cols"]) == (500, 40)
|
|
assert data["has_more"] is False
|
|
assert 'data-cell="A1"' in data["html"]
|
|
|
|
def test_window_keeps_the_real_a1_coordinates(self, client, xlsx_file):
|
|
"""A window must be indistinguishable from a full render: the A1
|
|
references and the row numbers have to be the sheet's, not the
|
|
window's, or an edit would land on the wrong cell. The CONTENT matters
|
|
as much as the label — row 2 of "Budget" is "Total", not "Poste"."""
|
|
data = self._get(client, sheet="Budget", offset=1, limit=1).json()
|
|
assert data["rows"] == 1
|
|
assert 'data-cell="A2"' in data["html"]
|
|
assert 'data-cell="A1"' not in data["html"]
|
|
assert "<th class=\"xlsx-rownum\">2</th>" in data["html"]
|
|
assert "Total" in data["html"] and "Poste" not in data["html"]
|
|
|
|
def test_window_offsets_walk_the_whole_sheet(self, client, wide_xlsx):
|
|
first = self._get(client, path="grand.xlsx", sheet="Data", offset=0, limit=10).json()
|
|
last = self._get(client, path="grand.xlsx", sheet="Data", offset=500, limit=10).json()
|
|
assert first["truncated"] is True and first["has_more"] is True
|
|
assert 'data-cell="A501"' in last["html"] # the row the caps used to hide
|
|
assert "dernière ligne" in last["html"]
|
|
assert "dernière ligne" not in first["html"]
|
|
assert last["rows"] == 1 and last["has_more"] is False
|
|
|
|
def test_offset_past_the_end_is_empty_not_an_error(self, client, xlsx_file):
|
|
resp = self._get(client, sheet="Budget", offset=9999, limit=10)
|
|
assert resp.status_code == 200
|
|
data = resp.json()
|
|
assert data["rows"] == 0
|
|
assert data["has_more"] is False
|
|
|
|
def test_cached_result_survives_the_window(self, client, lossy_xlsx):
|
|
"""A12 must not be lost on the lazy path."""
|
|
data = self._get(client, path="risky.xlsx", sheet="Data", offset=0, limit=10).json()
|
|
assert "xlsx-cached" in data["html"]
|
|
|
|
def test_unknown_sheet_is_404(self, client, xlsx_file):
|
|
resp = self._get(client, sheet="Nope")
|
|
assert resp.status_code == 404
|
|
assert "Feuille introuvable" in resp.json()["detail"]
|
|
|
|
def test_missing_file_is_404(self, client, test_vault_dir):
|
|
assert self._get(client, path="absent.xlsx", sheet="Budget").status_code == 404
|
|
|
|
def test_non_xlsx_file_is_415(self, client, test_vault_dir):
|
|
(Path(test_vault_dir) / "note.md").write_text("# hi", encoding="utf-8")
|
|
resp = self._get(client, path="note.md", sheet="Budget")
|
|
assert resp.status_code == 415
|
|
|
|
def test_limit_above_the_server_cap_is_rejected(self, client, xlsx_file):
|
|
"""The cap is a contract, not a silent truncation of the request."""
|
|
assert self._get(client, sheet="Budget", limit=100_000).status_code == 422
|
|
|
|
def test_reader_clamps_a_hostile_limit(self, xlsx_file):
|
|
"""Belt and braces: the reader caps too, whoever calls it."""
|
|
from backend.xlsx_reader import MAX_WINDOW_ROWS, read_sheet_window
|
|
|
|
window = read_sheet_window(Path(xlsx_file), "Budget", offset=0, limit=10**9)
|
|
assert window["limit"] == MAX_WINDOW_ROWS
|
|
|
|
def test_reader_rejects_a_negative_offset(self, xlsx_file):
|
|
from backend.xlsx_reader import read_sheet_window
|
|
|
|
window = read_sheet_window(Path(xlsx_file), "Budget", offset=-5, limit=10)
|
|
assert window["offset"] == 0
|
|
|
|
|
|
# ── #153 A4 — formula injection ──────────────────────────────────────────
|
|
|
|
|
|
class TestXlsxFormulaGuard:
|
|
def _save(self, client, cells, **extra):
|
|
return client.put(
|
|
f"/api/file/{VAULT}/xlsx/save",
|
|
params={"path": "budget.xlsx"},
|
|
json={"sheet": "Budget", "cells": cells, **extra},
|
|
)
|
|
|
|
def test_formula_like_value_is_stored_as_text(self, client, xlsx_file):
|
|
resp = self._save(client, {"A3": "=cmd|'/c calc'!A1"})
|
|
assert resp.status_code == 200
|
|
cell = openpyxl.load_workbook(xlsx_file)["Budget"]["A3"]
|
|
assert cell.value == "=cmd|'/c calc'!A1"
|
|
assert cell.data_type == "s" # pas de <f> dans l'archive
|
|
|
|
def test_at_prefix_is_stored_as_text(self, client, xlsx_file):
|
|
self._save(client, {"A4": "@SUM(A1:A2)"})
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A4"].data_type == "s"
|
|
|
|
def test_allow_formula_keeps_a_real_formula(self, client, xlsx_file):
|
|
resp = self._save(client, {"A3": "=B1+5"}, allow_formula=True)
|
|
assert resp.status_code == 200
|
|
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A3"].data_type == "f"
|
|
|
|
def test_numbers_are_unaffected(self, client, xlsx_file):
|
|
self._save(client, {"B3": "42", "B4": "-3.5"})
|
|
ws = openpyxl.load_workbook(xlsx_file)["Budget"]
|
|
assert ws["B3"].value == 42 and isinstance(ws["B3"].value, int)
|
|
assert ws["B4"].value == -3.5
|