Files
ObsiGate/tests/test_xlsx_viewer.py
T
bruno 06f8e63d06 feat: tableurs indexables, saisie typée et valeurs calculées #153
Les tableurs étaient invisibles à la recherche, chaque saisie devenait
du texte, et une formule n'affichait que sa formule.

- A5 : extract_indexable_text() indexe les noms de feuilles et les 20
  premières lignes (plafond 5 k caractères) pour le TF-IDF et la
  recherche sémantique. Un mot tapé dans une cellule rend le fichier
  trouvable ; un classeur chiffré s'indexe par son seul nom.
- A10 : la saisie est typée comme dans Excel — booléens
  (TRUE/FAUX/OUI/NON) et dates FR JJ/MM/AAAA en ordre jour-first, donc
  01/02/2026 est le 1er février. Une saisie ressemblant à une formule
  n'est jamais convertie (BUG-088 préservé).
- A12 : la valeur calculée par Excel s'affiche sous la formule, via une
  2e lecture data_only=True faite seulement si l'archive contient un
  <v>. Info-bulle traduite FR/EN, aucun texte d'interface côté backend.
- BUG-089 : un reindex manuel ne reconstruisait pas l'index inversé, et
  backend/search.py lisait l'index via un import par valeur — après un
  rechargement du module, la recherche écrivait dans un dict périmé.

Contre-preuves vérifiées pour A5, A10 et A12 (neutralisation de chaque
fonction → échec des tests concernés). Tests : 1402 pytest, 10 JSDOM,
4 E2E, suite E2E complète verte, ruff/mypy 0.

🤖 Generated with Codebuff
Co-Authored-By: Codebuff <[email protected]>
2026-09-27 22:43:24 -04:00

554 lines
21 KiB
Python

"""Display / edit / download for .xlsx files (viewer + PUT xlsx/save).
Covers #152 (affichage / édition) and #153 P0 : A1 alerte de fidélité avant
écriture, A2 écriture atomique, A3 verrou par fichier, A4 neutralisation de
l'injection de formule.
"""
from __future__ import annotations
import zipfile
from pathlib import Path
import pytest
openpyxl = pytest.importorskip("openpyxl")
VAULT = "TestVault"
@pytest.fixture
def xlsx_file(test_vault_dir: str) -> str:
from openpyxl import Workbook
path = Path(test_vault_dir) / "budget.xlsx"
wb = Workbook()
ws = wb.active
ws.title = "Budget"
ws["A1"] = "Poste"
ws["B1"] = 100
ws["A2"] = "Total"
ws["B2"] = "=B1*2"
notes = wb.create_sheet("Notes")
notes["A1"] = "hello"
wb.save(path)
return str(path)
def _add_lossy_parts(path: Path, parts: dict[str, bytes]) -> None:
"""Re-pack *path* with extra OPC parts openpyxl cannot write back."""
with zipfile.ZipFile(path) as zf:
items = {n: zf.read(n) for n in zf.namelist()}
items.update(parts)
with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf:
for name, blob in items.items():
zf.writestr(name, blob)
@pytest.fixture
def lossy_xlsx(test_vault_dir: str) -> str:
"""Workbook with a slicer + a formula carrying its cached result."""
from openpyxl import Workbook
path = Path(test_vault_dir) / "risky.xlsx"
wb = Workbook()
ws = wb.active
ws.title = "Data"
ws["A1"] = 3
ws["A2"] = "=A1*3"
wb.save(path)
# <f>…</f><v>…</v> : openpyxl keeps the formula, drops the cached result.
with zipfile.ZipFile(path) as zf:
items = {n: zf.read(n) for n in zf.namelist()}
sheet = next(n for n in items if n.startswith("xl/worksheets/sheet"))
xml = items[sheet].decode("utf-8").replace(
"<f>A1*3</f>", "<f>A1*3</f><v>9</v>"
)
items[sheet] = xml.encode("utf-8")
items["xl/slicers/slicer1.xml"] = b"<slicer/>"
with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf:
for name, blob in items.items():
zf.writestr(name, blob)
return str(path)
# ── Display ───────────────────────────────────────────────────────────────
class TestXlsxDisplay:
def test_renders_all_sheets(self, client, xlsx_file):
resp = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"})
assert resp.status_code == 200
data = resp.json()
assert data["is_xlsx"] is True
assert data["unsupported"] is False
assert [s["name"] for s in data["xlsx_sheets"]] == ["Budget", "Notes"]
first = data["xlsx_sheets"][0]["html"]
assert "Poste" in first
assert 'data-cell="B1"' in first
assert "=B1*2" in first # formula kept as text (data_only=False)
assert 'data-cell="A1"' in data["xlsx_sheets"][1]["html"]
def test_corrupt_xlsx_returns_500(self, client, test_vault_dir):
bad = Path(test_vault_dir) / "corrupt.xlsx"
bad.write_bytes(b"this is not a zip archive")
resp = client.get(f"/api/file/{VAULT}", params={"path": "corrupt.xlsx"})
assert resp.status_code == 500
# ── Index parity (tree visibility) ────────────────────────────────────────
class TestXlsxIndexing:
def test_xlsx_in_supported_extensions(self):
from backend.indexer import SUPPORTED_EXTENSIONS
assert ".xlsx" in SUPPORTED_EXTENSIONS
def test_xlsx_indexes_sheet_names_and_headers(self, test_vault_dir, xlsx_file):
"""#153 A5 — a workbook is searchable by its cell values."""
from backend.indexer import _index_single_file_sync
info = _index_single_file_sync(VAULT, test_vault_dir, xlsx_file)
assert info is not None
assert info["extension"] == ".xlsx"
# Sheet names + header rows reach TF-IDF (was metadata-only before A5).
assert "Budget" in info["content"]
assert "Poste" in info["content"]
assert info["content_preview"]
assert info["title"] # filename-derived title
class TestXlsxCachedValues:
"""#153 A12 — show what Excel last computed next to each formula."""
def test_cached_result_is_shown_beside_the_formula(self, client, lossy_xlsx):
from backend.xlsx_reader import render_sheets
html = render_sheets(Path(lossy_xlsx))[0]["html"]
# A2 is "=A1*3" with <v>9</v> in the fixture.
assert 'data-cell="A2"' in html
assert "=A1*3" in html
assert "xlsx-cached" in html
assert ">9<" in html # the cached result Excel computed
def test_no_shadow_when_no_formula_carries_a_result(self, client, xlsx_file):
from backend.xlsx_reader import render_sheets
html = render_sheets(Path(xlsx_file))[0]["html"]
assert "xlsx-cached" not in html # budget.xlsx has no <v> at all
def test_plain_cells_are_never_duplicated(self, client, lossy_xlsx):
from backend.xlsx_reader import render_sheets
html = render_sheets(Path(lossy_xlsx))[0]["html"]
# A1 is the literal 3: the two reads agree, so only one value shows.
assert 'data-cell="A1"' in html
assert html.count("xlsx-cached") == 1 # only the formula cell
class TestXlsxValueCoercion:
"""#153 A10 — a typed value comes back with the type Excel would infer."""
def _write(self, client, xlsx_file, ref, value):
return client.put(
f"/api/file/{VAULT}/xlsx/save",
params={"path": "budget.xlsx"},
json={"sheet": "Budget", "cells": {ref: value}, "force": True},
)
def test_number_and_bool_are_stored_as_typed(self, client, xlsx_file):
resp = self._write(
client, xlsx_file, "D1", "42"
)
assert resp.status_code == 200
resp = self._write(client, xlsx_file, "D2", "VRAI")
assert resp.status_code == 200
resp = self._write(client, xlsx_file, "D3", "12/03/2026")
assert resp.status_code == 200
from openpyxl import load_workbook
wb = load_workbook(xlsx_file)
ws = wb["Budget"]
assert ws["D1"].value == 42 and isinstance(ws["D1"].value, int)
assert ws["D2"].value is True
assert ws["D3"].value.year == 2026 and ws["D3"].value.month == 3
assert ws["D3"].value.day == 12 # FR day-first, not 3 December
wb.close()
def test_day_first_date_is_not_read_as_us(self, client, xlsx_file):
"""'01/02/2026' is 1 February in French, not 2 January."""
assert self._write(client, xlsx_file, "E1", "01/02/2026").status_code == 200
from openpyxl import load_workbook
wb = load_workbook(xlsx_file)
d = wb["Budget"]["E1"].value
wb.close()
assert (d.month, d.day) == (2, 1)
def test_ambiguous_text_is_left_alone(self, client, xlsx_file):
"""A version string or a partial date stays text, never a date."""
assert self._write(client, xlsx_file, "F1", "3.14.2").status_code == 200
assert self._write(client, xlsx_file, "F2", "Ref 12/34").status_code == 200
assert self._write(client, xlsx_file, "F3", "12/2026").status_code == 200
from openpyxl import load_workbook
wb = load_workbook(xlsx_file)
ws = wb["Budget"]
assert ws["F1"].value == "3.14.2"
assert ws["F2"].value == "Ref 12/34"
assert ws["F3"].value == "12/2026"
wb.close()
def test_plain_integer_text_becomes_a_number(self, client, xlsx_file):
"""Typing a bare number yields a number, as it did before #153 A10."""
assert self._write(client, xlsx_file, "H1", "75001").status_code == 200
from openpyxl import load_workbook
wb = load_workbook(xlsx_file)
assert wb["Budget"]["H1"].value == 75001
wb.close()
def test_formula_looking_date_stays_text(self, client, xlsx_file):
"""A formula is never mistaken for a date (BUG-088 must not regress)."""
assert self._write(client, xlsx_file, "G1", "=12/03/2026").status_code == 200
from openpyxl import load_workbook
wb = load_workbook(xlsx_file)
c = wb["Budget"]["G1"]
wb.close()
assert c.data_type == "s"
assert c.value == "=12/03/2026"
class TestXlsxSearchable:
"""#153 A5 — a keyword living in a CELL must make the file findable."""
def test_search_finds_a_word_stored_in_a_cell(self, test_vault_dir):
"""A word typed in a CELL must make the workbook findable.
Deliberately hermetic: it drives the indexer and the inverted index
directly instead of going through the HTTP reload, because both are
process-wide singletons that other test modules mutate (some reload the
module outright), which would make this test order-dependent.
"""
from openpyxl import Workbook
import backend.indexer as indexer
import backend.search as search_mod
path = Path(test_vault_dir) / "fournisseurs.xlsx"
wb = Workbook()
ws = wb.active
ws.title = "Contacts"
ws["A1"] = "Fournisseur"
ws["A2"] = "Menuiserie Beaulieu"
wb.save(path)
# Index exactly this vault, from scratch.
indexer.index[VAULT] = {
"files": [indexer._index_single_file_sync(VAULT, test_vault_dir, str(path))],
"tags": {},
"paths": [],
"config": {},
}
entries = [f for f in indexer.index[VAULT]["files"] if f]
assert entries, "le tableur n'a pas ete indexe"
assert "Menuiserie" in entries[0]["content"], (
f"contenu indexe : {entries[0]['content'][:80]!r}"
)
search_mod.init_inverted_index()
hits = [r["path"] for r in search_mod.search("Menuiserie", "all")]
assert "fournisseurs.xlsx" in hits
def test_indexable_text_is_capped(self, tmp_path):
"""A data dump must not flood the index."""
from openpyxl import Workbook
from backend.xlsx_reader import MAX_INDEX_CHARS, extract_indexable_text
path = tmp_path / "huge.xlsx"
wb = Workbook()
ws = wb.active
ws.title = "Big"
for r in range(1, 400):
ws.cell(row=r, column=1, value=f"ligne {r} " + "x" * 60)
wb.save(path)
text = extract_indexable_text(path)
assert len(text) <= MAX_INDEX_CHARS
assert "Big" in text # the sheet name survives
def test_corrupt_workbook_indexes_as_empty_not_a_crash(self, tmp_path):
from backend.xlsx_reader import extract_indexable_text
path = tmp_path / "broken.xlsx"
path.write_bytes(b"not a zip at all")
assert extract_indexable_text(path) == ""
def test_blank_rows_are_skipped(self, tmp_path):
from openpyxl import Workbook
from backend.xlsx_reader import extract_indexable_text
path = tmp_path / "sparse.xlsx"
wb = Workbook()
ws = wb.active
ws.title = "S"
ws["A1"] = "Alpha"
ws["A50"] = "Omega" # beyond the indexed prefix -> ignored on purpose
wb.save(path)
text = extract_indexable_text(path)
assert "Alpha" in text
assert "Omega" not in text
assert "\t\t" not in text # no run of empty columns
# ── Edit ──────────────────────────────────────────────────────────────────
class TestXlsxSave:
def _save(self, client, body, path="budget.xlsx"):
return client.put(
f"/api/file/{VAULT}/xlsx/save",
params={"path": path},
json=body,
)
def test_save_updates_cell_with_number_coercion(self, client, xlsx_file):
resp = self._save(client, {"sheet": "Budget", "cells": {"B1": "250"}})
assert resp.status_code == 200
assert resp.json()["status"] == "ok"
from openpyxl import load_workbook
wb = load_workbook(xlsx_file)
assert wb["Budget"]["B1"].value == 250 # int, not "250"
def test_save_leaves_other_sheets_and_formulas(self, client, xlsx_file):
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "Titre", "B1": 42}})
assert resp.status_code == 200
from openpyxl import load_workbook
wb = load_workbook(xlsx_file)
assert wb["Budget"]["B2"].value == "=B1*2"
assert wb["Notes"]["A1"].value == "hello"
def test_save_empty_string_clears_cell(self, client, xlsx_file):
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": ""}})
assert resp.status_code == 200
from openpyxl import load_workbook
assert load_workbook(xlsx_file)["Budget"]["A1"].value is None
def test_save_creates_backup(self, client, xlsx_file):
from backend.services.backups import get_backup_dir
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "backup-me"}})
assert resp.status_code == 200
backup_dir = Path(get_backup_dir(VAULT, "budget.xlsx"))
assert backup_dir.is_dir()
assert list(backup_dir.glob("*.bak"))
def test_unknown_sheet_400(self, client, xlsx_file):
resp = self._save(client, {"sheet": "Nope", "cells": {"A1": "x"}})
assert resp.status_code == 400
def test_invalid_cell_ref_400(self, client, xlsx_file):
resp = self._save(client, {"sheet": "Budget", "cells": {"A-1": "x"}})
assert resp.status_code == 400
def test_wrong_extension_400(self, client, test_vault_dir):
(Path(test_vault_dir) / "note.md").write_text("# hi\n", encoding="utf-8")
resp = self._save(client, {"sheet": "Sheet", "cells": {"A1": "x"}}, path="note.md")
assert resp.status_code == 400
def test_missing_file_404(self, client):
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "x"}}, path="absent.xlsx")
assert resp.status_code == 404
def test_too_many_cells_400(self, client, xlsx_file):
cells = {f"A{i}": i for i in range(1, 502)}
resp = self._save(client, {"sheet": "Budget", "cells": cells})
assert resp.status_code == 400
def test_nested_value_400(self, client, xlsx_file):
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": {"nested": 1}}})
assert resp.status_code == 400
def test_missing_sheet_field_400(self, client, xlsx_file):
resp = self._save(client, {"cells": {"A1": "x"}})
assert resp.status_code == 400
def test_non_boolean_flag_400(self, client, xlsx_file):
for flag in ("force", "allow_formula"):
resp = self._save(client, {"sheet": "Budget", "cells": {"A1": "x"}, flag: "yes"})
assert resp.status_code == 400, flag
# ── #153 A1 — lossy-write guard ──────────────────────────────────────────
class TestXlsxLossyGuard:
def _save(self, client, body, path):
return client.put(
f"/api/file/{VAULT}/xlsx/save", params={"path": path}, json=body,
)
def test_read_reports_lossy_features(self, client, lossy_xlsx):
data = client.get(f"/api/file/{VAULT}", params={"path": "risky.xlsx"}).json()
assert data["is_xlsx"] is True
assert "slicers" in data["xlsx_lossy_features"]
assert "cached_values" in data["xlsx_lossy_features"]
def test_read_reports_nothing_for_a_plain_workbook(self, client, xlsx_file):
data = client.get(f"/api/file/{VAULT}", params={"path": "budget.xlsx"}).json()
# B2 holds "=B1*2" but openpyxl wrote no cached <v> for it.
assert data["xlsx_lossy_features"] == []
def test_save_refuses_without_force(self, client, lossy_xlsx):
resp = self._save(client, {"sheet": "Data", "cells": {"B1": "hello"}}, "risky.xlsx")
assert resp.status_code == 409
body = resp.json()
assert body["code"] == "xlsx_lossy_content"
assert "slicers" in body["details"]["features"]
def test_refused_save_leaves_the_file_untouched(self, client, lossy_xlsx):
before = Path(lossy_xlsx).read_bytes()
self._save(client, {"sheet": "Data", "cells": {"B1": "hello"}}, "risky.xlsx")
assert Path(lossy_xlsx).read_bytes() == before
def test_save_with_force_succeeds(self, client, lossy_xlsx):
resp = self._save(
client,
{"sheet": "Data", "cells": {"B1": "hello"}, "force": True},
"risky.xlsx",
)
assert resp.status_code == 200
assert openpyxl.load_workbook(lossy_xlsx)["Data"]["B1"].value == "hello"
def test_inspect_flags_every_known_family(self, lossy_xlsx):
from backend.xlsx_reader import LOSSY_PARTS, inspect_workbook
for key, prefixes in LOSSY_PARTS.items():
_add_lossy_parts(
Path(lossy_xlsx),
{f"{prefixes[0]}probe.xml": b"<x/>" for _ in [0]},
)
assert key in inspect_workbook(Path(lossy_xlsx)), key
def test_inspect_is_quiet_on_a_corrupt_archive(self, test_vault_dir):
from backend.xlsx_reader import inspect_workbook
bad = Path(test_vault_dir) / "broken.xlsx"
bad.write_bytes(b"not a zip at all")
assert inspect_workbook(bad) == []
# ── #153 A2 — atomic write ───────────────────────────────────────────────
class TestXlsxAtomicWrite:
def test_failed_save_keeps_the_original(self, client, xlsx_file, monkeypatch):
from openpyxl.workbook.workbook import Workbook
before = Path(xlsx_file).read_bytes()
def boom(self, *args, **kwargs):
raise OSError("disk full")
monkeypatch.setattr(Workbook, "save", boom)
# TestClient re-raises the server exception (in production: 500).
with pytest.raises(OSError):
client.put(
f"/api/file/{VAULT}/xlsx/save",
params={"path": "budget.xlsx"},
json={"sheet": "Budget", "cells": {"A1": "perdu"}},
)
# The workbook on disk is byte-identical : the write never reached it.
assert Path(xlsx_file).read_bytes() == before
# No temporary file left behind in the vault.
assert list(Path(xlsx_file).parent.glob("*.tmp")) == []
def test_no_tmp_left_after_a_successful_save(self, client, xlsx_file):
client.put(
f"/api/file/{VAULT}/xlsx/save",
params={"path": "budget.xlsx"},
json={"sheet": "Budget", "cells": {"A1": "ok"}},
)
assert list(Path(xlsx_file).parent.glob("*.tmp")) == []
# ── #153 A3 — per-file write lock ────────────────────────────────────────
class TestXlsxWriteLock:
def test_concurrent_write_returns_409(self, client, xlsx_file):
from backend.services import mutations
key = str(Path(xlsx_file).resolve())
with mutations._xlsx_write_lock(key):
resp = client.put(
f"/api/file/{VAULT}/xlsx/save",
params={"path": "budget.xlsx"},
json={"sheet": "Budget", "cells": {"A1": "concurrent"}},
)
assert resp.status_code == 409
assert resp.json()["code"] == "conflict"
# The blocked call wrote nothing.
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A1"].value == "Poste"
def test_lock_is_released_after_a_normal_save(self, client, xlsx_file):
from backend.services import mutations
key = str(Path(xlsx_file).resolve())
client.put(
f"/api/file/{VAULT}/xlsx/save",
params={"path": "budget.xlsx"},
json={"sheet": "Budget", "cells": {"A1": "premier"}},
)
# The lock must be free again once the request returned.
with mutations._xlsx_write_lock(key):
pass
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A1"].value == "premier"
# ── #153 A4 — formula injection ──────────────────────────────────────────
class TestXlsxFormulaGuard:
def _save(self, client, cells, **extra):
return client.put(
f"/api/file/{VAULT}/xlsx/save",
params={"path": "budget.xlsx"},
json={"sheet": "Budget", "cells": cells, **extra},
)
def test_formula_like_value_is_stored_as_text(self, client, xlsx_file):
resp = self._save(client, {"A3": "=cmd|'/c calc'!A1"})
assert resp.status_code == 200
cell = openpyxl.load_workbook(xlsx_file)["Budget"]["A3"]
assert cell.value == "=cmd|'/c calc'!A1"
assert cell.data_type == "s" # pas de <f> dans l'archive
def test_at_prefix_is_stored_as_text(self, client, xlsx_file):
self._save(client, {"A4": "@SUM(A1:A2)"})
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A4"].data_type == "s"
def test_allow_formula_keeps_a_real_formula(self, client, xlsx_file):
resp = self._save(client, {"A3": "=B1+5"}, allow_formula=True)
assert resp.status_code == 200
assert openpyxl.load_workbook(xlsx_file)["Budget"]["A3"].data_type == "f"
def test_numbers_are_unaffected(self, client, xlsx_file):
self._save(client, {"B3": "42", "B4": "-3.5"})
ws = openpyxl.load_workbook(xlsx_file)["Budget"]
assert ws["B3"].value == 42 and isinstance(ws["B3"].value, int)
assert ws["B4"].value == -3.5