Files
ObsiGate/tests/test_search_advanced.py
T
bruno 31d4616baf feat: garde-fous d'écriture des classeurs Excel #153 (P0)
L'édition d'un .xlsx pouvait détruire une partie du classeur, le
concurrencer en silence, ou diffuser une injection de formule.

- BUG-085 : inspect_workbook() détecte ce qu'un round-trip openpyxl perd
  (valeurs calculées en cache, slicers, contrôles, connexions, custom
  XML, signature, commentaires enrichis, macros) → xlsx_lossy_features
  exposé en lecture, bandeau FR/EN, et 409 xlsx_lossy_content sans
  `force` (confirmation explicite puis reprise). Périmètre réel
  revalidé : graphiques, images et TCD survivent au round-trip.
- BUG-086 : écriture atomique (fichier .tmp + os.replace) : un plantage
  ne peut plus tronquer le classeur, le backup reste intact.
- BUG-087 : verrou par fichier autour du read-modify-write (timeout 15 s,
  409 conflict) ; endpoint xlsx/save devenu synchrone pour que
  l'attente s'exécute dans le threadpool.
- BUG-088 : une saisie en '=' ou '@' est stockée en texte, sauf opt-in
  `allow_formula` ou le bouton f(x) de la visionneuse. Le handler
  ServiceError expose désormais code + details, que api() propage.
- BUG-084 : la suppression d'une vault purge enfin l'index inversé
  (documents fantômes qui continuaient de matcher) et is_stale() devient
  is_ready(), le nom étant trompeur (la staleness n'existe plus).

Tests : 1390 pytest, 10 JSDOM (xlsx-viewer.test.mjs, branché au CI),
3 E2E Playwright, suite E2E complète verte, ruff/mypy 0.

🤖 Generated with Codebuff
Co-Authored-By: Codebuff <[email protected]>
2026-09-27 20:39:12 -04:00

251 lines
10 KiB
Python

# tests/test_search_advanced.py — Tests for InvertedIndex, search/suggest functions
import pytest
from backend.search import (
InvertedIndex,
get_inverted_index,
search,
advanced_search,
suggest_titles,
suggest_tags,
get_all_tags,
)
# ═══════════════════════════════════════════════════════════════════
# InvertedIndex — unit tests
# ═══════════════════════════════════════════════════════════════════
class TestInvertedIndex:
@pytest.fixture(autouse=True)
def _setup(self):
self.inv = InvertedIndex()
yield
def test_initial_state(self):
assert self.inv.doc_count == 0
assert len(self.inv.word_index) == 0
assert len(self.inv.title_index) == 0
assert len(self.inv._sorted_tokens) == 0
assert self.inv._ready is False
def test_rebuild_from_index(self, client):
"""rebuild() populates from the global index."""
self.inv.rebuild()
assert self.inv.doc_count >= 3
assert len(self.inv.word_index) > 0
assert self.inv._ready is True
def test_add_document(self, client):
self.inv.rebuild()
old_count = self.inv.doc_count
file_info = {
"path": "new/file.md",
"title": "Nouveau Fichier",
"tags": ["test", "nouveau"],
"content": "Ceci est un nouveau document de test.",
}
self.inv.add_document("TestVault", "new/file.md", file_info)
assert self.inv.doc_count == old_count + 1
doc_key = "TestVault::new/file.md"
assert doc_key in self.inv.doc_info
assert doc_key in self.inv.vault_docs["TestVault"]
def test_add_document_updates_existing(self, client):
self.inv.rebuild()
# Find an existing doc
existing_doc = next(iter(self.inv.doc_info.values()))
old_count = self.inv.doc_count
existing_doc_copy = dict(existing_doc)
existing_doc_copy["tags"] = ["updated"]
vault = self.inv.doc_vault.get(
f"{list(self.inv.vault_docs.keys())[0]}::{existing_doc['path']}",
list(self.inv.vault_docs.keys())[0]
)
doc_key = f"{vault}::{existing_doc['path']}"
if doc_key in self.inv.doc_info:
self.inv.add_document(vault, existing_doc["path"], existing_doc_copy)
assert self.inv.doc_count == old_count # No change
def test_remove_document(self, client):
self.inv.rebuild()
# Get the first document
doc_keys = list(self.inv.doc_info.keys())
if not doc_keys:
pytest.skip("No documents in inverted index")
doc_key = doc_keys[0]
vault, path = doc_key.split("::", 1)
old_count = self.inv.doc_count
self.inv.remove_document(vault, path)
assert self.inv.doc_count == old_count - 1
assert doc_key not in self.inv.doc_info
def test_remove_nonexistent_document(self, client):
self.inv.rebuild()
old_count = self.inv.doc_count
self.inv.remove_document("FakeVault", "nonexistent.md")
assert self.inv.doc_count == old_count # No change
def test_tag_indexing(self, client):
self.inv.rebuild()
# "python" tag should exist in tag_docs
assert any("python" in tag.lower() for tag in self.inv.tag_docs) or len(self.inv.tag_docs) > 0
def test_title_indexing(self, client):
self.inv.rebuild()
assert len(self.inv.title_index) > 0
def test_sorted_tokens(self, client):
self.inv.rebuild()
assert len(self.inv._sorted_tokens) > 0
# Check sorted order
for i in range(len(self.inv._sorted_tokens) - 1):
assert self.inv._sorted_tokens[i] <= self.inv._sorted_tokens[i + 1]
def test_get_inverted_index_singleton(self, client):
inv1 = get_inverted_index()
inv2 = get_inverted_index()
assert inv1 is inv2 # Same singleton
def test_skip_when_not_ready(self, client):
"""add_document/remove_document are no-ops when _ready is False."""
inv = InvertedIndex()
assert inv._ready is False
inv.add_document("V", "p.md", {"path": "p.md", "title": "T", "tags": [], "content": ""})
assert inv.doc_count == 0 # Skipped
inv.remove_document("V", "p.md")
assert inv.doc_count == 0 # Skipped
def test_is_ready_tracks_initial_build(self, client):
"""is_ready() is the only freshness signal: no generation counter,
no cooldown, no lazy rebuild (plan.md step 6)."""
inv = InvertedIndex()
assert inv.is_ready() is False
inv.rebuild()
assert inv.is_ready() is True
# The old staleness API is gone for good.
assert not hasattr(inv, "is_stale")
def test_is_ready_survives_incremental_updates(self, client):
"""Incremental add/remove must not flip readiness back (which would
silently push search() onto the O(N) full-scan fallback)."""
self.inv.rebuild()
assert self.inv.is_ready() is True
self.inv.add_document("V", "p.md", {"path": "p.md", "title": "T", "tags": [], "content": "x"})
self.inv.remove_document("V", "p.md")
assert self.inv.is_ready() is True
# ═══════════════════════════════════════════════════════════════════
# Search / Advanced Search integration tests
# ═══════════════════════════════════════════════════════════════════
class TestSearchFunctions:
def test_search_basic(self, client):
results = search("python", vault_filter="all")
assert len(results) >= 1
def test_search_vault_filter(self, client):
results = search("python", vault_filter="TestVault")
assert len(results) >= 1
for r in results:
assert r["vault"] == "TestVault"
def test_search_tag_filter(self, client):
results = search("", vault_filter="all", tag_filter="python")
assert len(results) >= 1
def test_search_no_results(self, client):
results = search("xyznonexistent12345", vault_filter="all")
assert len(results) == 0
def test_get_all_tags(self, client):
tags = get_all_tags(vault_filter="all")
assert isinstance(tags, dict)
assert len(tags) > 0
def test_get_all_tags_vault_filter(self, client):
tags = get_all_tags(vault_filter="TestVault")
assert "python" in tags or "docker" in tags
def test_advanced_search_basic(self, client):
result = advanced_search("python", vault_filter="all")
assert isinstance(result, dict)
results = result.get("results", [])
if len(results) == 0:
pytest.skip("No results from advanced search")
r = results[0]
assert "title" in r
assert "score" in r
assert "snippet" in r
def test_advanced_search_with_tag(self, client):
result = advanced_search("", vault_filter="all", tag_filter="python")
assert isinstance(result, dict)
assert "results" in result
def test_advanced_search_relevance_order(self, client):
"""Results should be sorted by score descending."""
result = advanced_search("python", vault_filter="all")
results = result.get("results", [])
if len(results) < 2:
pytest.skip("Not enough results to test ordering")
for i in range(len(results) - 1):
assert results[i]["score"] >= results[i + 1]["score"]
def test_suggest_titles(self, client):
suggestions = suggest_titles("intr", vault_filter="all")
assert len(suggestions) >= 1
s = suggestions[0]
assert "title" in s
assert "vault" in s
assert "path" in s
def test_suggest_titles_no_match(self, client):
suggestions = suggest_titles("xyznonexistent", vault_filter="all")
assert len(suggestions) == 0
def test_suggest_tags(self, client):
suggestions = suggest_tags("py", vault_filter="all")
assert len(suggestions) >= 1 # Should find "python"
def test_suggest_tags_no_match(self, client):
suggestions = suggest_tags("xyznonexistent", vault_filter="all")
assert len(suggestions) == 0
# ═══════════════════════════════════════════════════════════════════
# Vault removal — inverted index must not keep ghost documents
# ═══════════════════════════════════════════════════════════════════
class TestVaultRemovalPurgesInvertedIndex:
"""`remove_vault_from_index()` must notify the inverted-index hook.
Regression: it only cleaned the indexer's own structures, so every document
of the removed vault survived in the inverted index (postings, doc_info,
doc_vault, vault_docs) and kept matching searches for a vault that no
longer exists — a leak that only a manual reindex used to clear.
"""
def test_removing_a_vault_purges_its_documents(self, client):
import asyncio
import backend.indexer as ix
import backend.search as bs
ix.set_index_change_hook(bs._on_index_change_hook)
inv = bs._inverted_index
inv.rebuild()
vault_keys = [k for k in inv.doc_info if k.startswith("TestVault::")]
assert vault_keys, "vault non indexe, test sans valeur"
before = inv.doc_count
asyncio.run(ix.remove_vault_from_index("TestVault"))
ghosts = [k for k in inv.doc_info if k.startswith("TestVault::")]
assert not ghosts, f"documents fantomes dans l'index inverse : {ghosts[:5]}"
assert inv.doc_count == before - len(vault_keys)
assert "TestVault" not in inv.vault_docs
# The index stays usable for the remaining vaults.
assert inv.is_ready() is True