L'édition d'un .xlsx pouvait détruire une partie du classeur, le concurrencer en silence, ou diffuser une injection de formule. - BUG-085 : inspect_workbook() détecte ce qu'un round-trip openpyxl perd (valeurs calculées en cache, slicers, contrôles, connexions, custom XML, signature, commentaires enrichis, macros) → xlsx_lossy_features exposé en lecture, bandeau FR/EN, et 409 xlsx_lossy_content sans `force` (confirmation explicite puis reprise). Périmètre réel revalidé : graphiques, images et TCD survivent au round-trip. - BUG-086 : écriture atomique (fichier .tmp + os.replace) : un plantage ne peut plus tronquer le classeur, le backup reste intact. - BUG-087 : verrou par fichier autour du read-modify-write (timeout 15 s, 409 conflict) ; endpoint xlsx/save devenu synchrone pour que l'attente s'exécute dans le threadpool. - BUG-088 : une saisie en '=' ou '@' est stockée en texte, sauf opt-in `allow_formula` ou le bouton f(x) de la visionneuse. Le handler ServiceError expose désormais code + details, que api() propage. - BUG-084 : la suppression d'une vault purge enfin l'index inversé (documents fantômes qui continuaient de matcher) et is_stale() devient is_ready(), le nom étant trompeur (la staleness n'existe plus). Tests : 1390 pytest, 10 JSDOM (xlsx-viewer.test.mjs, branché au CI), 3 E2E Playwright, suite E2E complète verte, ruff/mypy 0. 🤖 Generated with Codebuff Co-Authored-By: Codebuff <[email protected]>
251 lines
10 KiB
Python
251 lines
10 KiB
Python
# tests/test_search_advanced.py — Tests for InvertedIndex, search/suggest functions
|
|
import pytest
|
|
from backend.search import (
|
|
InvertedIndex,
|
|
get_inverted_index,
|
|
search,
|
|
advanced_search,
|
|
suggest_titles,
|
|
suggest_tags,
|
|
get_all_tags,
|
|
)
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# InvertedIndex — unit tests
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
class TestInvertedIndex:
|
|
@pytest.fixture(autouse=True)
|
|
def _setup(self):
|
|
self.inv = InvertedIndex()
|
|
yield
|
|
|
|
def test_initial_state(self):
|
|
assert self.inv.doc_count == 0
|
|
assert len(self.inv.word_index) == 0
|
|
assert len(self.inv.title_index) == 0
|
|
assert len(self.inv._sorted_tokens) == 0
|
|
assert self.inv._ready is False
|
|
|
|
def test_rebuild_from_index(self, client):
|
|
"""rebuild() populates from the global index."""
|
|
self.inv.rebuild()
|
|
assert self.inv.doc_count >= 3
|
|
assert len(self.inv.word_index) > 0
|
|
assert self.inv._ready is True
|
|
|
|
def test_add_document(self, client):
|
|
self.inv.rebuild()
|
|
old_count = self.inv.doc_count
|
|
file_info = {
|
|
"path": "new/file.md",
|
|
"title": "Nouveau Fichier",
|
|
"tags": ["test", "nouveau"],
|
|
"content": "Ceci est un nouveau document de test.",
|
|
}
|
|
self.inv.add_document("TestVault", "new/file.md", file_info)
|
|
assert self.inv.doc_count == old_count + 1
|
|
doc_key = "TestVault::new/file.md"
|
|
assert doc_key in self.inv.doc_info
|
|
assert doc_key in self.inv.vault_docs["TestVault"]
|
|
|
|
def test_add_document_updates_existing(self, client):
|
|
self.inv.rebuild()
|
|
# Find an existing doc
|
|
existing_doc = next(iter(self.inv.doc_info.values()))
|
|
old_count = self.inv.doc_count
|
|
existing_doc_copy = dict(existing_doc)
|
|
existing_doc_copy["tags"] = ["updated"]
|
|
vault = self.inv.doc_vault.get(
|
|
f"{list(self.inv.vault_docs.keys())[0]}::{existing_doc['path']}",
|
|
list(self.inv.vault_docs.keys())[0]
|
|
)
|
|
doc_key = f"{vault}::{existing_doc['path']}"
|
|
if doc_key in self.inv.doc_info:
|
|
self.inv.add_document(vault, existing_doc["path"], existing_doc_copy)
|
|
assert self.inv.doc_count == old_count # No change
|
|
|
|
def test_remove_document(self, client):
|
|
self.inv.rebuild()
|
|
# Get the first document
|
|
doc_keys = list(self.inv.doc_info.keys())
|
|
if not doc_keys:
|
|
pytest.skip("No documents in inverted index")
|
|
doc_key = doc_keys[0]
|
|
vault, path = doc_key.split("::", 1)
|
|
old_count = self.inv.doc_count
|
|
self.inv.remove_document(vault, path)
|
|
assert self.inv.doc_count == old_count - 1
|
|
assert doc_key not in self.inv.doc_info
|
|
|
|
def test_remove_nonexistent_document(self, client):
|
|
self.inv.rebuild()
|
|
old_count = self.inv.doc_count
|
|
self.inv.remove_document("FakeVault", "nonexistent.md")
|
|
assert self.inv.doc_count == old_count # No change
|
|
|
|
def test_tag_indexing(self, client):
|
|
self.inv.rebuild()
|
|
# "python" tag should exist in tag_docs
|
|
assert any("python" in tag.lower() for tag in self.inv.tag_docs) or len(self.inv.tag_docs) > 0
|
|
|
|
def test_title_indexing(self, client):
|
|
self.inv.rebuild()
|
|
assert len(self.inv.title_index) > 0
|
|
|
|
def test_sorted_tokens(self, client):
|
|
self.inv.rebuild()
|
|
assert len(self.inv._sorted_tokens) > 0
|
|
# Check sorted order
|
|
for i in range(len(self.inv._sorted_tokens) - 1):
|
|
assert self.inv._sorted_tokens[i] <= self.inv._sorted_tokens[i + 1]
|
|
|
|
def test_get_inverted_index_singleton(self, client):
|
|
inv1 = get_inverted_index()
|
|
inv2 = get_inverted_index()
|
|
assert inv1 is inv2 # Same singleton
|
|
|
|
def test_skip_when_not_ready(self, client):
|
|
"""add_document/remove_document are no-ops when _ready is False."""
|
|
inv = InvertedIndex()
|
|
assert inv._ready is False
|
|
inv.add_document("V", "p.md", {"path": "p.md", "title": "T", "tags": [], "content": ""})
|
|
assert inv.doc_count == 0 # Skipped
|
|
inv.remove_document("V", "p.md")
|
|
assert inv.doc_count == 0 # Skipped
|
|
|
|
def test_is_ready_tracks_initial_build(self, client):
|
|
"""is_ready() is the only freshness signal: no generation counter,
|
|
no cooldown, no lazy rebuild (plan.md step 6)."""
|
|
inv = InvertedIndex()
|
|
assert inv.is_ready() is False
|
|
inv.rebuild()
|
|
assert inv.is_ready() is True
|
|
# The old staleness API is gone for good.
|
|
assert not hasattr(inv, "is_stale")
|
|
|
|
def test_is_ready_survives_incremental_updates(self, client):
|
|
"""Incremental add/remove must not flip readiness back (which would
|
|
silently push search() onto the O(N) full-scan fallback)."""
|
|
self.inv.rebuild()
|
|
assert self.inv.is_ready() is True
|
|
self.inv.add_document("V", "p.md", {"path": "p.md", "title": "T", "tags": [], "content": "x"})
|
|
self.inv.remove_document("V", "p.md")
|
|
assert self.inv.is_ready() is True
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# Search / Advanced Search integration tests
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
class TestSearchFunctions:
|
|
def test_search_basic(self, client):
|
|
results = search("python", vault_filter="all")
|
|
assert len(results) >= 1
|
|
|
|
def test_search_vault_filter(self, client):
|
|
results = search("python", vault_filter="TestVault")
|
|
assert len(results) >= 1
|
|
for r in results:
|
|
assert r["vault"] == "TestVault"
|
|
|
|
def test_search_tag_filter(self, client):
|
|
results = search("", vault_filter="all", tag_filter="python")
|
|
assert len(results) >= 1
|
|
|
|
def test_search_no_results(self, client):
|
|
results = search("xyznonexistent12345", vault_filter="all")
|
|
assert len(results) == 0
|
|
|
|
def test_get_all_tags(self, client):
|
|
tags = get_all_tags(vault_filter="all")
|
|
assert isinstance(tags, dict)
|
|
assert len(tags) > 0
|
|
|
|
def test_get_all_tags_vault_filter(self, client):
|
|
tags = get_all_tags(vault_filter="TestVault")
|
|
assert "python" in tags or "docker" in tags
|
|
|
|
def test_advanced_search_basic(self, client):
|
|
result = advanced_search("python", vault_filter="all")
|
|
assert isinstance(result, dict)
|
|
results = result.get("results", [])
|
|
if len(results) == 0:
|
|
pytest.skip("No results from advanced search")
|
|
r = results[0]
|
|
assert "title" in r
|
|
assert "score" in r
|
|
assert "snippet" in r
|
|
|
|
def test_advanced_search_with_tag(self, client):
|
|
result = advanced_search("", vault_filter="all", tag_filter="python")
|
|
assert isinstance(result, dict)
|
|
assert "results" in result
|
|
|
|
def test_advanced_search_relevance_order(self, client):
|
|
"""Results should be sorted by score descending."""
|
|
result = advanced_search("python", vault_filter="all")
|
|
results = result.get("results", [])
|
|
if len(results) < 2:
|
|
pytest.skip("Not enough results to test ordering")
|
|
for i in range(len(results) - 1):
|
|
assert results[i]["score"] >= results[i + 1]["score"]
|
|
|
|
def test_suggest_titles(self, client):
|
|
suggestions = suggest_titles("intr", vault_filter="all")
|
|
assert len(suggestions) >= 1
|
|
s = suggestions[0]
|
|
assert "title" in s
|
|
assert "vault" in s
|
|
assert "path" in s
|
|
|
|
def test_suggest_titles_no_match(self, client):
|
|
suggestions = suggest_titles("xyznonexistent", vault_filter="all")
|
|
assert len(suggestions) == 0
|
|
|
|
def test_suggest_tags(self, client):
|
|
suggestions = suggest_tags("py", vault_filter="all")
|
|
assert len(suggestions) >= 1 # Should find "python"
|
|
|
|
def test_suggest_tags_no_match(self, client):
|
|
suggestions = suggest_tags("xyznonexistent", vault_filter="all")
|
|
assert len(suggestions) == 0
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# Vault removal — inverted index must not keep ghost documents
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
class TestVaultRemovalPurgesInvertedIndex:
|
|
"""`remove_vault_from_index()` must notify the inverted-index hook.
|
|
|
|
Regression: it only cleaned the indexer's own structures, so every document
|
|
of the removed vault survived in the inverted index (postings, doc_info,
|
|
doc_vault, vault_docs) and kept matching searches for a vault that no
|
|
longer exists — a leak that only a manual reindex used to clear.
|
|
"""
|
|
|
|
def test_removing_a_vault_purges_its_documents(self, client):
|
|
import asyncio
|
|
|
|
import backend.indexer as ix
|
|
import backend.search as bs
|
|
|
|
ix.set_index_change_hook(bs._on_index_change_hook)
|
|
inv = bs._inverted_index
|
|
inv.rebuild()
|
|
|
|
vault_keys = [k for k in inv.doc_info if k.startswith("TestVault::")]
|
|
assert vault_keys, "vault non indexe, test sans valeur"
|
|
before = inv.doc_count
|
|
|
|
asyncio.run(ix.remove_vault_from_index("TestVault"))
|
|
|
|
ghosts = [k for k in inv.doc_info if k.startswith("TestVault::")]
|
|
assert not ghosts, f"documents fantomes dans l'index inverse : {ghosts[:5]}"
|
|
assert inv.doc_count == before - len(vault_keys)
|
|
assert "TestVault" not in inv.vault_docs
|
|
# The index stays usable for the remaining vaults.
|
|
assert inv.is_ready() is True
|