# tests/test_perf_phase3.py — Optimisation globale des performances (#86, phase 3) """Non-regression tests for the #86 performance work. Covers the three remaining #86 items (the inverted-index search, the lazy PDF extraction and the regex CPU caps were already delivered via BUG-033, BUG-040 and BUG-025): - differential vault scan (unchanged files reused without disk re-read), - deferred excalidraw text extraction (scan cheap, enrichment fills text), - file-size guard on ``replace_in_files`` (oversized files skipped). """ import json import os from pathlib import Path import pytest def _excalidraw_doc(*texts: str) -> str: elements = [ { "id": f"el{i}", "type": "text", "text": text, "x": 0, "y": i * 20, "width": 100, "height": 20, } for i, text in enumerate(texts) ] return json.dumps({"type": "excalidraw", "version": 2, "elements": elements}) # ═══════════════════════════════════════════════════════════════════ # Deferred excalidraw extraction # ═══════════════════════════════════════════════════════════════════ class TestExcalidrawDeferred: def test_scan_defers_excalidraw_text(self, tmp_path: Path): """The scan must not parse diagram JSON — content empty + pending flag.""" from backend.indexer import _scan_vault vault = tmp_path / "vault" vault.mkdir() (vault / "diagram.excalidraw").write_text( _excalidraw_doc("hello diagram"), encoding="utf-8" ) result = _scan_vault("V", str(vault), {}) entry = next(f for f in result["files"] if f["path"] == "diagram.excalidraw") assert entry["content"] == "" assert entry["content_preview"] == "" assert entry.get("excalidraw_text_pending") is True assert entry["title"] == "diagram" def test_scan_defers_excalidraw_md(self, tmp_path: Path): """``.excalidraw.md`` files are deferred too (no lz decompression at scan).""" from backend.indexer import _scan_vault vault = tmp_path / "vault" vault.mkdir() (vault / "board.excalidraw.md").write_text( "---\ntitle: Board\n---\n" + _excalidraw_doc("board text"), encoding="utf-8", ) result = _scan_vault("V", str(vault), {}) entry = next(f for f in result["files"] if f["path"] == "board.excalidraw.md") assert entry["content"] == "" assert entry.get("excalidraw_text_pending") is True @pytest.mark.asyncio async def test_enrich_fills_excalidraw_text(self, tmp_path: Path): """The deferred pass extracts diagram text and clears the flag.""" import backend.indexer as idx vault = tmp_path / "vault" vault.mkdir() (vault / "diagram.excalidraw").write_text( _excalidraw_doc("uniquediagword"), encoding="utf-8" ) file_info = { "path": "diagram.excalidraw", "title": "diagram", "tags": [], "content": "", "content_preview": "", "size": 0, "modified": "", "extension": ".excalidraw", "excalidraw_text_pending": True, } with idx._index_lock: idx.index["ExcalV"] = { "files": [file_info], "tags": {}, "path": str(vault), "paths": [], } try: count = await idx.enrich_pdf_texts("ExcalV") assert count == 1 assert "uniquediagword" in file_info["content"] assert file_info["content_preview"] assert "excalidraw_text_pending" not in file_info finally: with idx._index_lock: idx.index.pop("ExcalV", None) # ═══════════════════════════════════════════════════════════════════ # Differential scan # ═══════════════════════════════════════════════════════════════════ class TestDifferentialScan: def test_first_scan_reports_zero_reused(self, test_vault_dir): from backend.indexer import _scan_vault result = _scan_vault("TestVault", test_vault_dir) assert result["reused"] == 0 assert len(result["files"]) >= 3 def test_unchanged_files_are_reused(self, test_vault_dir): from backend.indexer import _scan_vault first = _scan_vault("TestVault", test_vault_dir) previous = {f["path"]: f for f in first["files"]} second = _scan_vault("TestVault", test_vault_dir, None, previous) assert second["reused"] == len(first["files"]) assert {f["path"] for f in second["files"]} == {f["path"] for f in first["files"]} # Tags and titles survive the reuse path. assert second["tags"] == first["tags"] for f in second["files"]: assert f["content"] == previous[f["path"]]["content"] def test_changed_file_is_reparsed(self, test_vault_dir): from backend.indexer import _scan_vault first = _scan_vault("TestVault", test_vault_dir) previous = {f["path"]: f for f in first["files"]} target = Path(test_vault_dir) / "note1.md" target.write_text( target.read_text(encoding="utf-8") + "\nMot unique de reparse differentials.", encoding="utf-8", ) # Force a visibly different mtime (coarse filesystems). os.utime(target, (9999999999, 9999999999)) second = _scan_vault("TestVault", test_vault_dir, None, previous) assert second["reused"] == len(first["files"]) - 1 changed = next(f for f in second["files"] if f["path"] == "note1.md") assert "Mot unique de reparse differentials" in changed["content"] def test_added_and_deleted_files(self, test_vault_dir): from backend.indexer import _scan_vault first = _scan_vault("TestVault", test_vault_dir) previous = {f["path"]: f for f in first["files"]} (Path(test_vault_dir) / "brand-new.md").write_text( "# Brand new\nFresh content here.", encoding="utf-8" ) (Path(test_vault_dir) / "config.json").unlink() second = _scan_vault("TestVault", test_vault_dir, None, previous) paths = {f["path"] for f in second["files"]} assert "brand-new.md" in paths assert "config.json" not in paths assert second["reused"] == len(first["files"]) - 1 # ═══════════════════════════════════════════════════════════════════ # replace_in_files size guard # ═══════════════════════════════════════════════════════════════════ class TestReplaceSizeGuard: def test_oversized_file_is_skipped(self, tmp_path: Path, monkeypatch): """Files over MAX_REPLACE_FILE_BYTES are skipped, not read.""" from backend.services import mutations big = tmp_path / "big.md" big.write_text("findme " * 100, encoding="utf-8") monkeypatch.setattr(mutations, "MAX_REPLACE_FILE_BYTES", 10) monkeypatch.setattr( "backend.services.search.advanced_search_vaults", lambda *a, **k: { "results": [{"vault": "V", "path": "big.md", "title": "big"}], }, ) monkeypatch.setattr( mutations, "get_vault_root", lambda vault: tmp_path ) out = mutations.replace_in_files("findme", "replaced", vault="V", dry_run=True) assert out["matches"] == [] assert out["total_matches"] == 0 def test_small_file_still_processed(self, tmp_path: Path, monkeypatch): from backend.services import mutations small = tmp_path / "small.md" small.write_text("findme once", encoding="utf-8") monkeypatch.setattr(mutations, "MAX_REPLACE_FILE_BYTES", 10_000_000) monkeypatch.setattr( "backend.services.search.advanced_search_vaults", lambda *a, **k: { "results": [{"vault": "V", "path": "small.md", "title": "small"}], }, ) monkeypatch.setattr( mutations, "get_vault_root", lambda vault: tmp_path ) out = mutations.replace_in_files("findme", "replaced", vault="V", dry_run=True) assert out["total_matches"] == 1 assert out["matches"][0]["path"] == "small.md"