221 lines
8.9 KiB
Python
221 lines
8.9 KiB
Python
# tests/test_perf_phase3.py — Optimisation globale des performances (#86, phase 3)
|
|
"""Non-regression tests for the #86 performance work.
|
|
|
|
Covers the three remaining #86 items (the inverted-index search, the lazy PDF
|
|
extraction and the regex CPU caps were already delivered via BUG-033, BUG-040
|
|
and BUG-025):
|
|
|
|
- differential vault scan (unchanged files reused without disk re-read),
|
|
- deferred excalidraw text extraction (scan cheap, enrichment fills text),
|
|
- file-size guard on ``replace_in_files`` (oversized files skipped).
|
|
"""
|
|
import json
|
|
import os
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
|
|
def _excalidraw_doc(*texts: str) -> str:
|
|
elements = [
|
|
{
|
|
"id": f"el{i}",
|
|
"type": "text",
|
|
"text": text,
|
|
"x": 0,
|
|
"y": i * 20,
|
|
"width": 100,
|
|
"height": 20,
|
|
}
|
|
for i, text in enumerate(texts)
|
|
]
|
|
return json.dumps({"type": "excalidraw", "version": 2, "elements": elements})
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# Deferred excalidraw extraction
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
class TestExcalidrawDeferred:
|
|
def test_scan_defers_excalidraw_text(self, tmp_path: Path):
|
|
"""The scan must not parse diagram JSON — content empty + pending flag."""
|
|
from backend.indexer import _scan_vault
|
|
|
|
vault = tmp_path / "vault"
|
|
vault.mkdir()
|
|
(vault / "diagram.excalidraw").write_text(
|
|
_excalidraw_doc("hello diagram"), encoding="utf-8"
|
|
)
|
|
|
|
result = _scan_vault("V", str(vault), {})
|
|
entry = next(f for f in result["files"] if f["path"] == "diagram.excalidraw")
|
|
assert entry["content"] == ""
|
|
assert entry["content_preview"] == ""
|
|
assert entry.get("excalidraw_text_pending") is True
|
|
assert entry["title"] == "diagram"
|
|
|
|
def test_scan_defers_excalidraw_md(self, tmp_path: Path):
|
|
"""``.excalidraw.md`` files are deferred too (no lz decompression at scan)."""
|
|
from backend.indexer import _scan_vault
|
|
|
|
vault = tmp_path / "vault"
|
|
vault.mkdir()
|
|
(vault / "board.excalidraw.md").write_text(
|
|
"---\ntitle: Board\n---\n" + _excalidraw_doc("board text"),
|
|
encoding="utf-8",
|
|
)
|
|
|
|
result = _scan_vault("V", str(vault), {})
|
|
entry = next(f for f in result["files"] if f["path"] == "board.excalidraw.md")
|
|
assert entry["content"] == ""
|
|
assert entry.get("excalidraw_text_pending") is True
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_enrich_fills_excalidraw_text(self, tmp_path: Path):
|
|
"""The deferred pass extracts diagram text and clears the flag."""
|
|
import backend.indexer as idx
|
|
|
|
vault = tmp_path / "vault"
|
|
vault.mkdir()
|
|
(vault / "diagram.excalidraw").write_text(
|
|
_excalidraw_doc("uniquediagword"), encoding="utf-8"
|
|
)
|
|
file_info = {
|
|
"path": "diagram.excalidraw",
|
|
"title": "diagram",
|
|
"tags": [],
|
|
"content": "",
|
|
"content_preview": "",
|
|
"size": 0,
|
|
"modified": "",
|
|
"extension": ".excalidraw",
|
|
"excalidraw_text_pending": True,
|
|
}
|
|
with idx._index_lock:
|
|
idx.index["ExcalV"] = {
|
|
"files": [file_info],
|
|
"tags": {},
|
|
"path": str(vault),
|
|
"paths": [],
|
|
}
|
|
try:
|
|
count = await idx.enrich_pdf_texts("ExcalV")
|
|
assert count == 1
|
|
assert "uniquediagword" in file_info["content"]
|
|
assert file_info["content_preview"]
|
|
assert "excalidraw_text_pending" not in file_info
|
|
finally:
|
|
with idx._index_lock:
|
|
idx.index.pop("ExcalV", None)
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# Differential scan
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
class TestDifferentialScan:
|
|
def test_first_scan_reports_zero_reused(self, test_vault_dir):
|
|
from backend.indexer import _scan_vault
|
|
|
|
result = _scan_vault("TestVault", test_vault_dir)
|
|
assert result["reused"] == 0
|
|
assert len(result["files"]) >= 3
|
|
|
|
def test_unchanged_files_are_reused(self, test_vault_dir):
|
|
from backend.indexer import _scan_vault
|
|
|
|
first = _scan_vault("TestVault", test_vault_dir)
|
|
previous = {f["path"]: f for f in first["files"]}
|
|
second = _scan_vault("TestVault", test_vault_dir, None, previous)
|
|
assert second["reused"] == len(first["files"])
|
|
assert {f["path"] for f in second["files"]} == {f["path"] for f in first["files"]}
|
|
# Tags and titles survive the reuse path.
|
|
assert second["tags"] == first["tags"]
|
|
for f in second["files"]:
|
|
assert f["content"] == previous[f["path"]]["content"]
|
|
|
|
def test_changed_file_is_reparsed(self, test_vault_dir):
|
|
from backend.indexer import _scan_vault
|
|
|
|
first = _scan_vault("TestVault", test_vault_dir)
|
|
previous = {f["path"]: f for f in first["files"]}
|
|
|
|
target = Path(test_vault_dir) / "note1.md"
|
|
target.write_text(
|
|
target.read_text(encoding="utf-8") + "\nMot unique de reparse differentials.",
|
|
encoding="utf-8",
|
|
)
|
|
# Force a visibly different mtime (coarse filesystems).
|
|
os.utime(target, (9999999999, 9999999999))
|
|
|
|
second = _scan_vault("TestVault", test_vault_dir, None, previous)
|
|
assert second["reused"] == len(first["files"]) - 1
|
|
changed = next(f for f in second["files"] if f["path"] == "note1.md")
|
|
assert "Mot unique de reparse differentials" in changed["content"]
|
|
|
|
def test_added_and_deleted_files(self, test_vault_dir):
|
|
from backend.indexer import _scan_vault
|
|
|
|
first = _scan_vault("TestVault", test_vault_dir)
|
|
previous = {f["path"]: f for f in first["files"]}
|
|
|
|
(Path(test_vault_dir) / "brand-new.md").write_text(
|
|
"# Brand new\nFresh content here.", encoding="utf-8"
|
|
)
|
|
(Path(test_vault_dir) / "config.json").unlink()
|
|
|
|
second = _scan_vault("TestVault", test_vault_dir, None, previous)
|
|
paths = {f["path"] for f in second["files"]}
|
|
assert "brand-new.md" in paths
|
|
assert "config.json" not in paths
|
|
assert second["reused"] == len(first["files"]) - 1
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
# replace_in_files size guard
|
|
# ═══════════════════════════════════════════════════════════════════
|
|
|
|
class TestReplaceSizeGuard:
|
|
def test_oversized_file_is_skipped(self, tmp_path: Path, monkeypatch):
|
|
"""Files over MAX_REPLACE_FILE_BYTES are skipped, not read."""
|
|
from backend.services import mutations
|
|
|
|
big = tmp_path / "big.md"
|
|
big.write_text("findme " * 100, encoding="utf-8")
|
|
|
|
monkeypatch.setattr(mutations, "MAX_REPLACE_FILE_BYTES", 10)
|
|
monkeypatch.setattr(
|
|
"backend.services.search.advanced_search_vaults",
|
|
lambda *a, **k: {
|
|
"results": [{"vault": "V", "path": "big.md", "title": "big"}],
|
|
},
|
|
)
|
|
monkeypatch.setattr(
|
|
mutations, "get_vault_root", lambda vault: tmp_path
|
|
)
|
|
|
|
out = mutations.replace_in_files("findme", "replaced", vault="V", dry_run=True)
|
|
assert out["matches"] == []
|
|
assert out["total_matches"] == 0
|
|
|
|
def test_small_file_still_processed(self, tmp_path: Path, monkeypatch):
|
|
from backend.services import mutations
|
|
|
|
small = tmp_path / "small.md"
|
|
small.write_text("findme once", encoding="utf-8")
|
|
|
|
monkeypatch.setattr(mutations, "MAX_REPLACE_FILE_BYTES", 10_000_000)
|
|
monkeypatch.setattr(
|
|
"backend.services.search.advanced_search_vaults",
|
|
lambda *a, **k: {
|
|
"results": [{"vault": "V", "path": "small.md", "title": "small"}],
|
|
},
|
|
)
|
|
monkeypatch.setattr(
|
|
mutations, "get_vault_root", lambda vault: tmp_path
|
|
)
|
|
|
|
out = mutations.replace_in_files("findme", "replaced", vault="V", dry_run=True)
|
|
assert out["total_matches"] == 1
|
|
assert out["matches"][0]["path"] == "small.md"
|