From f621620593d49b6a9c8914722f600a01ad584b5b Mon Sep 17 00:00:00 2001 From: Bruno Charest Date: Tue, 22 Sep 2026 19:04:23 -0400 Subject: [PATCH] feat: optimisation globale des performances #86 (scan differentiel, excalidraw differe, garde-fou replace; inverted index/PDF lazy/caps regex deja livres via BUG-033/040/025) --- CHANGELOG.md | 19 ++- README.fr.md | 6 +- README.md | 6 +- VERSION | 2 +- backend/indexer.py | 162 ++++++++++++++++++----- backend/services/mutations.py | 15 +++ desktop/Cargo.lock | 2 +- desktop/Cargo.toml | 2 +- desktop/tauri.conf.json | 2 +- docs/ROADMAP.md | 18 +-- docs/features/perf-phase3-86.md | 69 ++++++++++ package.json | 2 +- tests/test_perf_phase3.py | 220 ++++++++++++++++++++++++++++++++ 13 files changed, 470 insertions(+), 55 deletions(-) create mode 100644 docs/features/perf-phase3-86.md create mode 100644 tests/test_perf_phase3.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 653eac6..cf815cc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,7 +6,7 @@ Format basé sur [Keep a Changelog](https://keepachangelog.com/fr/1.1.0/), et [Semantic Versioning](https://semver.org/spec/v2.0.0.html). > **En cours de développement** : les changements à venir sont listés dans la section -> [Unreleased](#unreleased). La dernière version livrée est **2.15.0**. +> [Unreleased](#unreleased). La dernière version livrée est **2.16.0**. --- @@ -14,6 +14,23 @@ et [Semantic Versioning](https://semver.org/spec/v2.0.0.html). --- +## [2.16.0] — 2026-09-22 + +### Modifié + +- **#86 - Optimisation globale des performances (phase 3)** : ferme les deux derniers + points de la phase 3 (recherche via inverted index, PDF lazy et caps regex déjà livrés + via BUG-033/BUG-040/BUG-025). Scan **différentiel** : `_scan_vault` réutilise les + entrées inchangées (`size` + `modified`) d'un snapshot précédent — seuls `os.walk` + + `stat` tournent à chaque rebuild (`build_index`, `reload_single_vault`). Extraction + **excalidraw différée** : le scan ne lit plus les `.excalidraw` / `.excalidraw.md` + (flag `excalidraw_text_pending`), `enrich_pdf_texts()` extrait leur texte après index + comme pour les PDF. Garde-fou `MAX_REPLACE_FILE_BYTES` (5 Mio) sur `replace_in_files`. + Tests : `tests/test_perf_phase3.py` (9). Voir + [docs/features/perf-phase3-86.md](./docs/features/perf-phase3-86.md). + +--- + ## [2.15.0] — 2026-09-22 ### Ajouté diff --git a/README.fr.md b/README.fr.md index 77207f3..1996f35 100644 --- a/README.fr.md +++ b/README.fr.md @@ -4,7 +4,7 @@ **Porte d'entrée web ultra-léger pour vos vaults Obsidian** — Accédez, naviguez et recherchez dans toutes vos notes Obsidian depuis n'importe quel appareil via une interface web moderne et responsive. -[![Version](https://img.shields.io/badge/Version-2.15.0-blue.svg)]() +[![Version](https://img.shields.io/badge/Version-2.16.0-blue.svg)]() [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) [![Docker](https://img.shields.io/badge/Docker-Ready-blue.svg)](https://www.docker.com/) [![Python](https://img.shields.io/badge/Python-3.11+-green.svg)](https://www.python.org/) @@ -927,8 +927,8 @@ Ce projet est sous licence **MIT** — voir le fichier [LICENSE](LICENSE) pour l ## 📝 Changelog -Consultez le [CHANGELOG.md](./CHANGELOG.md) pour l'historique complet de toutes les versions (v1.0.0 → v2.15.0). +Consultez le [CHANGELOG.md](./CHANGELOG.md) pour l'historique complet de toutes les versions (v1.0.0 → v2.16.0). --- -*Projet : ObsiGate | Version : 2.15.0 | Dernière mise à jour : Juin 2026* +*Projet : ObsiGate | Version : 2.16.0 | Dernière mise à jour : Juin 2026* diff --git a/README.md b/README.md index c80f8c5..6061b36 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ **Ultra-light web gateway for your Obsidian vaults** — Access, browse, and search all your Obsidian notes from any device via a modern, responsive web interface. -[![Version](https://img.shields.io/badge/Version-2.15.0-blue.svg)]() +[![Version](https://img.shields.io/badge/Version-2.16.0-blue.svg)]() [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) [![Docker](https://img.shields.io/badge/Docker-Ready-blue.svg)](https://www.docker.com/) [![Python](https://img.shields.io/badge/Python-3.11+-green.svg)](https://www.python.org/) @@ -1096,8 +1096,8 @@ This project is licensed under the **MIT License** - see the [LICENSE](LICENSE) ## 📝 Changelog -See [CHANGELOG.md](./CHANGELOG.md) for the complete version history (v1.0.0 → v2.15.0). +See [CHANGELOG.md](./CHANGELOG.md) for the complete version history (v1.0.0 → v2.16.0). --- -*Project: ObsiGate | Version: 2.15.0 | Last updated: May 2026* +*Project: ObsiGate | Version: 2.16.0 | Last updated: May 2026* diff --git a/VERSION b/VERSION index 68e69e4..7524906 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -2.15.0 +2.16.0 diff --git a/backend/indexer.py b/backend/indexer.py index 69866cf..21111cf 100644 --- a/backend/indexer.py +++ b/backend/indexer.py @@ -397,31 +397,52 @@ def parse_markdown_file(raw: str) -> frontmatter.Post: return frontmatter.Post(content) -def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | None = None) -> dict[str, Any]: +def _scan_vault( + vault_name: str, + vault_path: str, + vault_cfg: dict[str, Any] | None = None, + previous_files: dict[str, dict[str, Any]] | None = None, +) -> dict[str, Any]: """Synchronously scan a single vault directory and build file index. Walks the vault tree, reads supported files, extracts metadata (tags, title, content preview) and stores a capped content snapshot for in-memory full-text search. - + All files and directories are indexed, including hidden files (starting with '.'). + Differential scan (#86): when ``previous_files`` maps a relative path to + its previous ``file_info`` dict, entries whose ``size`` and ``modified`` + timestamp are unchanged are reused verbatim (no disk read, no re-parse). + Only the cheap ``os.walk`` + ``stat`` runs on every pass; heavy content + extraction (PDF metadata excepted — always cheap) is skipped for + unchanged files. This replaces the full ``rglob`` re-read on rebuilds. + + Excalidraw diagrams (#86, like PDFs since BUG-040) are deferred: the scan + only records the title and sets ``excalidraw_text_pending``; the expensive + JSON/lz-string text extraction runs in ``enrich_pdf_texts()`` after the + index is queryable. + Args: vault_name: Display name of the vault. vault_path: Absolute filesystem path to the vault root. vault_cfg: Optional vault configuration dict (unused for indexing, kept for compatibility). + previous_files: Optional ``{relative_path: file_info}`` snapshot from a + previous scan used for differential reuse. Returns: - Dict with keys ``files`` (list), ``tags`` (counter dict), ``path`` (str), ``paths`` (list). + Dict with keys ``files`` (list), ``tags`` (counter dict), ``path`` (str), + ``paths`` (list) and ``reused`` (int, differential hits). """ vault_root = Path(vault_path) files: list[dict[str, Any]] = [] tag_counts: dict[str, int] = {} paths: list[dict[str, str]] = [] + reused = 0 if not vault_root.exists(): logger.warning(f"Vault path does not exist: {vault_path}") - return {"files": [], "tags": {}, "path": vault_path, "paths": []} + return {"files": [], "tags": {}, "path": vault_path, "paths": [], "reused": 0} root_resolved = vault_root.resolve(strict=False) @@ -479,9 +500,37 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | No stat = fpath.stat() modified = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc).isoformat() + # #86 differential scan: reuse the previous entry when neither + # size nor mtime changed — skips the disk read + parse below. + if previous_files: + prev = previous_files.get(rel_path_str) + if ( + prev is not None + and prev.get("size") == stat.st_size + and prev.get("modified") == modified + ): + file_info = {**prev, "tags": list(prev.get("tags", []))} + files.append(file_info) + for tag in file_info.get("tags", []): + tag_counts[tag] = tag_counts.get(tag, 0) + 1 + reused += 1 + # The global backlink index is rebuilt on every scan, + # so re-register this file's wikilinks from its + # (cached) content instead of re-reading the disk. + if file_info.get("extension") == ".md" and file_info.get("content"): + try: + _extract_wikilinks_for_backlinks( + vault_name, file_info["path"], + file_info.get("title", ""), file_info["content"], + ) + except Exception: + pass + continue + # PDF handling — special path (binary, uses pdf_reader) tags: list[str] = [] pdf_text_pending = False + excalidraw_text_pending = False if ext == ".pdf": from backend.pdf_reader import extract_pdf_metadata # BUG-040: only the (cheap) metadata is read during the @@ -494,10 +543,13 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | No content_preview = "" pdf_text_pending = True elif ext == ".excalidraw" or fpath.name.lower().endswith(".excalidraw.md"): - raw = fpath.read_text(encoding="utf-8", errors="replace") - raw = extract_excalidraw_indexable(raw) + # #86: defer the expensive JSON/lz-string text extraction + # (read + decompress + element walk) to ``enrich_pdf_texts`` + # so the scan stays cheap; title comes from the filename. + raw = "" title = fpath.stem.replace(".excalidraw", "").replace("-", " ").replace("_", " ") - content_preview = raw[:200].strip() + content_preview = "" + excalidraw_text_pending = True else: raw = fpath.read_text(encoding="utf-8", errors="replace") title = fpath.stem.replace("-", " ").replace("_", " ") @@ -528,6 +580,8 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | No } if pdf_text_pending: file_info["pdf_text_pending"] = True + if excalidraw_text_pending: + file_info["excalidraw_text_pending"] = True files.append(file_info) for tag in tags: @@ -540,28 +594,49 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | No logger.error(f"Error indexing {fpath}: {e}") continue - logger.info(f"Vault '{vault_name}': indexed {len(files)} files, {len(paths)} paths, {len(tag_counts)} unique tags") - return {"files": files, "tags": tag_counts, "path": vault_path, "paths": paths, "config": {}} + logger.info( + f"Vault '{vault_name}': indexed {len(files)} files " + f"({reused} reused), {len(paths)} paths, {len(tag_counts)} unique tags" + ) + return {"files": files, "tags": tag_counts, "path": vault_path, "paths": paths, "config": {}, "reused": reused} + + +def _read_excalidraw_indexable_text(file_path: Path) -> str: + """Read an excalidraw file and return its indexable text (blocking helper). + + Runs inside an executor via ``enrich_pdf_texts`` so the lz-string + decompression of large diagrams never blocks the event loop. + """ + try: + raw = file_path.read_text(encoding="utf-8", errors="replace") + except OSError: + return "" + try: + return extract_excalidraw_indexable(raw) + except Exception: # pragma: no cover - defensive + return "" async def enrich_pdf_texts(vault_name: str | None = None) -> int: - """Extract text from PDFs whose extraction was deferred during the scan (BUG-040). + """Extract text deferred during the scan: PDFs (BUG-040) + excalidraw (#86). - ``_scan_vault`` only reads PDF metadata so a vault with many or large PDFs - starts serving immediately. This coroutine runs *after* the index (and the - inverted index) is ready, extracts the missing text off the event loop and - updates the in-memory entry plus the incremental index hooks. + ``_scan_vault`` only reads PDF metadata and excalidraw filenames so a vault + with many or large heavy files starts serving immediately. This coroutine + runs *after* the index (and the inverted index) is ready, extracts the + missing text off the event loop and updates the in-memory entry plus the + incremental index hooks. Args: vault_name: Restrict the pass to a single vault; ``None`` covers every indexed vault. Returns: - Number of deferred PDFs whose text extraction was attempted. + Number of deferred files (PDF + excalidraw) whose text extraction was + attempted. """ from backend.pdf_reader import extract_pdf_text - pending: list[tuple[str, dict[str, Any], Path]] = [] + pending: list[tuple[str, dict[str, Any], Path, str]] = [] with _index_lock: for name, vault_data in index.items(): if vault_name is not None and name != vault_name: @@ -569,32 +644,38 @@ async def enrich_pdf_texts(vault_name: str | None = None) -> int: vault_root = Path(vault_data.get("path", "")) for file_info in vault_data.get("files", []): if file_info.get("pdf_text_pending"): - pending.append((name, file_info, vault_root / file_info["path"])) + pending.append((name, file_info, vault_root / file_info["path"], "pdf")) + elif file_info.get("excalidraw_text_pending"): + pending.append((name, file_info, vault_root / file_info["path"], "excalidraw")) if not pending: return 0 loop = asyncio.get_running_loop() enriched = 0 - for name, file_info, file_path in pending: + for name, file_info, file_path, kind in pending: try: - raw = await loop.run_in_executor(None, extract_pdf_text, file_path, 100000) + if kind == "pdf": + raw = await loop.run_in_executor(None, extract_pdf_text, file_path, 100000) + else: + raw = await loop.run_in_executor(None, _read_excalidraw_indexable_text, file_path) except Exception as exc: # pragma: no cover - defensive - logger.warning("PDF enrichment failed for %s: %s", file_path, exc) + logger.warning("Deferred text enrichment failed for %s: %s", file_path, exc) raw = "" file_info["content"] = raw[:SEARCH_CONTENT_LIMIT] file_info["content_preview"] = raw[:200].strip() file_info.pop("pdf_text_pending", None) + file_info.pop("excalidraw_text_pending", None) enriched += 1 if _on_index_change: try: _on_index_change("add", name, file_info["path"], file_info) except Exception as exc: # pragma: no cover - defensive logger.warning( - "Index hook failed after PDF enrichment for %s: %s", file_path, exc + "Index hook failed after deferred enrichment for %s: %s", file_path, exc ) - logger.info("PDF enrichment: extracted text for %d deferred PDF(s)", enriched) + logger.info("Deferred text enrichment: extracted text for %d file(s)", enriched) return enriched @@ -603,16 +684,24 @@ async def build_index(progress_callback=None) -> None: Runs vault scans concurrently, inserting them incrementally into the global index. Notifies progress via the provided callback. + + #86 differential rebuild: the previous per-vault ``{path: file_info}`` + snapshots are captured before the clear and handed to ``_scan_vault`` so + unchanged files (same size + mtime) are reused without disk re-reads. """ global index, vault_config vault_config.clear() vault_config.update(load_vault_config()) - + # Note: vault_settings are now only used for UI display preferences (hideHiddenFiles) # Indexing always includes all files regardless of settings - + global _index_generation with _index_lock: + previous_snapshot: dict[str, dict[str, dict[str, Any]]] = { + name: {f["path"]: f for f in vdata.get("files", [])} + for name, vdata in index.items() + } index.clear() _file_lookup.clear() path_index.clear() @@ -631,8 +720,13 @@ async def build_index(progress_callback=None) -> None: loop = asyncio.get_event_loop() async def _process_vault(name: str, config: dict[str, Any]): + import functools + vault_path = config["path"] - vault_data = await loop.run_in_executor(None, _scan_vault, name, vault_path, config) + scan = functools.partial( + _scan_vault, name, vault_path, config, previous_snapshot.get(name) + ) + vault_data = await loop.run_in_executor(None, scan) vault_data["config"] = config # Build lookup entries for the new vault @@ -695,7 +789,7 @@ async def reload_index() -> dict[str, Any]: Dict mapping vault names to their file/tag counts. """ await build_index() - # BUG-040: complete the deferred PDF extraction for the rebuilt index. + # BUG-040/#86: complete the deferred PDF + excalidraw extraction. await enrich_pdf_texts() stats = {} for name, data in index.items(): @@ -724,14 +818,22 @@ async def reload_single_vault(vault_name: str) -> dict[str, Any]: raise ValueError(f"Vault '{vault_name}' not found in configuration") config = vault_config[vault_name] - + + # #86 differential rescan: snapshot this vault's entries before removal so + # unchanged files are reused without disk re-reads. + with _index_lock: + _previous = {f["path"]: f for f in index.get(vault_name, {}).get("files", [])} + # Remove old vault data from index structures await remove_vault_from_index(vault_name) - + # Re-add the vault with updated configuration + import functools + vault_path = config["path"] loop = asyncio.get_event_loop() - vault_data = await loop.run_in_executor(None, _scan_vault, vault_name, vault_path, config) + scan = functools.partial(_scan_vault, vault_name, vault_path, config, _previous) + vault_data = await loop.run_in_executor(None, scan) vault_data["config"] = config # Build lookup entries for the vault @@ -761,7 +863,7 @@ async def reload_single_vault(vault_name: str) -> dict[str, Any]: from backend.attachment_indexer import build_attachment_index await build_attachment_index({vault_name: config}) - # BUG-040: complete the deferred PDF extraction for this vault. + # BUG-040/#86: complete the deferred PDF + excalidraw extraction. await enrich_pdf_texts(vault_name) stats = {"file_count": len(vault_data["files"]), "tag_count": len(vault_data["tags"])} diff --git a/backend/services/mutations.py b/backend/services/mutations.py index 425ffb2..f9dba68 100644 --- a/backend/services/mutations.py +++ b/backend/services/mutations.py @@ -26,6 +26,11 @@ from backend.services.vaults import get_vault_root logger = logging.getLogger("obsigate.services.mutations") +# #86: per-file size cap for find/replace passes (CPU guard — complements the +# BUG-025 regex caps). Files larger than this are skipped instead of being +# read fully into memory and scanned with a user-supplied pattern. +MAX_REPLACE_FILE_BYTES = 5_000_000 + # Skeleton injected into empty ``.excalidraw`` files (mirrors the route logic). _EXCALIDRAW_SKELETON = ( '{"type":"excalidraw","version":2,"elements":[],' @@ -509,6 +514,16 @@ def replace_in_files( continue if not file_path.exists() or not file_path.is_file(): continue + # #86 CPU guard: skip files too large to scan safely in one pass. + try: + if file_path.stat().st_size > MAX_REPLACE_FILE_BYTES: + logger.warning( + "replace_in_files: skipping oversized file %s/%s (%d bytes)", + result_vault, result["path"], file_path.stat().st_size, + ) + continue + except OSError: + continue try: original = file_path.read_text(encoding="utf-8", errors="replace") except OSError: diff --git a/desktop/Cargo.lock b/desktop/Cargo.lock index 1522402..073985b 100644 --- a/desktop/Cargo.lock +++ b/desktop/Cargo.lock @@ -2626,7 +2626,7 @@ dependencies = [ [[package]] name = "obsigate-desktop" -version = "2.15.0" +version = "2.16.0" dependencies = [ "chrono", "env_logger", diff --git a/desktop/Cargo.toml b/desktop/Cargo.toml index 89ccd3b..9a17d22 100644 --- a/desktop/Cargo.toml +++ b/desktop/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "obsigate-desktop" -version = "2.15.0" +version = "2.16.0" description = "ObsiGate Desktop — Porte d'entrée native pour vos vaults Obsidian" authors = ["Bruno Charest"] edition = "2021" diff --git a/desktop/tauri.conf.json b/desktop/tauri.conf.json index 8e37def..bd120c1 100644 --- a/desktop/tauri.conf.json +++ b/desktop/tauri.conf.json @@ -1,7 +1,7 @@ { "$schema": "https://raw.githubusercontent.com/nicedoc/obsigate/main/desktop/tauri.conf.schema.json", "productName": "ObsiGate", - "version": "2.15.0", + "version": "2.16.0", "identifier": "com.obsigate.desktop", "build": { "frontendDist": "../frontend", diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index a58a659..ecca184 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -1,6 +1,6 @@ # ObsiGate — Roadmap -> **Version :** 2.15.0 | **Dernière mise à jour :** 2026-09-22 +> **Version :** 2.16.0 | **Dernière mise à jour :** 2026-09-22 > **Ce fichier ne contient que le travail à venir** (🔵 En cours + ⚪ Backlog) et un index compact > vers les fonctionnalités livrées. > - **Méthode de livraison à appliquer pour toute tâche : [DELIVERY_WORKFLOW.md](./DELIVERY_WORKFLOW.md)** @@ -107,15 +107,6 @@ - [ ] Persister index, JTI révoqués et compteurs de rate-limit (SQLite/Redis) - [ ] Verrous asyncio autour de l'index global et des stores JSON ; service de partage public (expiration, révocation, quotas) -### 86. Optimisation globale des performances (phase 3) - -- **Effort :** 4-6 jours | **Impact :** 🟡 | **Zone :** backend (`search.py`, `indexer.py`, `mutations.py`) -- **Description :** brancher l'inverted index (déjà construit) sur la recherche simple et le tool IA `search_fulltext`, indexation incrémentale, extraction PDF lazy. -- **Sous-tâches :** - - [ ] Recherche simple + tool IA via l'inverted index (suppression du balayage O(N) en mémoire) - - [ ] Indexation incrémentale + scan différentiel au démarrage (remplace le `rglob` complet) - - [ ] Extraction PDF/excalidraw différée (hors scan) ; caps CPU sur les opérations regex - ### 87. Amélioration continue — tests, CI/CD, revues de sécurité (phase 4) - **Effort :** 3-5 jours | **Impact :** 🟡 | **Zone :** `.gitea/workflows/`, `tests/` @@ -190,6 +181,7 @@ | 105 | Guide d'utilisation — audit de couverture complet, téléchargement Markdown/PDF, guide desktop élargi, section Architecture (Mermaid) + BUG-067 | 2.13.0 | [features/guide-coverage-105.md](./features/guide-coverage-105.md) | | 106 | Assistant IA — Actions instantanées contextuelles, catalogue « Toutes les actions » & frontmatter complet | 2.14.0 | [features/ai-quick-actions.md](./features/ai-quick-actions.md) | | 107 | Configuration — Gestion des clés API & MCP : création/révocation de jetons longue durée (1 j, 1 mois, 6 mois, 1 an, sans fin), une seule clé pour l'API REST et le serveur MCP, « dernière utilisation », store `data/api_tokens.json` sans secret persisté | 2.15.0 | [features/api-mcp-tokens-107.md](./features/api-mcp-tokens-107.md) | +| 86 | Optimisation globale des performances (phase 3) — scan différentiel, excalidraw différé, garde-fou `replace` (inverted index / PDF lazy / caps regex déjà livrés via BUG-033/040/025) | 2.16.0 | [features/perf-phase3-86.md](./features/perf-phase3-86.md) | --- @@ -197,11 +189,11 @@ | Priorité | Items | Effort total estimé | |---|---|---| -| ✅ Complété | #1 → #59, #61–72, #74–76, #78–84, #88–93, #94–100, #102–107, #92 | ~116 jours réalisés | +| ✅ Complété | #1 → #59, #61–72, #74–76, #78–84, #86, #88–93, #94–100, #102–107, #92 | ~120 jours réalisés | | 🔵 P2 restant | #77 Desktop : signature de code (non retenue), 6 tests E2E **manuels** ([protocole](./DESKTOP_E2E_CHECKLIST.md)) | ~0,5-1 jour | | ⚪ P4 restant | #73 Sync (6-8j) | 6-8 jours | -| ⚪ P0/P1 restant | #85-87 Refonte architecturale, performance, CI/CD (BUG-035 → BUG-040 corrigés) | ~15-23 jours | -| **Total restant** | **7 items + finitions** | **~27-42 jours** | +| ⚪ P0/P1 restant | #85, #87 Refonte architecturale, CI/CD (BUG-035 → BUG-040 corrigés, #86 livré) | ~11-17 jours | +| **Total restant** | **6 items + finitions** | **~23-36 jours** | --- diff --git a/docs/features/perf-phase3-86.md b/docs/features/perf-phase3-86.md new file mode 100644 index 0000000..cfebdf7 --- /dev/null +++ b/docs/features/perf-phase3-86.md @@ -0,0 +1,69 @@ +# #86 — Optimisation globale des performances (phase 3) + +> **Statut :** ✅ livré | **Zone :** backend (`indexer.py`, `mutations.py`) +> **Prérequis déjà livrés :** BUG-033 (recherche via inverted index), BUG-040 (PDF lazy), +> BUG-025 (caps regex). + +## Contexte + +La roadmap #86 demandait : recherche simple + tool IA via l'inverted index, indexation +incrémentale + scan différentiel au démarrage, extraction PDF/excalidraw différée et caps +CPU sur les opérations regex. Trois de ces cinq points étaient déjà couverts par des +correctifs antérieurs ; cette livraison ferme les deux points restants. + +## Ce qui était déjà livré (rappel) + +| Point #86 | Livré par | État | +|---|---|---| +| Recherche simple + `search_fulltext` via inverted index | BUG-033 | `search()` récupère ses candidats via l'inverted index (intersection + expansion de préfixes), repli scan pendant la construction | +| Extraction PDF différée | BUG-040 | `_scan_vault` ne lit que les métadonnées ; `enrich_pdf_texts()` extrait après index | +| Caps CPU regex | BUG-025 | `validate_regex` (longueur ≤ 500, rejet quantificateurs imbriqués), contenu tronqué à 200 kio, matchs plafonnés à 1 000 | + +## Livré ici + +### 1. Scan différentiel au démarrage (`backend/indexer.py`) + +- `_scan_vault(..., previous_files)` : quand un snapshot `{relative_path: file_info}` est + fourni, toute entrée dont `size` **et** `modified` sont inchangés est réutilisée sans + lecture disque ni re-parse (copie du dict, tags recomptés, wikilinks ré-enregistrés + depuis le contenu caché pour reconstruire l'index de backlinks). +- `build_index()` capture le snapshot précédent avant le `clear()` et le transmet à + chaque scan de vault (via `functools.partial` pour l'executor) ; `reload_single_vault()` + fait de même pour son vault. Seuls le `os.walk` + `stat` (bon marché) tournent à + chaque passe ; le retour inclut `reused` (hits différentiels, loggé par vault). +- Premier démarrage (aucun snapshot) : comportement identique à avant. + +### 2. Extraction excalidraw différée (`backend/indexer.py`) + +- Le scan ne lit plus les `.excalidraw` / `.excalidraw.md` : titre dérivé du nom de + fichier, `content` vide, flag `excalidraw_text_pending` (plus de décompression + lz-string pendant le scan). +- `enrich_pdf_texts()` traite désormais les deux flags (`pdf` + `excalidraw`) via + `_read_excalidraw_indexable_text()` exécuté dans l'executor, avec notification du hook + d'index incrémental comme pour les PDF. Nom conservé pour compatibilité (tests, + `main.py`, `reload_index`, `reload_single_vault` inchangés côté appel). +- Le chemin incrémental fichier-à-fichier (`_index_single_file_sync`, watcher) reste + immédiat : un seul fichier ne justifie pas le différé. + +### 3. Garde-fou taille sur `replace_in_files` (`backend/services/mutations.py`) + +- Nouvelle constante `MAX_REPLACE_FILE_BYTES` (5 Mio) : tout fichier dépassant le plafond + est sauté (warning loggé) au lieu d'être lu intégralement puis balayé par le pattern + utilisateur. Complète les caps BUG-025 (qui couvrent la recherche, pas le remplacement + qui opère sur le contenu disque complet par nature). + +## Tests + +`tests/test_perf_phase3.py` (9 tests) : différé excalidraw au scan (`.excalidraw` + +`.excalidraw.md`), remplissage par l'enrichissement, `reused == 0` au premier scan, +réutilisation à l'identique, re-parse du fichier modifié, ajout/suppression, skip +`replace` sur fichier surdimensionné + cas passant nominal. + +## Limites connues + +- Pas de persistance disque de l'index : le différentiel joue sur les rebuilds dans le + même processus (`reload_index`, `reload_single_vault`), pas entre deux redémarrages + (persistance = #85, phase 2). +- La comparaison `size + mtime` ne détecte pas une modification qui conserverait taille + et mtime à la milliseconde près (cas pathologique, le watcher temps réel couvre les + modifications en cours d'exécution). diff --git a/package.json b/package.json index 9783b96..da7b11b 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "obsigate", - "version": "2.15.0", + "version": "2.16.0", "description": "**Porte d'entrée web ultra-léger pour vos vaults Obsidian** — Accédez, naviguez et recherchez dans toutes vos notes Obsidian depuis n'importe quel appareil via une interface web moderne et responsive.", "main": "patch.js", "directories": { diff --git a/tests/test_perf_phase3.py b/tests/test_perf_phase3.py new file mode 100644 index 0000000..abb72ea --- /dev/null +++ b/tests/test_perf_phase3.py @@ -0,0 +1,220 @@ +# tests/test_perf_phase3.py — Optimisation globale des performances (#86, phase 3) +"""Non-regression tests for the #86 performance work. + +Covers the three remaining #86 items (the inverted-index search, the lazy PDF +extraction and the regex CPU caps were already delivered via BUG-033, BUG-040 +and BUG-025): + +- differential vault scan (unchanged files reused without disk re-read), +- deferred excalidraw text extraction (scan cheap, enrichment fills text), +- file-size guard on ``replace_in_files`` (oversized files skipped). +""" +import json +import os +from pathlib import Path + +import pytest + + +def _excalidraw_doc(*texts: str) -> str: + elements = [ + { + "id": f"el{i}", + "type": "text", + "text": text, + "x": 0, + "y": i * 20, + "width": 100, + "height": 20, + } + for i, text in enumerate(texts) + ] + return json.dumps({"type": "excalidraw", "version": 2, "elements": elements}) + + +# ═══════════════════════════════════════════════════════════════════ +# Deferred excalidraw extraction +# ═══════════════════════════════════════════════════════════════════ + +class TestExcalidrawDeferred: + def test_scan_defers_excalidraw_text(self, tmp_path: Path): + """The scan must not parse diagram JSON — content empty + pending flag.""" + from backend.indexer import _scan_vault + + vault = tmp_path / "vault" + vault.mkdir() + (vault / "diagram.excalidraw").write_text( + _excalidraw_doc("hello diagram"), encoding="utf-8" + ) + + result = _scan_vault("V", str(vault), {}) + entry = next(f for f in result["files"] if f["path"] == "diagram.excalidraw") + assert entry["content"] == "" + assert entry["content_preview"] == "" + assert entry.get("excalidraw_text_pending") is True + assert entry["title"] == "diagram" + + def test_scan_defers_excalidraw_md(self, tmp_path: Path): + """``.excalidraw.md`` files are deferred too (no lz decompression at scan).""" + from backend.indexer import _scan_vault + + vault = tmp_path / "vault" + vault.mkdir() + (vault / "board.excalidraw.md").write_text( + "---\ntitle: Board\n---\n" + _excalidraw_doc("board text"), + encoding="utf-8", + ) + + result = _scan_vault("V", str(vault), {}) + entry = next(f for f in result["files"] if f["path"] == "board.excalidraw.md") + assert entry["content"] == "" + assert entry.get("excalidraw_text_pending") is True + + @pytest.mark.asyncio + async def test_enrich_fills_excalidraw_text(self, tmp_path: Path): + """The deferred pass extracts diagram text and clears the flag.""" + import backend.indexer as idx + + vault = tmp_path / "vault" + vault.mkdir() + (vault / "diagram.excalidraw").write_text( + _excalidraw_doc("uniquediagword"), encoding="utf-8" + ) + file_info = { + "path": "diagram.excalidraw", + "title": "diagram", + "tags": [], + "content": "", + "content_preview": "", + "size": 0, + "modified": "", + "extension": ".excalidraw", + "excalidraw_text_pending": True, + } + with idx._index_lock: + idx.index["ExcalV"] = { + "files": [file_info], + "tags": {}, + "path": str(vault), + "paths": [], + } + try: + count = await idx.enrich_pdf_texts("ExcalV") + assert count == 1 + assert "uniquediagword" in file_info["content"] + assert file_info["content_preview"] + assert "excalidraw_text_pending" not in file_info + finally: + with idx._index_lock: + idx.index.pop("ExcalV", None) + + +# ═══════════════════════════════════════════════════════════════════ +# Differential scan +# ═══════════════════════════════════════════════════════════════════ + +class TestDifferentialScan: + def test_first_scan_reports_zero_reused(self, test_vault_dir): + from backend.indexer import _scan_vault + + result = _scan_vault("TestVault", test_vault_dir) + assert result["reused"] == 0 + assert len(result["files"]) >= 3 + + def test_unchanged_files_are_reused(self, test_vault_dir): + from backend.indexer import _scan_vault + + first = _scan_vault("TestVault", test_vault_dir) + previous = {f["path"]: f for f in first["files"]} + second = _scan_vault("TestVault", test_vault_dir, None, previous) + assert second["reused"] == len(first["files"]) + assert {f["path"] for f in second["files"]} == {f["path"] for f in first["files"]} + # Tags and titles survive the reuse path. + assert second["tags"] == first["tags"] + for f in second["files"]: + assert f["content"] == previous[f["path"]]["content"] + + def test_changed_file_is_reparsed(self, test_vault_dir): + from backend.indexer import _scan_vault + + first = _scan_vault("TestVault", test_vault_dir) + previous = {f["path"]: f for f in first["files"]} + + target = Path(test_vault_dir) / "note1.md" + target.write_text( + target.read_text(encoding="utf-8") + "\nMot unique de reparse differentials.", + encoding="utf-8", + ) + # Force a visibly different mtime (coarse filesystems). + os.utime(target, (9999999999, 9999999999)) + + second = _scan_vault("TestVault", test_vault_dir, None, previous) + assert second["reused"] == len(first["files"]) - 1 + changed = next(f for f in second["files"] if f["path"] == "note1.md") + assert "Mot unique de reparse differentials" in changed["content"] + + def test_added_and_deleted_files(self, test_vault_dir): + from backend.indexer import _scan_vault + + first = _scan_vault("TestVault", test_vault_dir) + previous = {f["path"]: f for f in first["files"]} + + (Path(test_vault_dir) / "brand-new.md").write_text( + "# Brand new\nFresh content here.", encoding="utf-8" + ) + (Path(test_vault_dir) / "config.json").unlink() + + second = _scan_vault("TestVault", test_vault_dir, None, previous) + paths = {f["path"] for f in second["files"]} + assert "brand-new.md" in paths + assert "config.json" not in paths + assert second["reused"] == len(first["files"]) - 1 + + +# ═══════════════════════════════════════════════════════════════════ +# replace_in_files size guard +# ═══════════════════════════════════════════════════════════════════ + +class TestReplaceSizeGuard: + def test_oversized_file_is_skipped(self, tmp_path: Path, monkeypatch): + """Files over MAX_REPLACE_FILE_BYTES are skipped, not read.""" + from backend.services import mutations + + big = tmp_path / "big.md" + big.write_text("findme " * 100, encoding="utf-8") + + monkeypatch.setattr(mutations, "MAX_REPLACE_FILE_BYTES", 10) + monkeypatch.setattr( + "backend.services.search.advanced_search_vaults", + lambda *a, **k: { + "results": [{"vault": "V", "path": "big.md", "title": "big"}], + }, + ) + monkeypatch.setattr( + mutations, "get_vault_root", lambda vault: tmp_path + ) + + out = mutations.replace_in_files("findme", "replaced", vault="V", dry_run=True) + assert out["matches"] == [] + assert out["total_matches"] == 0 + + def test_small_file_still_processed(self, tmp_path: Path, monkeypatch): + from backend.services import mutations + + small = tmp_path / "small.md" + small.write_text("findme once", encoding="utf-8") + + monkeypatch.setattr(mutations, "MAX_REPLACE_FILE_BYTES", 10_000_000) + monkeypatch.setattr( + "backend.services.search.advanced_search_vaults", + lambda *a, **k: { + "results": [{"vault": "V", "path": "small.md", "title": "small"}], + }, + ) + monkeypatch.setattr( + mutations, "get_vault_root", lambda vault: tmp_path + ) + + out = mutations.replace_in_files("findme", "replaced", vault="V", dry_run=True) + assert out["total_matches"] == 1 + assert out["matches"][0]["path"] == "small.md"