feat: optimisation globale des performances #86 (scan differentiel, excalidraw differe, garde-fou replace; inverted index/PDF lazy/caps regex deja livres via BUG-033/040/025)
This commit is contained in:
+18
-1
@@ -6,7 +6,7 @@ Format basé sur [Keep a Changelog](https://keepachangelog.com/fr/1.1.0/),
|
||||
et [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
> **En cours de développement** : les changements à venir sont listés dans la section
|
||||
> [Unreleased](#unreleased). La dernière version livrée est **2.15.0**.
|
||||
> [Unreleased](#unreleased). La dernière version livrée est **2.16.0**.
|
||||
|
||||
---
|
||||
|
||||
@@ -14,6 +14,23 @@ et [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
---
|
||||
|
||||
## [2.16.0] — 2026-09-22
|
||||
|
||||
### Modifié
|
||||
|
||||
- **#86 - Optimisation globale des performances (phase 3)** : ferme les deux derniers
|
||||
points de la phase 3 (recherche via inverted index, PDF lazy et caps regex déjà livrés
|
||||
via BUG-033/BUG-040/BUG-025). Scan **différentiel** : `_scan_vault` réutilise les
|
||||
entrées inchangées (`size` + `modified`) d'un snapshot précédent — seuls `os.walk` +
|
||||
`stat` tournent à chaque rebuild (`build_index`, `reload_single_vault`). Extraction
|
||||
**excalidraw différée** : le scan ne lit plus les `.excalidraw` / `.excalidraw.md`
|
||||
(flag `excalidraw_text_pending`), `enrich_pdf_texts()` extrait leur texte après index
|
||||
comme pour les PDF. Garde-fou `MAX_REPLACE_FILE_BYTES` (5 Mio) sur `replace_in_files`.
|
||||
Tests : `tests/test_perf_phase3.py` (9). Voir
|
||||
[docs/features/perf-phase3-86.md](./docs/features/perf-phase3-86.md).
|
||||
|
||||
---
|
||||
|
||||
## [2.15.0] — 2026-09-22
|
||||
|
||||
### Ajouté
|
||||
|
||||
+3
-3
@@ -4,7 +4,7 @@
|
||||
|
||||
**Porte d'entrée web ultra-léger pour vos vaults Obsidian** — Accédez, naviguez et recherchez dans toutes vos notes Obsidian depuis n'importe quel appareil via une interface web moderne et responsive.
|
||||
|
||||
[]()
|
||||
[]()
|
||||
[](https://opensource.org/licenses/MIT)
|
||||
[](https://www.docker.com/)
|
||||
[](https://www.python.org/)
|
||||
@@ -927,8 +927,8 @@ Ce projet est sous licence **MIT** — voir le fichier [LICENSE](LICENSE) pour l
|
||||
|
||||
## 📝 Changelog
|
||||
|
||||
Consultez le [CHANGELOG.md](./CHANGELOG.md) pour l'historique complet de toutes les versions (v1.0.0 → v2.15.0).
|
||||
Consultez le [CHANGELOG.md](./CHANGELOG.md) pour l'historique complet de toutes les versions (v1.0.0 → v2.16.0).
|
||||
|
||||
---
|
||||
|
||||
*Projet : ObsiGate | Version : 2.15.0 | Dernière mise à jour : Juin 2026*
|
||||
*Projet : ObsiGate | Version : 2.16.0 | Dernière mise à jour : Juin 2026*
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
**Ultra-light web gateway for your Obsidian vaults** — Access, browse, and search all your Obsidian notes from any device via a modern, responsive web interface.
|
||||
|
||||
[]()
|
||||
[]()
|
||||
[](https://opensource.org/licenses/MIT)
|
||||
[](https://www.docker.com/)
|
||||
[](https://www.python.org/)
|
||||
@@ -1096,8 +1096,8 @@ This project is licensed under the **MIT License** - see the [LICENSE](LICENSE)
|
||||
|
||||
## 📝 Changelog
|
||||
|
||||
See [CHANGELOG.md](./CHANGELOG.md) for the complete version history (v1.0.0 → v2.15.0).
|
||||
See [CHANGELOG.md](./CHANGELOG.md) for the complete version history (v1.0.0 → v2.16.0).
|
||||
|
||||
---
|
||||
|
||||
*Project: ObsiGate | Version: 2.15.0 | Last updated: May 2026*
|
||||
*Project: ObsiGate | Version: 2.16.0 | Last updated: May 2026*
|
||||
|
||||
+126
-24
@@ -397,7 +397,12 @@ def parse_markdown_file(raw: str) -> frontmatter.Post:
|
||||
return frontmatter.Post(content)
|
||||
|
||||
|
||||
def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | None = None) -> dict[str, Any]:
|
||||
def _scan_vault(
|
||||
vault_name: str,
|
||||
vault_path: str,
|
||||
vault_cfg: dict[str, Any] | None = None,
|
||||
previous_files: dict[str, dict[str, Any]] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Synchronously scan a single vault directory and build file index.
|
||||
|
||||
Walks the vault tree, reads supported files, extracts metadata
|
||||
@@ -406,22 +411,38 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | No
|
||||
|
||||
All files and directories are indexed, including hidden files (starting with '.').
|
||||
|
||||
Differential scan (#86): when ``previous_files`` maps a relative path to
|
||||
its previous ``file_info`` dict, entries whose ``size`` and ``modified``
|
||||
timestamp are unchanged are reused verbatim (no disk read, no re-parse).
|
||||
Only the cheap ``os.walk`` + ``stat`` runs on every pass; heavy content
|
||||
extraction (PDF metadata excepted — always cheap) is skipped for
|
||||
unchanged files. This replaces the full ``rglob`` re-read on rebuilds.
|
||||
|
||||
Excalidraw diagrams (#86, like PDFs since BUG-040) are deferred: the scan
|
||||
only records the title and sets ``excalidraw_text_pending``; the expensive
|
||||
JSON/lz-string text extraction runs in ``enrich_pdf_texts()`` after the
|
||||
index is queryable.
|
||||
|
||||
Args:
|
||||
vault_name: Display name of the vault.
|
||||
vault_path: Absolute filesystem path to the vault root.
|
||||
vault_cfg: Optional vault configuration dict (unused for indexing, kept for compatibility).
|
||||
previous_files: Optional ``{relative_path: file_info}`` snapshot from a
|
||||
previous scan used for differential reuse.
|
||||
|
||||
Returns:
|
||||
Dict with keys ``files`` (list), ``tags`` (counter dict), ``path`` (str), ``paths`` (list).
|
||||
Dict with keys ``files`` (list), ``tags`` (counter dict), ``path`` (str),
|
||||
``paths`` (list) and ``reused`` (int, differential hits).
|
||||
"""
|
||||
vault_root = Path(vault_path)
|
||||
files: list[dict[str, Any]] = []
|
||||
tag_counts: dict[str, int] = {}
|
||||
paths: list[dict[str, str]] = []
|
||||
reused = 0
|
||||
|
||||
if not vault_root.exists():
|
||||
logger.warning(f"Vault path does not exist: {vault_path}")
|
||||
return {"files": [], "tags": {}, "path": vault_path, "paths": []}
|
||||
return {"files": [], "tags": {}, "path": vault_path, "paths": [], "reused": 0}
|
||||
|
||||
root_resolved = vault_root.resolve(strict=False)
|
||||
|
||||
@@ -479,9 +500,37 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | No
|
||||
stat = fpath.stat()
|
||||
modified = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc).isoformat()
|
||||
|
||||
# #86 differential scan: reuse the previous entry when neither
|
||||
# size nor mtime changed — skips the disk read + parse below.
|
||||
if previous_files:
|
||||
prev = previous_files.get(rel_path_str)
|
||||
if (
|
||||
prev is not None
|
||||
and prev.get("size") == stat.st_size
|
||||
and prev.get("modified") == modified
|
||||
):
|
||||
file_info = {**prev, "tags": list(prev.get("tags", []))}
|
||||
files.append(file_info)
|
||||
for tag in file_info.get("tags", []):
|
||||
tag_counts[tag] = tag_counts.get(tag, 0) + 1
|
||||
reused += 1
|
||||
# The global backlink index is rebuilt on every scan,
|
||||
# so re-register this file's wikilinks from its
|
||||
# (cached) content instead of re-reading the disk.
|
||||
if file_info.get("extension") == ".md" and file_info.get("content"):
|
||||
try:
|
||||
_extract_wikilinks_for_backlinks(
|
||||
vault_name, file_info["path"],
|
||||
file_info.get("title", ""), file_info["content"],
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
continue
|
||||
|
||||
# PDF handling — special path (binary, uses pdf_reader)
|
||||
tags: list[str] = []
|
||||
pdf_text_pending = False
|
||||
excalidraw_text_pending = False
|
||||
if ext == ".pdf":
|
||||
from backend.pdf_reader import extract_pdf_metadata
|
||||
# BUG-040: only the (cheap) metadata is read during the
|
||||
@@ -494,10 +543,13 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | No
|
||||
content_preview = ""
|
||||
pdf_text_pending = True
|
||||
elif ext == ".excalidraw" or fpath.name.lower().endswith(".excalidraw.md"):
|
||||
raw = fpath.read_text(encoding="utf-8", errors="replace")
|
||||
raw = extract_excalidraw_indexable(raw)
|
||||
# #86: defer the expensive JSON/lz-string text extraction
|
||||
# (read + decompress + element walk) to ``enrich_pdf_texts``
|
||||
# so the scan stays cheap; title comes from the filename.
|
||||
raw = ""
|
||||
title = fpath.stem.replace(".excalidraw", "").replace("-", " ").replace("_", " ")
|
||||
content_preview = raw[:200].strip()
|
||||
content_preview = ""
|
||||
excalidraw_text_pending = True
|
||||
else:
|
||||
raw = fpath.read_text(encoding="utf-8", errors="replace")
|
||||
title = fpath.stem.replace("-", " ").replace("_", " ")
|
||||
@@ -528,6 +580,8 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | No
|
||||
}
|
||||
if pdf_text_pending:
|
||||
file_info["pdf_text_pending"] = True
|
||||
if excalidraw_text_pending:
|
||||
file_info["excalidraw_text_pending"] = True
|
||||
files.append(file_info)
|
||||
|
||||
for tag in tags:
|
||||
@@ -540,28 +594,49 @@ def _scan_vault(vault_name: str, vault_path: str, vault_cfg: dict[str, Any] | No
|
||||
logger.error(f"Error indexing {fpath}: {e}")
|
||||
continue
|
||||
|
||||
logger.info(f"Vault '{vault_name}': indexed {len(files)} files, {len(paths)} paths, {len(tag_counts)} unique tags")
|
||||
return {"files": files, "tags": tag_counts, "path": vault_path, "paths": paths, "config": {}}
|
||||
logger.info(
|
||||
f"Vault '{vault_name}': indexed {len(files)} files "
|
||||
f"({reused} reused), {len(paths)} paths, {len(tag_counts)} unique tags"
|
||||
)
|
||||
return {"files": files, "tags": tag_counts, "path": vault_path, "paths": paths, "config": {}, "reused": reused}
|
||||
|
||||
|
||||
def _read_excalidraw_indexable_text(file_path: Path) -> str:
|
||||
"""Read an excalidraw file and return its indexable text (blocking helper).
|
||||
|
||||
Runs inside an executor via ``enrich_pdf_texts`` so the lz-string
|
||||
decompression of large diagrams never blocks the event loop.
|
||||
"""
|
||||
try:
|
||||
raw = file_path.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return ""
|
||||
try:
|
||||
return extract_excalidraw_indexable(raw)
|
||||
except Exception: # pragma: no cover - defensive
|
||||
return ""
|
||||
|
||||
|
||||
async def enrich_pdf_texts(vault_name: str | None = None) -> int:
|
||||
"""Extract text from PDFs whose extraction was deferred during the scan (BUG-040).
|
||||
"""Extract text deferred during the scan: PDFs (BUG-040) + excalidraw (#86).
|
||||
|
||||
``_scan_vault`` only reads PDF metadata so a vault with many or large PDFs
|
||||
starts serving immediately. This coroutine runs *after* the index (and the
|
||||
inverted index) is ready, extracts the missing text off the event loop and
|
||||
updates the in-memory entry plus the incremental index hooks.
|
||||
``_scan_vault`` only reads PDF metadata and excalidraw filenames so a vault
|
||||
with many or large heavy files starts serving immediately. This coroutine
|
||||
runs *after* the index (and the inverted index) is ready, extracts the
|
||||
missing text off the event loop and updates the in-memory entry plus the
|
||||
incremental index hooks.
|
||||
|
||||
Args:
|
||||
vault_name: Restrict the pass to a single vault; ``None`` covers every
|
||||
indexed vault.
|
||||
|
||||
Returns:
|
||||
Number of deferred PDFs whose text extraction was attempted.
|
||||
Number of deferred files (PDF + excalidraw) whose text extraction was
|
||||
attempted.
|
||||
"""
|
||||
from backend.pdf_reader import extract_pdf_text
|
||||
|
||||
pending: list[tuple[str, dict[str, Any], Path]] = []
|
||||
pending: list[tuple[str, dict[str, Any], Path, str]] = []
|
||||
with _index_lock:
|
||||
for name, vault_data in index.items():
|
||||
if vault_name is not None and name != vault_name:
|
||||
@@ -569,32 +644,38 @@ async def enrich_pdf_texts(vault_name: str | None = None) -> int:
|
||||
vault_root = Path(vault_data.get("path", ""))
|
||||
for file_info in vault_data.get("files", []):
|
||||
if file_info.get("pdf_text_pending"):
|
||||
pending.append((name, file_info, vault_root / file_info["path"]))
|
||||
pending.append((name, file_info, vault_root / file_info["path"], "pdf"))
|
||||
elif file_info.get("excalidraw_text_pending"):
|
||||
pending.append((name, file_info, vault_root / file_info["path"], "excalidraw"))
|
||||
|
||||
if not pending:
|
||||
return 0
|
||||
|
||||
loop = asyncio.get_running_loop()
|
||||
enriched = 0
|
||||
for name, file_info, file_path in pending:
|
||||
for name, file_info, file_path, kind in pending:
|
||||
try:
|
||||
if kind == "pdf":
|
||||
raw = await loop.run_in_executor(None, extract_pdf_text, file_path, 100000)
|
||||
else:
|
||||
raw = await loop.run_in_executor(None, _read_excalidraw_indexable_text, file_path)
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.warning("PDF enrichment failed for %s: %s", file_path, exc)
|
||||
logger.warning("Deferred text enrichment failed for %s: %s", file_path, exc)
|
||||
raw = ""
|
||||
file_info["content"] = raw[:SEARCH_CONTENT_LIMIT]
|
||||
file_info["content_preview"] = raw[:200].strip()
|
||||
file_info.pop("pdf_text_pending", None)
|
||||
file_info.pop("excalidraw_text_pending", None)
|
||||
enriched += 1
|
||||
if _on_index_change:
|
||||
try:
|
||||
_on_index_change("add", name, file_info["path"], file_info)
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.warning(
|
||||
"Index hook failed after PDF enrichment for %s: %s", file_path, exc
|
||||
"Index hook failed after deferred enrichment for %s: %s", file_path, exc
|
||||
)
|
||||
|
||||
logger.info("PDF enrichment: extracted text for %d deferred PDF(s)", enriched)
|
||||
logger.info("Deferred text enrichment: extracted text for %d file(s)", enriched)
|
||||
return enriched
|
||||
|
||||
|
||||
@@ -603,6 +684,10 @@ async def build_index(progress_callback=None) -> None:
|
||||
|
||||
Runs vault scans concurrently, inserting them incrementally into the global index.
|
||||
Notifies progress via the provided callback.
|
||||
|
||||
#86 differential rebuild: the previous per-vault ``{path: file_info}``
|
||||
snapshots are captured before the clear and handed to ``_scan_vault`` so
|
||||
unchanged files (same size + mtime) are reused without disk re-reads.
|
||||
"""
|
||||
global index, vault_config
|
||||
vault_config.clear()
|
||||
@@ -613,6 +698,10 @@ async def build_index(progress_callback=None) -> None:
|
||||
|
||||
global _index_generation
|
||||
with _index_lock:
|
||||
previous_snapshot: dict[str, dict[str, dict[str, Any]]] = {
|
||||
name: {f["path"]: f for f in vdata.get("files", [])}
|
||||
for name, vdata in index.items()
|
||||
}
|
||||
index.clear()
|
||||
_file_lookup.clear()
|
||||
path_index.clear()
|
||||
@@ -631,8 +720,13 @@ async def build_index(progress_callback=None) -> None:
|
||||
loop = asyncio.get_event_loop()
|
||||
|
||||
async def _process_vault(name: str, config: dict[str, Any]):
|
||||
import functools
|
||||
|
||||
vault_path = config["path"]
|
||||
vault_data = await loop.run_in_executor(None, _scan_vault, name, vault_path, config)
|
||||
scan = functools.partial(
|
||||
_scan_vault, name, vault_path, config, previous_snapshot.get(name)
|
||||
)
|
||||
vault_data = await loop.run_in_executor(None, scan)
|
||||
vault_data["config"] = config
|
||||
|
||||
# Build lookup entries for the new vault
|
||||
@@ -695,7 +789,7 @@ async def reload_index() -> dict[str, Any]:
|
||||
Dict mapping vault names to their file/tag counts.
|
||||
"""
|
||||
await build_index()
|
||||
# BUG-040: complete the deferred PDF extraction for the rebuilt index.
|
||||
# BUG-040/#86: complete the deferred PDF + excalidraw extraction.
|
||||
await enrich_pdf_texts()
|
||||
stats = {}
|
||||
for name, data in index.items():
|
||||
@@ -725,13 +819,21 @@ async def reload_single_vault(vault_name: str) -> dict[str, Any]:
|
||||
|
||||
config = vault_config[vault_name]
|
||||
|
||||
# #86 differential rescan: snapshot this vault's entries before removal so
|
||||
# unchanged files are reused without disk re-reads.
|
||||
with _index_lock:
|
||||
_previous = {f["path"]: f for f in index.get(vault_name, {}).get("files", [])}
|
||||
|
||||
# Remove old vault data from index structures
|
||||
await remove_vault_from_index(vault_name)
|
||||
|
||||
# Re-add the vault with updated configuration
|
||||
import functools
|
||||
|
||||
vault_path = config["path"]
|
||||
loop = asyncio.get_event_loop()
|
||||
vault_data = await loop.run_in_executor(None, _scan_vault, vault_name, vault_path, config)
|
||||
scan = functools.partial(_scan_vault, vault_name, vault_path, config, _previous)
|
||||
vault_data = await loop.run_in_executor(None, scan)
|
||||
vault_data["config"] = config
|
||||
|
||||
# Build lookup entries for the vault
|
||||
@@ -761,7 +863,7 @@ async def reload_single_vault(vault_name: str) -> dict[str, Any]:
|
||||
from backend.attachment_indexer import build_attachment_index
|
||||
await build_attachment_index({vault_name: config})
|
||||
|
||||
# BUG-040: complete the deferred PDF extraction for this vault.
|
||||
# BUG-040/#86: complete the deferred PDF + excalidraw extraction.
|
||||
await enrich_pdf_texts(vault_name)
|
||||
|
||||
stats = {"file_count": len(vault_data["files"]), "tag_count": len(vault_data["tags"])}
|
||||
|
||||
@@ -26,6 +26,11 @@ from backend.services.vaults import get_vault_root
|
||||
|
||||
logger = logging.getLogger("obsigate.services.mutations")
|
||||
|
||||
# #86: per-file size cap for find/replace passes (CPU guard — complements the
|
||||
# BUG-025 regex caps). Files larger than this are skipped instead of being
|
||||
# read fully into memory and scanned with a user-supplied pattern.
|
||||
MAX_REPLACE_FILE_BYTES = 5_000_000
|
||||
|
||||
# Skeleton injected into empty ``.excalidraw`` files (mirrors the route logic).
|
||||
_EXCALIDRAW_SKELETON = (
|
||||
'{"type":"excalidraw","version":2,"elements":[],'
|
||||
@@ -509,6 +514,16 @@ def replace_in_files(
|
||||
continue
|
||||
if not file_path.exists() or not file_path.is_file():
|
||||
continue
|
||||
# #86 CPU guard: skip files too large to scan safely in one pass.
|
||||
try:
|
||||
if file_path.stat().st_size > MAX_REPLACE_FILE_BYTES:
|
||||
logger.warning(
|
||||
"replace_in_files: skipping oversized file %s/%s (%d bytes)",
|
||||
result_vault, result["path"], file_path.stat().st_size,
|
||||
)
|
||||
continue
|
||||
except OSError:
|
||||
continue
|
||||
try:
|
||||
original = file_path.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
|
||||
Generated
+1
-1
@@ -2626,7 +2626,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "obsigate-desktop"
|
||||
version = "2.15.0"
|
||||
version = "2.16.0"
|
||||
dependencies = [
|
||||
"chrono",
|
||||
"env_logger",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "obsigate-desktop"
|
||||
version = "2.15.0"
|
||||
version = "2.16.0"
|
||||
description = "ObsiGate Desktop — Porte d'entrée native pour vos vaults Obsidian"
|
||||
authors = ["Bruno Charest"]
|
||||
edition = "2021"
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"$schema": "https://raw.githubusercontent.com/nicedoc/obsigate/main/desktop/tauri.conf.schema.json",
|
||||
"productName": "ObsiGate",
|
||||
"version": "2.15.0",
|
||||
"version": "2.16.0",
|
||||
"identifier": "com.obsigate.desktop",
|
||||
"build": {
|
||||
"frontendDist": "../frontend",
|
||||
|
||||
+5
-13
@@ -1,6 +1,6 @@
|
||||
# ObsiGate — Roadmap
|
||||
|
||||
> **Version :** 2.15.0 | **Dernière mise à jour :** 2026-09-22
|
||||
> **Version :** 2.16.0 | **Dernière mise à jour :** 2026-09-22
|
||||
> **Ce fichier ne contient que le travail à venir** (🔵 En cours + ⚪ Backlog) et un index compact
|
||||
> vers les fonctionnalités livrées.
|
||||
> - **Méthode de livraison à appliquer pour toute tâche : [DELIVERY_WORKFLOW.md](./DELIVERY_WORKFLOW.md)**
|
||||
@@ -107,15 +107,6 @@
|
||||
- [ ] Persister index, JTI révoqués et compteurs de rate-limit (SQLite/Redis)
|
||||
- [ ] Verrous asyncio autour de l'index global et des stores JSON ; service de partage public (expiration, révocation, quotas)
|
||||
|
||||
### 86. Optimisation globale des performances (phase 3)
|
||||
|
||||
- **Effort :** 4-6 jours | **Impact :** 🟡 | **Zone :** backend (`search.py`, `indexer.py`, `mutations.py`)
|
||||
- **Description :** brancher l'inverted index (déjà construit) sur la recherche simple et le tool IA `search_fulltext`, indexation incrémentale, extraction PDF lazy.
|
||||
- **Sous-tâches :**
|
||||
- [ ] Recherche simple + tool IA via l'inverted index (suppression du balayage O(N) en mémoire)
|
||||
- [ ] Indexation incrémentale + scan différentiel au démarrage (remplace le `rglob` complet)
|
||||
- [ ] Extraction PDF/excalidraw différée (hors scan) ; caps CPU sur les opérations regex
|
||||
|
||||
### 87. Amélioration continue — tests, CI/CD, revues de sécurité (phase 4)
|
||||
|
||||
- **Effort :** 3-5 jours | **Impact :** 🟡 | **Zone :** `.gitea/workflows/`, `tests/`
|
||||
@@ -190,6 +181,7 @@
|
||||
| 105 | Guide d'utilisation — audit de couverture complet, téléchargement Markdown/PDF, guide desktop élargi, section Architecture (Mermaid) + BUG-067 | 2.13.0 | [features/guide-coverage-105.md](./features/guide-coverage-105.md) |
|
||||
| 106 | Assistant IA — Actions instantanées contextuelles, catalogue « Toutes les actions » & frontmatter complet | 2.14.0 | [features/ai-quick-actions.md](./features/ai-quick-actions.md) |
|
||||
| 107 | Configuration — Gestion des clés API & MCP : création/révocation de jetons longue durée (1 j, 1 mois, 6 mois, 1 an, sans fin), une seule clé pour l'API REST et le serveur MCP, « dernière utilisation », store `data/api_tokens.json` sans secret persisté | 2.15.0 | [features/api-mcp-tokens-107.md](./features/api-mcp-tokens-107.md) |
|
||||
| 86 | Optimisation globale des performances (phase 3) — scan différentiel, excalidraw différé, garde-fou `replace` (inverted index / PDF lazy / caps regex déjà livrés via BUG-033/040/025) | 2.16.0 | [features/perf-phase3-86.md](./features/perf-phase3-86.md) |
|
||||
|
||||
---
|
||||
|
||||
@@ -197,11 +189,11 @@
|
||||
|
||||
| Priorité | Items | Effort total estimé |
|
||||
|---|---|---|
|
||||
| ✅ Complété | #1 → #59, #61–72, #74–76, #78–84, #88–93, #94–100, #102–107, #92 | ~116 jours réalisés |
|
||||
| ✅ Complété | #1 → #59, #61–72, #74–76, #78–84, #86, #88–93, #94–100, #102–107, #92 | ~120 jours réalisés |
|
||||
| 🔵 P2 restant | #77 Desktop : signature de code (non retenue), 6 tests E2E **manuels** ([protocole](./DESKTOP_E2E_CHECKLIST.md)) | ~0,5-1 jour |
|
||||
| ⚪ P4 restant | #73 Sync (6-8j) | 6-8 jours |
|
||||
| ⚪ P0/P1 restant | #85-87 Refonte architecturale, performance, CI/CD (BUG-035 → BUG-040 corrigés) | ~15-23 jours |
|
||||
| **Total restant** | **7 items + finitions** | **~27-42 jours** |
|
||||
| ⚪ P0/P1 restant | #85, #87 Refonte architecturale, CI/CD (BUG-035 → BUG-040 corrigés, #86 livré) | ~11-17 jours |
|
||||
| **Total restant** | **6 items + finitions** | **~23-36 jours** |
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
# #86 — Optimisation globale des performances (phase 3)
|
||||
|
||||
> **Statut :** ✅ livré | **Zone :** backend (`indexer.py`, `mutations.py`)
|
||||
> **Prérequis déjà livrés :** BUG-033 (recherche via inverted index), BUG-040 (PDF lazy),
|
||||
> BUG-025 (caps regex).
|
||||
|
||||
## Contexte
|
||||
|
||||
La roadmap #86 demandait : recherche simple + tool IA via l'inverted index, indexation
|
||||
incrémentale + scan différentiel au démarrage, extraction PDF/excalidraw différée et caps
|
||||
CPU sur les opérations regex. Trois de ces cinq points étaient déjà couverts par des
|
||||
correctifs antérieurs ; cette livraison ferme les deux points restants.
|
||||
|
||||
## Ce qui était déjà livré (rappel)
|
||||
|
||||
| Point #86 | Livré par | État |
|
||||
|---|---|---|
|
||||
| Recherche simple + `search_fulltext` via inverted index | BUG-033 | `search()` récupère ses candidats via l'inverted index (intersection + expansion de préfixes), repli scan pendant la construction |
|
||||
| Extraction PDF différée | BUG-040 | `_scan_vault` ne lit que les métadonnées ; `enrich_pdf_texts()` extrait après index |
|
||||
| Caps CPU regex | BUG-025 | `validate_regex` (longueur ≤ 500, rejet quantificateurs imbriqués), contenu tronqué à 200 kio, matchs plafonnés à 1 000 |
|
||||
|
||||
## Livré ici
|
||||
|
||||
### 1. Scan différentiel au démarrage (`backend/indexer.py`)
|
||||
|
||||
- `_scan_vault(..., previous_files)` : quand un snapshot `{relative_path: file_info}` est
|
||||
fourni, toute entrée dont `size` **et** `modified` sont inchangés est réutilisée sans
|
||||
lecture disque ni re-parse (copie du dict, tags recomptés, wikilinks ré-enregistrés
|
||||
depuis le contenu caché pour reconstruire l'index de backlinks).
|
||||
- `build_index()` capture le snapshot précédent avant le `clear()` et le transmet à
|
||||
chaque scan de vault (via `functools.partial` pour l'executor) ; `reload_single_vault()`
|
||||
fait de même pour son vault. Seuls le `os.walk` + `stat` (bon marché) tournent à
|
||||
chaque passe ; le retour inclut `reused` (hits différentiels, loggé par vault).
|
||||
- Premier démarrage (aucun snapshot) : comportement identique à avant.
|
||||
|
||||
### 2. Extraction excalidraw différée (`backend/indexer.py`)
|
||||
|
||||
- Le scan ne lit plus les `.excalidraw` / `.excalidraw.md` : titre dérivé du nom de
|
||||
fichier, `content` vide, flag `excalidraw_text_pending` (plus de décompression
|
||||
lz-string pendant le scan).
|
||||
- `enrich_pdf_texts()` traite désormais les deux flags (`pdf` + `excalidraw`) via
|
||||
`_read_excalidraw_indexable_text()` exécuté dans l'executor, avec notification du hook
|
||||
d'index incrémental comme pour les PDF. Nom conservé pour compatibilité (tests,
|
||||
`main.py`, `reload_index`, `reload_single_vault` inchangés côté appel).
|
||||
- Le chemin incrémental fichier-à-fichier (`_index_single_file_sync`, watcher) reste
|
||||
immédiat : un seul fichier ne justifie pas le différé.
|
||||
|
||||
### 3. Garde-fou taille sur `replace_in_files` (`backend/services/mutations.py`)
|
||||
|
||||
- Nouvelle constante `MAX_REPLACE_FILE_BYTES` (5 Mio) : tout fichier dépassant le plafond
|
||||
est sauté (warning loggé) au lieu d'être lu intégralement puis balayé par le pattern
|
||||
utilisateur. Complète les caps BUG-025 (qui couvrent la recherche, pas le remplacement
|
||||
qui opère sur le contenu disque complet par nature).
|
||||
|
||||
## Tests
|
||||
|
||||
`tests/test_perf_phase3.py` (9 tests) : différé excalidraw au scan (`.excalidraw` +
|
||||
`.excalidraw.md`), remplissage par l'enrichissement, `reused == 0` au premier scan,
|
||||
réutilisation à l'identique, re-parse du fichier modifié, ajout/suppression, skip
|
||||
`replace` sur fichier surdimensionné + cas passant nominal.
|
||||
|
||||
## Limites connues
|
||||
|
||||
- Pas de persistance disque de l'index : le différentiel joue sur les rebuilds dans le
|
||||
même processus (`reload_index`, `reload_single_vault`), pas entre deux redémarrages
|
||||
(persistance = #85, phase 2).
|
||||
- La comparaison `size + mtime` ne détecte pas une modification qui conserverait taille
|
||||
et mtime à la milliseconde près (cas pathologique, le watcher temps réel couvre les
|
||||
modifications en cours d'exécution).
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "obsigate",
|
||||
"version": "2.15.0",
|
||||
"version": "2.16.0",
|
||||
"description": "**Porte d'entrée web ultra-léger pour vos vaults Obsidian** — Accédez, naviguez et recherchez dans toutes vos notes Obsidian depuis n'importe quel appareil via une interface web moderne et responsive.",
|
||||
"main": "patch.js",
|
||||
"directories": {
|
||||
|
||||
@@ -0,0 +1,220 @@
|
||||
# tests/test_perf_phase3.py — Optimisation globale des performances (#86, phase 3)
|
||||
"""Non-regression tests for the #86 performance work.
|
||||
|
||||
Covers the three remaining #86 items (the inverted-index search, the lazy PDF
|
||||
extraction and the regex CPU caps were already delivered via BUG-033, BUG-040
|
||||
and BUG-025):
|
||||
|
||||
- differential vault scan (unchanged files reused without disk re-read),
|
||||
- deferred excalidraw text extraction (scan cheap, enrichment fills text),
|
||||
- file-size guard on ``replace_in_files`` (oversized files skipped).
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _excalidraw_doc(*texts: str) -> str:
|
||||
elements = [
|
||||
{
|
||||
"id": f"el{i}",
|
||||
"type": "text",
|
||||
"text": text,
|
||||
"x": 0,
|
||||
"y": i * 20,
|
||||
"width": 100,
|
||||
"height": 20,
|
||||
}
|
||||
for i, text in enumerate(texts)
|
||||
]
|
||||
return json.dumps({"type": "excalidraw", "version": 2, "elements": elements})
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════
|
||||
# Deferred excalidraw extraction
|
||||
# ═══════════════════════════════════════════════════════════════════
|
||||
|
||||
class TestExcalidrawDeferred:
|
||||
def test_scan_defers_excalidraw_text(self, tmp_path: Path):
|
||||
"""The scan must not parse diagram JSON — content empty + pending flag."""
|
||||
from backend.indexer import _scan_vault
|
||||
|
||||
vault = tmp_path / "vault"
|
||||
vault.mkdir()
|
||||
(vault / "diagram.excalidraw").write_text(
|
||||
_excalidraw_doc("hello diagram"), encoding="utf-8"
|
||||
)
|
||||
|
||||
result = _scan_vault("V", str(vault), {})
|
||||
entry = next(f for f in result["files"] if f["path"] == "diagram.excalidraw")
|
||||
assert entry["content"] == ""
|
||||
assert entry["content_preview"] == ""
|
||||
assert entry.get("excalidraw_text_pending") is True
|
||||
assert entry["title"] == "diagram"
|
||||
|
||||
def test_scan_defers_excalidraw_md(self, tmp_path: Path):
|
||||
"""``.excalidraw.md`` files are deferred too (no lz decompression at scan)."""
|
||||
from backend.indexer import _scan_vault
|
||||
|
||||
vault = tmp_path / "vault"
|
||||
vault.mkdir()
|
||||
(vault / "board.excalidraw.md").write_text(
|
||||
"---\ntitle: Board\n---\n" + _excalidraw_doc("board text"),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
result = _scan_vault("V", str(vault), {})
|
||||
entry = next(f for f in result["files"] if f["path"] == "board.excalidraw.md")
|
||||
assert entry["content"] == ""
|
||||
assert entry.get("excalidraw_text_pending") is True
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_enrich_fills_excalidraw_text(self, tmp_path: Path):
|
||||
"""The deferred pass extracts diagram text and clears the flag."""
|
||||
import backend.indexer as idx
|
||||
|
||||
vault = tmp_path / "vault"
|
||||
vault.mkdir()
|
||||
(vault / "diagram.excalidraw").write_text(
|
||||
_excalidraw_doc("uniquediagword"), encoding="utf-8"
|
||||
)
|
||||
file_info = {
|
||||
"path": "diagram.excalidraw",
|
||||
"title": "diagram",
|
||||
"tags": [],
|
||||
"content": "",
|
||||
"content_preview": "",
|
||||
"size": 0,
|
||||
"modified": "",
|
||||
"extension": ".excalidraw",
|
||||
"excalidraw_text_pending": True,
|
||||
}
|
||||
with idx._index_lock:
|
||||
idx.index["ExcalV"] = {
|
||||
"files": [file_info],
|
||||
"tags": {},
|
||||
"path": str(vault),
|
||||
"paths": [],
|
||||
}
|
||||
try:
|
||||
count = await idx.enrich_pdf_texts("ExcalV")
|
||||
assert count == 1
|
||||
assert "uniquediagword" in file_info["content"]
|
||||
assert file_info["content_preview"]
|
||||
assert "excalidraw_text_pending" not in file_info
|
||||
finally:
|
||||
with idx._index_lock:
|
||||
idx.index.pop("ExcalV", None)
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════
|
||||
# Differential scan
|
||||
# ═══════════════════════════════════════════════════════════════════
|
||||
|
||||
class TestDifferentialScan:
|
||||
def test_first_scan_reports_zero_reused(self, test_vault_dir):
|
||||
from backend.indexer import _scan_vault
|
||||
|
||||
result = _scan_vault("TestVault", test_vault_dir)
|
||||
assert result["reused"] == 0
|
||||
assert len(result["files"]) >= 3
|
||||
|
||||
def test_unchanged_files_are_reused(self, test_vault_dir):
|
||||
from backend.indexer import _scan_vault
|
||||
|
||||
first = _scan_vault("TestVault", test_vault_dir)
|
||||
previous = {f["path"]: f for f in first["files"]}
|
||||
second = _scan_vault("TestVault", test_vault_dir, None, previous)
|
||||
assert second["reused"] == len(first["files"])
|
||||
assert {f["path"] for f in second["files"]} == {f["path"] for f in first["files"]}
|
||||
# Tags and titles survive the reuse path.
|
||||
assert second["tags"] == first["tags"]
|
||||
for f in second["files"]:
|
||||
assert f["content"] == previous[f["path"]]["content"]
|
||||
|
||||
def test_changed_file_is_reparsed(self, test_vault_dir):
|
||||
from backend.indexer import _scan_vault
|
||||
|
||||
first = _scan_vault("TestVault", test_vault_dir)
|
||||
previous = {f["path"]: f for f in first["files"]}
|
||||
|
||||
target = Path(test_vault_dir) / "note1.md"
|
||||
target.write_text(
|
||||
target.read_text(encoding="utf-8") + "\nMot unique de reparse differentials.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
# Force a visibly different mtime (coarse filesystems).
|
||||
os.utime(target, (9999999999, 9999999999))
|
||||
|
||||
second = _scan_vault("TestVault", test_vault_dir, None, previous)
|
||||
assert second["reused"] == len(first["files"]) - 1
|
||||
changed = next(f for f in second["files"] if f["path"] == "note1.md")
|
||||
assert "Mot unique de reparse differentials" in changed["content"]
|
||||
|
||||
def test_added_and_deleted_files(self, test_vault_dir):
|
||||
from backend.indexer import _scan_vault
|
||||
|
||||
first = _scan_vault("TestVault", test_vault_dir)
|
||||
previous = {f["path"]: f for f in first["files"]}
|
||||
|
||||
(Path(test_vault_dir) / "brand-new.md").write_text(
|
||||
"# Brand new\nFresh content here.", encoding="utf-8"
|
||||
)
|
||||
(Path(test_vault_dir) / "config.json").unlink()
|
||||
|
||||
second = _scan_vault("TestVault", test_vault_dir, None, previous)
|
||||
paths = {f["path"] for f in second["files"]}
|
||||
assert "brand-new.md" in paths
|
||||
assert "config.json" not in paths
|
||||
assert second["reused"] == len(first["files"]) - 1
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════
|
||||
# replace_in_files size guard
|
||||
# ═══════════════════════════════════════════════════════════════════
|
||||
|
||||
class TestReplaceSizeGuard:
|
||||
def test_oversized_file_is_skipped(self, tmp_path: Path, monkeypatch):
|
||||
"""Files over MAX_REPLACE_FILE_BYTES are skipped, not read."""
|
||||
from backend.services import mutations
|
||||
|
||||
big = tmp_path / "big.md"
|
||||
big.write_text("findme " * 100, encoding="utf-8")
|
||||
|
||||
monkeypatch.setattr(mutations, "MAX_REPLACE_FILE_BYTES", 10)
|
||||
monkeypatch.setattr(
|
||||
"backend.services.search.advanced_search_vaults",
|
||||
lambda *a, **k: {
|
||||
"results": [{"vault": "V", "path": "big.md", "title": "big"}],
|
||||
},
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
mutations, "get_vault_root", lambda vault: tmp_path
|
||||
)
|
||||
|
||||
out = mutations.replace_in_files("findme", "replaced", vault="V", dry_run=True)
|
||||
assert out["matches"] == []
|
||||
assert out["total_matches"] == 0
|
||||
|
||||
def test_small_file_still_processed(self, tmp_path: Path, monkeypatch):
|
||||
from backend.services import mutations
|
||||
|
||||
small = tmp_path / "small.md"
|
||||
small.write_text("findme once", encoding="utf-8")
|
||||
|
||||
monkeypatch.setattr(mutations, "MAX_REPLACE_FILE_BYTES", 10_000_000)
|
||||
monkeypatch.setattr(
|
||||
"backend.services.search.advanced_search_vaults",
|
||||
lambda *a, **k: {
|
||||
"results": [{"vault": "V", "path": "small.md", "title": "small"}],
|
||||
},
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
mutations, "get_vault_root", lambda vault: tmp_path
|
||||
)
|
||||
|
||||
out = mutations.replace_in_files("findme", "replaced", vault="V", dry_run=True)
|
||||
assert out["total_matches"] == 1
|
||||
assert out["matches"][0]["path"] == "small.md"
|
||||
Reference in New Issue
Block a user