From 31d4be80155603680d058f570335d485fd8fbcae Mon Sep 17 00:00:00 2001 From: Bruno Charest Date: Sat, 12 Sep 2026 00:12:20 -0400 Subject: [PATCH] feat(search): recherche semantique - embeddings vectoriels + hybride RRF (#70) --- .gitea/workflows/ci.yml | 6 +- CHANGELOG.md | 16 + README.fr.md | 9 +- README.md | 8 +- backend/main.py | 12 +- backend/requirements-semantic.txt | 18 + backend/search.py | 225 ++++++--- backend/semantic_search.py | 612 ++++++++++++++++++++++++ backend/services/search.py | 7 +- docs/ROADMAP.md | 30 +- docs/features/semantic-search.md | 128 +++++ frontend/index.html | 1 + frontend/js/search.js | 26 +- frontend/js/state.js | 2 + frontend/locales/en.json | 3 + frontend/locales/fr.json | 3 + tests/conftest.py | 7 + tests/frontend/semantic-search.test.mjs | 122 +++++ tests/test_semantic_search.py | 243 ++++++++++ 19 files changed, 1388 insertions(+), 90 deletions(-) create mode 100644 backend/requirements-semantic.txt create mode 100644 backend/semantic_search.py create mode 100644 docs/features/semantic-search.md create mode 100644 tests/frontend/semantic-search.test.mjs create mode 100644 tests/test_semantic_search.py diff --git a/.gitea/workflows/ci.yml b/.gitea/workflows/ci.yml index d757ddc..bfad483 100644 --- a/.gitea/workflows/ci.yml +++ b/.gitea/workflows/ci.yml @@ -38,7 +38,7 @@ jobs: - name: Frontend unit tests run: node tests/frontend/unit.test.mjs - - name: Frontend JSDOM tests (PaneManager + Excalidraw + Plugins + AI + SW + Collab + Mobile) + - name: Frontend JSDOM tests (PaneManager + Excalidraw + Plugins + AI + SW + Collab + Mobile + Semantic) run: | cd tests/frontend if [ -d node_modules ]; then @@ -49,8 +49,9 @@ jobs: node sw.test.mjs node collab.test.mjs node mobile-editor.test.mjs + node semantic-search.test.mjs else - echo "tests/frontend/node_modules missing — installing jsdom" + echo "tests/frontend/node_modules missing - installing jsdom" npm install --no-audit --no-fund --silent node pane-manager.test.mjs node excalidraw-viewer.test.mjs @@ -59,6 +60,7 @@ jobs: node sw.test.mjs node collab.test.mjs node mobile-editor.test.mjs + node semantic-search.test.mjs fi # ── Tests ───────────────────────────────────────────────────────── diff --git a/CHANGELOG.md b/CHANGELOG.md index 337175a..52c7f53 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,22 @@ et [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Ajouté +- **#70 Recherche sémantique — Embeddings vectoriels** — la recherche comprend désormais le + **sens** de la requête en plus des mots-clés. **Embeddings** : chaque document est découpé en + chunks de 512 mots (recouvrement 64) puis vectorisé (384 dim) via `all-MiniLM-L6-v2` + (`sentence-transformers`), un endpoint `/embeddings` compatible OpenAI, ou un provider de repli + **sans dépendance** (hachage déterministe). **Stockage vectoriel** : `numpy`/`faiss` + (`IndexFlatIP`) si disponibles, sinon cosinus pur Python. **Recherche hybride** : fusion du + classement TF-IDF et du classement sémantique par **RRF** (Reciprocal Rank Fusion), avec + `semantic_score` par résultat. **Indexation incrémentale** branchée sur le watcher (un fichier + modifié régénère son embedding). **UI** : toggle « Recherche sémantique » (`~`, raccourci + `Alt+S`) dans la barre de résultats + affichage du score de similarité, clés i18n FR/EN. + Nouveau module `backend/semantic_search.py`, paramètre `semantic` sur + `/api/search/advanced`, dépendances **optionnelles** dans + `backend/requirements-semantic.txt`. Tests : `tests/test_semantic_search.py` (27) + + `tests/frontend/semantic-search.test.mjs` (4). Détail : + [docs/features/semantic-search.md](./docs/features/semantic-search.md). + - **#69 Éditeur mobile natif — Interface tactile optimisée** — refonte de l'expérience d'édition sur téléphone/tablette. **Barre d'outils flottante** dans l'éditeur (gras, italique, code, liste à puces, lien) opérant directement sur la sélection CodeMirror (ou le textarea de diff --git a/README.fr.md b/README.fr.md index b28c5b2..ae56f4c 100644 --- a/README.fr.md +++ b/README.fr.md @@ -61,7 +61,7 @@ - **🗺️ Vue graphe interactive** — Canvas force-directed avec Barnes-Hut O(n log n), filtres (tag, type), profondeur, mode focus, historique de navigation ←→↑, export PNG, aperçu au survol (Ctrl+click) - **🗂️ Multi-vault** : Visualisez plusieurs vaults Obsidian simultanément - **🌳 Navigation arborescente** : Parcourez vos dossiers et fichiers dans la sidebar -- **🔍 Recherche avancée** : Moteur TF-IDF avec stemming français, normalisation des accents, snippets surlignés, facettes, pagination et tri +- **🔍 Recherche avancée** : Moteur TF-IDF avec stemming français, normalisation des accents, snippets surlignés, facettes, pagination et tri — plus une **recherche sémantique** optionnelle (embeddings `all-MiniLM-L6-v2`, fusion hybride TF-IDF + RRF) activable via le toggle `~` ([détail](docs/features/semantic-search.md)) - **💡 Autocomplétion intelligente** : Suggestions de fichiers, tags et historique avec navigation clavier - **🧩 Syntaxe de requête** : Opérateurs `tag:`, `#`, `vault:`, `title:`, `path:`, `ext:` avec chips visuels - **📜 Historique de recherche** : Persisté en localStorage (max 50 entrées, LIFO, dédupliqué) @@ -595,7 +595,7 @@ ObsiGate expose une API REST complète : | `/api/file/{vault}/download?path=` | Téléchargement d'un fichier | GET | Oui | | `/api/file/{vault}/save?path=` | Sauvegarder un fichier | PUT | Oui | | `/api/file/{vault}?path=` | Supprimer un fichier | DELETE | Oui | -| `/api/search/advanced` | Recherche avancée TF-IDF | GET | Oui | +| `/api/search/advanced` | Recherche avancée TF-IDF (+ `semantic=true` pour l'hybride) | GET | Oui | | `/api/suggest` / `/api/tags/suggest` | Autocomplétion | GET | Oui | | `/api/tags?vault=` | Tags uniques avec compteurs | GET | Oui | | `/api/index/reload` | Force un re-scan des vaults | GET | Admin | @@ -669,6 +669,11 @@ Les fichiers créés avec le **plugin Obsidian Excalidraw** (y compris le format - **Boost titre** : correspondances dans le titre ×3 - **Normalisation des accents** : `resume` trouve `résumé` - **Snippets surlignés** (``), **facettes** (compteurs par vault/tag), **pagination** (50/page), **tri** pertinence/date, **chips** de filtres, **historique** (50 recherches) +- **Recherche sémantique** (optionnelle) : le toggle `~` (ou `Alt+S`) fusionne le classement + TF-IDF avec un classement par embeddings (RRF). Fonctionne sans dépendance avec un provider de + hachage ; installez `backend/requirements-semantic.txt` et/ou renseignez `OBSIGATE_EMBEDDING_*` + pour de vrais embeddings `all-MiniLM-L6-v2`. Voir + [docs/features/semantic-search.md](docs/features/semantic-search.md). --- diff --git a/README.md b/README.md index 746e572..d774c2c 100644 --- a/README.md +++ b/README.md @@ -54,7 +54,7 @@ - **🗺️ Interactive Graph View** — Canvas force-directed with Barnes-Hut O(n log n), filters (tag, type), depth, focus mode, navigation history ←→↑, export PNG, preview on hover (Ctrl+click) - **🗂️ Multi-vault** : View multiple Obsidian vaults simultaneously - **🌳 Tree Navigation** : Browse your folders and files in the sidebar -- **🔍 Advanced Search** : TF-IDF search engine with French stemming, accent normalization, highlighted snippets, facets, pagination, and sorting +- **🔍 Advanced Search** : TF-IDF search engine with French stemming, accent normalization, highlighted snippets, facets, pagination, and sorting — plus an optional **semantic search** (embeddings via `all-MiniLM-L6-v2`, hybrid TF-IDF + RRF fusion) toggled with `~` ([details](docs/features/semantic-search.md)) - **💡 Smart Autocomplete** : Suggestions for files, tags, and history with keyboard navigation - **🧩 Query Syntax** : Operators `tag:`, `#`, `vault:`, `title:`, `path:`, `ext:` with visual chips - **📜 Search History** : Persisted in localStorage (max 50 entries, LIFO, deduplicated) @@ -708,7 +708,7 @@ ObsiGate exposes a complete REST API : | `/api/file/{vault}/download?path=` | Download a file | GET | Yes | | `/api/file/{vault}/save?path=` | Save a file | PUT | Yes | | `/api/file/{vault}?path=` | Delete a file | DELETE | Yes | -| `/api/search/advanced` | Advanced TF-IDF search | GET | Yes | +| `/api/search/advanced` | Advanced TF-IDF search (+ `semantic=true` for hybrid) | GET | Yes | | `/api/suggest` / `/api/tags/suggest` | Autocomplete | GET | Yes | | `/api/tags?vault=` | Unique tags with counters | GET | Yes | | `/api/index/reload` | Force a rescan of vaults | GET | Admin | @@ -802,6 +802,10 @@ Operators are combinable: `tag:linux vault:IT ext:md server web` searches for "s - **Sorting** : By relevance (TF-IDF) or modification date - **Visual chips** : Active filters are shown as removable colored chips - **History** : Last 50 searches are stored in localStorage +- **Semantic search** (optional) : Toggle `~` (or `Alt+S`) fuses the TF-IDF ranking with an + embedding-based ranking (RRF). Works out of the box with a dependency-free hashing embedder; + install `backend/requirements-semantic.txt` and/or set `OBSIGATE_EMBEDDING_*` for real + `all-MiniLM-L6-v2` embeddings. See [docs/features/semantic-search.md](docs/features/semantic-search.md). --- diff --git a/backend/main.py b/backend/main.py index 2629cb2..60ab37d 100644 --- a/backend/main.py +++ b/backend/main.py @@ -98,6 +98,7 @@ from backend.search import ( suggest_tags, suggest_titles, ) +from backend.semantic_search import init_semantic_index from backend.services.backups import diff_backup as service_diff_backup from backend.services.backups import get_backup_dir as service_get_backup_dir from backend.services.backups import list_backup_files as service_list_backup_files @@ -281,7 +282,8 @@ class AdvancedSearchResultItem(BaseModel): path: str = Field(description="Relative file path") title: str = Field(description="File title") tags: list[str] = Field(description="File tags") - score: float = Field(description="TF-IDF relevance score") + score: float = Field(description="TF-IDF relevance score (or fused RRF score in semantic mode)") + semantic_score: float = Field(default=0.0, description="Cosine similarity from the semantic index (0 when unavailable)") snippet: str = Field(description="Content excerpt with highlights") modified: str = Field(description="ISO 8601 modification timestamp") extension: str = Field(default="", description="File extension") @@ -301,6 +303,7 @@ class AdvancedSearchResponse(BaseModel): limit: int = Field(description="Page size") facets: SearchFacets = Field(description="Faceted counts by tag and vault") query_time_ms: float = Field(default=0, description="Server-side query time in milliseconds") + semantic_available: bool = Field(default=False, description="True when the semantic (embedding) index is ready") class TitleSuggestion(BaseModel): @@ -694,6 +697,8 @@ async def lifespan(app: FastAPI): # would freeze HTTP responses if run in the async event loop. loop = asyncio.get_running_loop() await loop.run_in_executor(_search_executor, init_inverted_index) + # Build the semantic (embedding) index in the same background thread pool. + await loop.run_in_executor(_search_executor, init_semantic_index) # Scan for plugins in all vaults logger.info("Scanning for plugins...") @@ -2589,6 +2594,7 @@ async def api_advanced_search( created: str | None = Query(None, description="Created date filter (>date, ``-highlighted snippets and faceted tag/vault counts. """ @@ -2615,7 +2623,7 @@ async def api_advanced_search( limit=limit, offset=offset, sort=sort, case_sensitive=case_sensitive, whole_word=whole_word, regex=regex, include_paths=include_paths, exclude_paths=exclude_paths, - created=created, modified=modified, size=size), + created=created, modified=modified, size=size, semantic=semantic), ) diff --git a/backend/requirements-semantic.txt b/backend/requirements-semantic.txt new file mode 100644 index 0000000..8d6e436 --- /dev/null +++ b/backend/requirements-semantic.txt @@ -0,0 +1,18 @@ +# ObsiGate — Optional dependencies for semantic search (#70) +# +# These are NOT required: the semantic search engine degrades gracefully to a +# dependency-free hashing embedder and a pure-Python cosine store when they are +# absent. Install this file to enable the full local model + fast vector index: +# +# pip install -r backend/requirements-semantic.txt +# +# NOTE: sentence-transformers pulls in PyTorch (large download). If you only +# want the vector acceleration, install numpy + faiss-cpu and configure an +# external embedding endpoint instead (OBSIGATE_EMBEDDING_*). + +# Local embedding model (all-MiniLM-L6-v2, ~80 MB, CPU) +sentence-transformers>=2.2.0 + +# Vector storage / similarity search +numpy>=1.24.0 +faiss-cpu>=1.7.4 diff --git a/backend/search.py b/backend/search.py index 9f8f6a9..954c401 100644 --- a/backend/search.py +++ b/backend/search.py @@ -4,12 +4,14 @@ import re import time import unicodedata from collections import defaultdict +from collections.abc import Callable from typing import Any from snowballstemmer import stemmer as _snowball_stemmer from sortedcontainers import SortedList from backend import indexer as _indexer +from backend import semantic_search as _semantic from backend.indexer import index logger = logging.getLogger("obsigate.search") @@ -655,6 +657,10 @@ def _on_index_change_hook(action: str, vault_name: str, path: str, file_info: di inv.remove_document(vault_name, path) except Exception as e: logger.warning(f"Inverted index incremental update failed ({action} {vault_name}/{path}): {e}") + try: + _semantic.on_index_change(action, vault_name, path, file_info) + except Exception as e: + logger.warning(f"Semantic index incremental update failed ({action} {vault_name}/{path}): {e}") # Register the hook with indexer (indexer is already imported at top of file) @@ -1102,6 +1108,7 @@ def advanced_search( created: str | None = None, modified: str | None = None, size: str | None = None, + semantic: bool = False, ) -> dict[str, Any]: """Advanced full-text search with TF-IDF scoring, facets, and pagination. @@ -1121,10 +1128,13 @@ def advanced_search( limit: Max results per page. offset: Pagination offset. sort_by: ``"relevance"`` or ``"modified"``. + semantic: When True, fuse the TF-IDF ranking with the semantic + (embedding) ranking via Reciprocal Rank Fusion and expose a + ``semantic_score`` per result. Returns: Dict with ``results``, ``total``, ``offset``, ``limit``, ``facets``, - ``query_time_ms``. + ``query_time_ms`` and ``semantic_available``. """ t0 = time.monotonic() query = query.strip() if query else "" @@ -1182,58 +1192,63 @@ def advanced_search( # ------------------------------------------------------------------ # Step 2: Apply filters on candidate set # ------------------------------------------------------------------ - if effective_vault != "all": - candidates &= inv.vault_docs.get(effective_vault, set()) - - if all_tags and has_terms: - for t in all_tags: - candidates &= inv.tag_docs.get(t.lower(), set()) - - if parsed["title"]: - norm_title_filter = normalize_text(parsed["title"]) - candidates = { - dk for dk in candidates - if norm_title_filter in normalize_text(inv.doc_info[dk].get("title", "")) - } - - if parsed["path"]: - norm_path_filter = normalize_text(parsed["path"]) - candidates = { - dk for dk in candidates - if norm_path_filter in normalize_text(inv.doc_info[dk].get("path", "")) - } - - if parsed["ext"]: - ext_filter = parsed["ext"] - candidates = { - dk for dk in candidates - if ( - inv.doc_info[dk].get("path", "").rsplit("/", 1)[-1].lower() == ext_filter - or inv.doc_info[dk].get("path", "").rsplit("/", 1)[-1].lower().endswith(f".{ext_filter}") - ) - } - - # Date and size filters (from query operators or API params) date_range_created = _parse_date_range(created or parsed.get("created")) - if date_range_created: - candidates = { - dk for dk in candidates - if _matches_date_range(inv.doc_info[dk].get("created"), date_range_created) - } - date_range_modified = _parse_date_range(modified or parsed.get("modified")) - if date_range_modified: - candidates = { - dk for dk in candidates - if _matches_date_range(inv.doc_info[dk].get("modified"), date_range_modified) - } - size_range = _parse_size_range(size or parsed.get("size")) - if size_range: - candidates = { - dk for dk in candidates - if _matches_size_range(inv.doc_info[dk].get("size", 0), size_range) - } + + def _apply_metadata_filters(docs: set) -> set: + """Restrict a document-key set to the query's metadata filters.""" + if effective_vault != "all": + docs &= inv.vault_docs.get(effective_vault, set()) + + if all_tags: + for t in all_tags: + docs &= inv.tag_docs.get(t.lower(), set()) + + if parsed["title"]: + norm_title_filter = normalize_text(parsed["title"]) + docs = { + dk for dk in docs + if norm_title_filter in normalize_text(inv.doc_info[dk].get("title", "")) + } + + if parsed["path"]: + norm_path_filter = normalize_text(parsed["path"]) + docs = { + dk for dk in docs + if norm_path_filter in normalize_text(inv.doc_info[dk].get("path", "")) + } + + if parsed["ext"]: + ext_filter = parsed["ext"] + docs = { + dk for dk in docs + if ( + inv.doc_info[dk].get("path", "").rsplit("/", 1)[-1].lower() == ext_filter + or inv.doc_info[dk].get("path", "").rsplit("/", 1)[-1].lower().endswith(f".{ext_filter}") + ) + } + + if date_range_created: + docs = { + dk for dk in docs + if _matches_date_range(inv.doc_info[dk].get("created"), date_range_created) + } + + if date_range_modified: + docs = { + dk for dk in docs + if _matches_date_range(inv.doc_info[dk].get("modified"), date_range_modified) + } + + if size_range: + docs = { + dk for dk in docs + if _matches_size_range(inv.doc_info[dk].get("size", 0), size_range) + } + return docs + + candidates = _apply_metadata_filters(candidates) # ------------------------------------------------------------------ # Step 3: Score only the candidates (not all N documents) @@ -1312,16 +1327,22 @@ def advanced_search( "title": file_info["title"], "tags": file_info.get("tags", []), "score": round(score, 4), + "semantic_score": 0.0, "snippet": snippet, "modified": file_info.get("modified", ""), "extension": file_info.get("extension", file_info.get("path", "").rsplit(".", 1)[-1] if "." in file_info.get("path", "") else ""), } scored_results.append((score, result)) - # Facets - facet_vaults[vault_name] = facet_vaults.get(vault_name, 0) + 1 - for tag in file_info.get("tags", []): - facet_tags[tag] = facet_tags.get(tag, 0) + 1 + # ------------------------------------------------------------------ + # Step 4: Optional semantic fusion (RRF with the TF-IDF ranking) + # ------------------------------------------------------------------ + semantic_available = _semantic.get_semantic_index().is_ready() + if semantic and has_terms and not regex: + scored_results = _fuse_semantic_results( + scored_results, query, effective_vault, inv, _apply_metadata_filters, + include_paths, exclude_paths, limit, + ) # Sort if sort_by == "modified": @@ -1329,6 +1350,12 @@ def advanced_search( else: scored_results.sort(key=lambda x: -x[0]) + # Facets are recomputed from the final result set (covers semantic-only docs) + for _, result in scored_results: + facet_vaults[result["vault"]] = facet_vaults.get(result["vault"], 0) + 1 + for tag in result.get("tags", []): + facet_tags[tag] = facet_tags.get(tag, 0) + 1 + total = len(scored_results) page = scored_results[offset: offset + limit] elapsed_ms = round((time.monotonic() - t0) * 1000, 1) @@ -1343,9 +1370,99 @@ def advanced_search( "vaults": dict(sorted(facet_vaults.items(), key=lambda x: -x[1])), }, "query_time_ms": elapsed_ms, + "semantic_available": semantic_available, } +def _fuse_semantic_results( + scored_results: list[tuple[float, dict[str, Any]]], + query: str, + vault_filter: str, + inv: InvertedIndex, + apply_metadata_filters: Callable[[set], set], + include_paths: str | None, + exclude_paths: str | None, + limit: int, +) -> list[tuple[float, dict[str, Any]]]: + """Fuse the TF-IDF ranking with the semantic ranking using RRF. + + Documents found only by the semantic engine are materialized from the + inverted index metadata (with a plain, non-highlighted snippet). The + returned tuples carry the fused score, and every result dict gets a + ``semantic_score`` (cosine similarity, 0.0 when absent). + + Args: + scored_results: Existing ``(tfidf_score, result_dict)`` tuples. + query: Raw free-text query. + vault_filter: Effective vault filter. + inv: Inverted index (document metadata source). + apply_metadata_filters: Callable restricting a doc-key set to the + query's tag/title/path/ext/date/size filters. + include_paths: Include glob patterns (or None). + exclude_paths: Exclude glob patterns (or None). + limit: Requested page size (drives how many semantic hits to fetch). + + Returns: + New ``(fused_score, result_dict)`` list (unsorted). + """ + sem_hits = _semantic.semantic_search_docs(query, vault_filter=vault_filter, top_k=max(limit * 5, 200)) + if not sem_hits: + return scored_results + + # Semantic candidates must satisfy the same metadata + path filters. + semantic_universe = apply_metadata_filters(set(inv.doc_info.keys())) + semantic_universe = { + dk for dk in semantic_universe + if _passes_path_filters(inv.doc_info[dk].get("path", ""), include_paths, exclude_paths) + } + + sem_scores: dict[str, float] = {} + sem_ranked: list[str] = [] + for doc_key, similarity in sem_hits: + if doc_key not in semantic_universe: + continue + sem_scores[doc_key] = similarity + sem_ranked.append(doc_key) + + if not sem_ranked: + return scored_results + + existing: dict[str, dict[str, Any]] = {} + lexical_ranked: list[str] = [] + for _, lex_result in sorted(scored_results, key=lambda item: -item[0]): + key = f"{lex_result['vault']}::{lex_result['path']}" + existing[key] = lex_result + lexical_ranked.append(key) + + fused = _semantic.rrf_fuse([lexical_ranked, sem_ranked]) + + merged: list[tuple[float, dict[str, Any]]] = [] + for doc_key, fused_score in fused.items(): + result = existing.get(doc_key) + if result is None: + file_info = inv.doc_info.get(doc_key) + if file_info is None: + continue + content = file_info.get("content", "") + result = { + "vault": inv.doc_vault[doc_key], + "path": file_info["path"], + "title": file_info["title"], + "tags": file_info.get("tags", []), + "score": 0.0, + "semantic_score": 0.0, + "snippet": _escape_html(content[:200].strip()) if content else "", + "modified": file_info.get("modified", ""), + "extension": file_info.get( + "extension", + file_info.get("path", "").rsplit(".", 1)[-1] if "." in file_info.get("path", "") else "", + ), + } + result["semantic_score"] = round(sem_scores.get(doc_key, 0.0), 4) + merged.append((fused_score, result)) + return merged + + # --------------------------------------------------------------------------- # Suggestion helpers # --------------------------------------------------------------------------- diff --git a/backend/semantic_search.py b/backend/semantic_search.py new file mode 100644 index 0000000..7bfceb3 --- /dev/null +++ b/backend/semantic_search.py @@ -0,0 +1,612 @@ +"""ObsiGate — Semantic search: embeddings, vector store and hybrid retrieval. + +This module adds a *semantic* layer on top of the existing TF-IDF search. Each +document is split into overlapping chunks, each chunk is converted into a dense +vector, and queries are matched by cosine similarity. Results are combined with +the lexical ranking through Reciprocal Rank Fusion (RRF). + +Design goals +------------ +* **Zero mandatory dependency.** ``sentence-transformers`` (local model), + ``numpy`` and ``faiss`` are *optional*. They are imported lazily and, when + missing, the module falls back to a deterministic pure-Python hashing embedder + and a pure-Python cosine store. The feature therefore degrades gracefully and + the default CI (which only installs ``backend/requirements.txt``) keeps working. +* **Plug-in providers.** Embeddings can come from the local + ``all-MiniLM-L6-v2`` model, from an OpenAI-compatible ``/embeddings`` endpoint + (configured via env vars), or from the deterministic fallback. +* **Incremental.** The index is updated document-by-document from the indexer + change hook (file watcher + API mutations), never rebuilt on each search. + +Optional extras are listed in ``backend/requirements-semantic.txt``. +""" + +from __future__ import annotations + +import hashlib +import logging +import math +import os +import re +import threading +from abc import ABC, abstractmethod +from collections import Counter +from itertools import pairwise +from typing import Any + +logger = logging.getLogger("obsigate.semantic") + +# --------------------------------------------------------------------------- +# Constants +# --------------------------------------------------------------------------- +EMBEDDING_DIM = 384 # all-MiniLM-L6-v2 output dimension +CHUNK_TOKENS = 512 # target chunk size (whitespace tokens) +CHUNK_OVERLAP_TOKENS = 64 # overlap between consecutive chunks +DEFAULT_TOP_K = 200 # max documents returned by a semantic query +RRF_K = 60 # Reciprocal Rank Fusion smoothing constant +MAX_QUERY_CHARS = 2000 # guard against pathological queries + +_WORD_RE = re.compile(r"[\w]+", re.UNICODE) + + +# --------------------------------------------------------------------------- +# Tokenization / chunking +# --------------------------------------------------------------------------- +def _simple_tokens(text: str) -> list[str]: + """Split *text* into lowercase word tokens (keeps accents).""" + return _WORD_RE.findall(text.lower()) + + +def chunk_text( + text: str, + chunk_tokens: int = CHUNK_TOKENS, + overlap: int = CHUNK_OVERLAP_TOKENS, +) -> list[str]: + """Split *text* into overlapping windows of roughly *chunk_tokens* words. + + Args: + text: Raw document text. + chunk_tokens: Target number of whitespace tokens per chunk. + overlap: Number of tokens shared by two consecutive chunks. + + Returns: + A list of chunk strings. Empty input yields an empty list. + """ + if not text or not text.strip(): + return [] + if chunk_tokens <= 0: + chunk_tokens = CHUNK_TOKENS + overlap = max(0, min(overlap, chunk_tokens - 1)) + + words = text.split() + if len(words) <= chunk_tokens: + return [" ".join(words)] + + step = max(1, chunk_tokens - overlap) + chunks: list[str] = [] + for start in range(0, len(words), step): + window = words[start:start + chunk_tokens] + if not window: + break + chunks.append(" ".join(window)) + if start + chunk_tokens >= len(words): + break + return chunks + + +# --------------------------------------------------------------------------- +# Embedding providers +# --------------------------------------------------------------------------- +class EmbeddingProvider(ABC): + """Base class for embedding backends.""" + + name: str = "base" + + def __init__(self, dimension: int = EMBEDDING_DIM) -> None: + self.dimension = dimension + + @abstractmethod + def encode(self, texts: list[str]) -> list[list[float]]: + """Return one L2-normalized vector per input text.""" + + def encode_one(self, text: str) -> list[float]: + """Convenience wrapper returning the vector for a single text.""" + vectors = self.encode([text]) + return vectors[0] if vectors else [0.0] * self.dimension + + +class HashEmbeddingProvider(EmbeddingProvider): + """Deterministic, dependency-free hashing embedder. + + This is a *lexical* fallback: it hashes word unigrams, word bigrams and + character trigrams into fixed-size signed buckets (the "hashing trick"), + then L2-normalizes the result. It captures shared vocabulary and + morphological variants (``backup``/``backups``), so it already improves + recall over exact TF-IDF matching, but it does not understand synonyms the + way a real transformer model does. + """ + + name = "hash" + + def _add_feature(self, vec: list[float], key: str, weight: float) -> None: + digest = hashlib.blake2b(key.encode("utf-8"), digest_size=8).digest() + h = int.from_bytes(digest, "big") + idx = h % self.dimension + sign = 1.0 if (h >> 63) & 1 else -1.0 + vec[idx] += sign * weight + + def _encode_one(self, text: str) -> list[float]: + vec = [0.0] * self.dimension + tokens = _simple_tokens(text) + if not tokens: + return vec + + tf = Counter(tokens) + for token, count in tf.items(): + weight = 1.0 + math.log(count) + self._add_feature(vec, "w:" + token, weight) + for gram in _char_ngrams(token, 3): + self._add_feature(vec, "g:" + gram, weight * 0.5) + + for first, second in pairwise(tokens): + self._add_feature(vec, "b:" + first + "_" + second, 0.5) + + norm = math.sqrt(sum(v * v for v in vec)) + if norm > 0.0: + vec = [v / norm for v in vec] + return vec + + def encode(self, texts: list[str]) -> list[list[float]]: + return [self._encode_one(t or "") for t in texts] + + +def _char_ngrams(token: str, n: int) -> list[str]: + """Return padded character n-grams for *token* (bounded to avoid blow-up).""" + if len(token) < n: + return [token] + if len(token) > 24: + token = token[:24] + return [token[i:i + n] for i in range(len(token) - n + 1)] + + +class SentenceTransformerProvider(EmbeddingProvider): + """Local ``all-MiniLM-L6-v2`` embeddings via ``sentence-transformers``.""" + + name = "sentence-transformers" + + def __init__(self, model_name: str = "all-MiniLM-L6-v2") -> None: + super().__init__(EMBEDDING_DIM) + self.model_name = model_name + self._model: Any | None = None + + @staticmethod + def is_available() -> bool: + try: + import sentence_transformers # noqa: F401 + except Exception: + return False + return True + + def _get_model(self) -> Any: + if self._model is None: + from sentence_transformers import SentenceTransformer + + self._model = SentenceTransformer(self.model_name) + return self._model + + def encode(self, texts: list[str]) -> list[list[float]]: + if not texts: + return [] + model = self._get_model() + vectors = model.encode(texts, normalize_embeddings=True) + return [[float(x) for x in vec] for vec in vectors] + + +class RemoteEmbeddingProvider(EmbeddingProvider): + """OpenAI-compatible ``/embeddings`` endpoint (API key based).""" + + name = "remote" + + def __init__( + self, + base_url: str, + api_key: str, + model: str = "text-embedding-3-small", + dimension: int = EMBEDDING_DIM, + ) -> None: + super().__init__(dimension) + self.base_url = base_url.rstrip("/") + self.api_key = api_key + self.model = model + + def encode(self, texts: list[str]) -> list[list[float]]: + if not texts: + return [] + import httpx + + response = httpx.post( + f"{self.base_url}/embeddings", + headers={"Authorization": f"Bearer {self.api_key}"}, + json={"model": self.model, "input": texts}, + timeout=30.0, + ) + response.raise_for_status() + payload = response.json() + data = sorted(payload.get("data", []), key=lambda item: item.get("index", 0)) + return [self._normalize([float(x) for x in item["embedding"]]) for item in data] + + @staticmethod + def _normalize(vec: list[float]) -> list[float]: + norm = math.sqrt(sum(v * v for v in vec)) + if norm > 0.0: + return [v / norm for v in vec] + return vec + + +_provider: EmbeddingProvider | None = None +_provider_lock = threading.Lock() + + +def _build_provider() -> EmbeddingProvider: + """Select the best available provider (respecting ``OBSIGATE_EMBEDDING_PROVIDER``).""" + requested = os.getenv("OBSIGATE_EMBEDDING_PROVIDER", "auto").strip().lower() + + if requested in ("auto", "local", "sentence-transformers") and SentenceTransformerProvider.is_available(): + model = os.getenv("OBSIGATE_EMBEDDING_MODEL", "all-MiniLM-L6-v2") + return SentenceTransformerProvider(model) + + remote_key = os.getenv("OBSIGATE_EMBEDDING_API_KEY", "") + remote_url = os.getenv("OBSIGATE_EMBEDDING_BASE_URL", "") + if requested in ("auto", "remote") and remote_key and remote_url: + model = os.getenv("OBSIGATE_EMBEDDING_MODEL", "text-embedding-3-small") + dim = int(os.getenv("OBSIGATE_EMBEDDING_DIM", str(EMBEDDING_DIM))) + return RemoteEmbeddingProvider(remote_url, remote_key, model, dim) + + if requested == "remote": + logger.warning( + "OBSIGATE_EMBEDDING_PROVIDER=remote but OBSIGATE_EMBEDDING_API_KEY/BASE_URL missing; using hash fallback" + ) + return HashEmbeddingProvider() + + +def get_embedding_provider() -> EmbeddingProvider: + """Return the cached embedding provider (built on first use).""" + global _provider + with _provider_lock: + if _provider is None: + _provider = _build_provider() + logger.info("Semantic embedding provider: %s (dim=%d)", _provider.name, _provider.dimension) + return _provider + + +def reset_embedding_provider() -> None: + """Forget the cached provider (used by tests and config reloads).""" + global _provider + with _provider_lock: + _provider = None + + +# --------------------------------------------------------------------------- +# Vector store +# --------------------------------------------------------------------------- +class VectorStore: + """In-memory vector store with optional numpy / faiss acceleration. + + Vectors are always kept as Python lists (source of truth). A numpy matrix + and/or a faiss ``IndexFlatIP`` are built lazily and invalidated on mutation. + All vectors are expected to be L2-normalized, so the inner product equals + the cosine similarity. + """ + + def __init__(self, dimension: int = EMBEDDING_DIM) -> None: + self.dimension = dimension + self._keys: list[str] = [] + self._chunks: list[str] = [] + self._vectors: list[list[float]] = [] + self._dirty = True + self._numpy: Any | None = None + self._numpy_checked = False + self._matrix: Any | None = None + self._faiss: Any | None = None + self._faiss_checked = False + self._faiss_index: Any | None = None + + def __len__(self) -> int: + return len(self._vectors) + + def clear(self) -> None: + self._keys = [] + self._chunks = [] + self._vectors = [] + self._dirty = True + + def add(self, key: str, chunk: str, vector: list[float]) -> None: + self._keys.append(key) + self._chunks.append(chunk) + self._vectors.append(vector) + self._dirty = True + + def remove_document(self, key: str) -> None: + """Remove every chunk belonging to *key*.""" + kept = [(k, c, v) for k, c, v in zip(self._keys, self._chunks, self._vectors) if k != key] + if len(kept) == len(self._vectors): + return + self._keys = [k for k, _, _ in kept] + self._chunks = [c for _, c, _ in kept] + self._vectors = [v for _, _, v in kept] + self._dirty = True + + # -- optional accelerators ------------------------------------------------- + def _get_numpy(self) -> Any | None: + if not self._numpy_checked: + self._numpy_checked = True + try: + import numpy as np + + self._numpy = np + except Exception: + self._numpy = None + return self._numpy + + def _get_faiss(self) -> Any | None: + if not self._faiss_checked: + self._faiss_checked = True + try: + import faiss + + self._faiss = faiss + except Exception: + self._faiss = None + return self._faiss + + def _rebuild_accelerators(self) -> None: + self._dirty = False + np = self._get_numpy() + if np is None or not self._vectors: + self._matrix = None + self._faiss_index = None + return + self._matrix = np.asarray(self._vectors, dtype="float32") + faiss = self._get_faiss() + if faiss is not None: + index = faiss.IndexFlatIP(self.dimension) + index.add(self._matrix) + self._faiss_index = index + else: + self._faiss_index = None + + def search(self, query_vector: list[float], top_k: int = DEFAULT_TOP_K) -> list[tuple[str, float]]: + """Return ``(doc_key, cosine_similarity)`` pairs sorted by similarity.""" + if not self._vectors: + return [] + top_k = max(1, min(top_k, len(self._vectors))) + if self._dirty: + self._rebuild_accelerators() + + np = self._get_numpy() + if np is not None and self._faiss_index is not None and self._matrix is not None: + query = np.asarray([query_vector], dtype="float32") + scores, indices = self._faiss_index.search(query, top_k) + return [ + (self._keys[int(idx)], float(score)) + for score, idx in zip(scores[0], indices[0]) + if idx >= 0 + ] + + if np is not None and self._matrix is not None: + query = np.asarray(query_vector, dtype="float32") + scores = self._matrix @ query + order = np.argsort(scores)[::-1][:top_k] + return [(self._keys[int(i)], float(scores[int(i)])) for i in order] + + scored = [(self._keys[i], _dot(self._vectors[i], query_vector)) for i in range(len(self._vectors))] + scored.sort(key=lambda item: item[1], reverse=True) + return scored[:top_k] + + def chunk_of(self, index: int) -> str: + """Return the stored chunk text at *index* (used by diagnostics/tests).""" + return self._chunks[index] + + +def _dot(a: list[float], b: list[float]) -> float: + """Dot product for two equal-length vectors.""" + return sum(x * y for x, y in zip(a, b)) + + +# --------------------------------------------------------------------------- +# Reciprocal Rank Fusion +# --------------------------------------------------------------------------- +def rrf_fuse(rankings: list[list[str]], k: int = RRF_K) -> dict[str, float]: + """Fuse several ranked key lists into a single score map. + + ``score(key) = Σ_rankings 1 / (k + rank(key))`` where ``rank`` is + 1-based. Documents ranked highly by several methods rise to the top. + + Args: + rankings: Ordered lists of document keys (best first). + k: RRF smoothing constant. + + Returns: + Mapping ``doc_key -> fused score`` (insertion order is unspecified). + """ + scores: dict[str, float] = {} + for ranking in rankings: + seen: set[str] = set() + for rank, key in enumerate(ranking): + if key in seen: + continue + seen.add(key) + scores[key] = scores.get(key, 0.0) + 1.0 / (k + rank + 1) + return scores + + +# --------------------------------------------------------------------------- +# Semantic index +# --------------------------------------------------------------------------- +class SemanticIndex: + """Holds document chunk embeddings and answers similarity queries.""" + + def __init__(self, provider: EmbeddingProvider | None = None) -> None: + self.provider = provider + self.store = VectorStore(provider.dimension if provider else EMBEDDING_DIM) + self.doc_keys: set[str] = set() + self._ready = False + self._lock = threading.Lock() + + def is_ready(self) -> bool: + """Return True once a full rebuild has completed.""" + return self._ready + + def is_stale(self) -> bool: + """Alias used by callers that check index freshness.""" + return not self._ready + + def _ensure_provider(self) -> EmbeddingProvider: + if self.provider is None: + self.provider = get_embedding_provider() + self.store = VectorStore(self.provider.dimension) + return self.provider + + @staticmethod + def _document_text(file_info: dict[str, Any]) -> str: + title = file_info.get("title", "") or "" + content = file_info.get("content", "") or "" + return (title + "\n\n" + content).strip() + + def _embed_document(self, doc_key: str, file_info: dict[str, Any]) -> None: + text = self._document_text(file_info) + if not text: + return + provider = self._ensure_provider() + chunks = chunk_text(text) + if not chunks: + return + vectors = provider.encode(chunks) + for chunk, vector in zip(chunks, vectors): + self.store.add(doc_key, chunk, vector) + self.doc_keys.add(doc_key) + + def rebuild(self) -> None: + """Rebuild the whole index from the global in-memory index.""" + from backend.indexer import index + + provider = self._ensure_provider() + with self._lock: + self.store = VectorStore(provider.dimension) + self.doc_keys = set() + for vault_name, vault_data in index.items(): + for file_info in vault_data.get("files", []): + doc_key = f"{vault_name}::{file_info.get('path', '')}" + try: + self._embed_document(doc_key, file_info) + except Exception as exc: + logger.warning("Semantic embedding failed for %s: %s", doc_key, exc) + self._ready = True + logger.info( + "Semantic index built: %d documents, %d chunks (provider=%s)", + len(self.doc_keys), + len(self.store), + provider.name, + ) + + def add_document(self, vault_name: str, path: str, file_info: dict[str, Any]) -> None: + """Add or refresh a single document (no-op until the index is ready).""" + if not self._ready or not file_info: + return + doc_key = f"{vault_name}::{path}" + with self._lock: + self.store.remove_document(doc_key) + self.doc_keys.discard(doc_key) + try: + self._embed_document(doc_key, file_info) + except Exception as exc: + logger.warning("Semantic embedding failed for %s: %s", doc_key, exc) + + def remove_document(self, vault_name: str, path: str) -> None: + """Remove a single document (no-op until the index is ready).""" + if not self._ready: + return + doc_key = f"{vault_name}::{path}" + with self._lock: + self.store.remove_document(doc_key) + self.doc_keys.discard(doc_key) + + def search( + self, + query: str, + vault_filter: str = "all", + top_k: int = DEFAULT_TOP_K, + ) -> list[tuple[str, float]]: + """Return ``(doc_key, best_chunk_similarity)`` pairs, best first.""" + if not self._ready or not query or not query.strip(): + return [] + provider = self._ensure_provider() + query_vector = provider.encode_one(query[:MAX_QUERY_CHARS]) + hits = self.store.search(query_vector, top_k=max(top_k * 4, top_k)) + best: dict[str, float] = {} + for doc_key, score in hits: + if vault_filter != "all" and not doc_key.startswith(vault_filter + "::"): + continue + if doc_key not in best or score > best[doc_key]: + best[doc_key] = score + ordered = sorted(best.items(), key=lambda item: item[1], reverse=True) + return ordered[:top_k] + + +_semantic_index: SemanticIndex | None = None +_index_lock = threading.Lock() + + +def get_semantic_index() -> SemanticIndex: + """Return the process-wide semantic index (created on first access).""" + global _semantic_index + with _index_lock: + if _semantic_index is None: + _semantic_index = SemanticIndex() + return _semantic_index + + +def reset_semantic_index() -> None: + """Drop the singleton index (tests).""" + global _semantic_index + with _index_lock: + _semantic_index = None + + +def init_semantic_index() -> None: + """Force a full semantic index build. Called after ``build_index`` on startup.""" + from backend.indexer import index + + if any(vdata.get("files") for vdata in index.values()): + get_semantic_index().rebuild() + + +def on_index_change(action: str, vault_name: str, path: str, file_info: dict[str, Any]) -> None: + """Incremental hook registered with the indexer change notifier.""" + index_obj = get_semantic_index() + if action == "add" and file_info: + index_obj.add_document(vault_name, path, file_info) + elif action == "remove": + index_obj.remove_document(vault_name, path) + + +def semantic_search_docs( + query: str, + vault_filter: str = "all", + top_k: int = DEFAULT_TOP_K, +) -> list[tuple[str, float]]: + """Convenience wrapper around :meth:`SemanticIndex.search`.""" + return get_semantic_index().search(query, vault_filter=vault_filter, top_k=top_k) + + +def semantic_status() -> dict[str, Any]: + """Return provider/index diagnostics for the API and the UI.""" + index_obj = get_semantic_index() + provider = index_obj.provider or get_embedding_provider() + return { + "available": index_obj.is_ready(), + "provider": provider.name, + "dimension": provider.dimension, + "documents": len(index_obj.doc_keys), + "chunks": len(index_obj.store), + } diff --git a/backend/services/search.py b/backend/services/search.py index 6be7f18..d58e831 100644 --- a/backend/services/search.py +++ b/backend/services/search.py @@ -56,11 +56,13 @@ def advanced_search_vaults( created: str | None = None, modified: str | None = None, size: str | None = None, + semantic: bool = False, ) -> dict[str, Any]: """Advanced full-text search (TF-IDF, facets, operators). - No permission filtering is applied: callers that need it (the tool layer) - filter the ``results`` list themselves. + When ``semantic`` is True, the TF-IDF ranking is fused with the embedding + (semantic) ranking via RRF. No permission filtering is applied: callers + that need it (the tool layer) filter the ``results`` list themselves. """ from backend.search import advanced_search @@ -79,6 +81,7 @@ def advanced_search_vaults( created=created, modified=modified, size=size, + semantic=semantic, ) diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index f1cf22a..18a2964 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -1,6 +1,6 @@ # ObsiGate — Roadmap -> **Version :** 2.2.1 | **Dernière mise à jour :** 2026-09-11 +> **Version :** 2.3.0-dev | **Dernière mise à jour :** 2026-09-12 > **Ce fichier ne contient que le travail à venir** (🔵 En cours + ⚪ Backlog) et un index compact > vers les fonctionnalités livrées. > - **Méthode de livraison à appliquer pour toute tâche : [DELIVERY_WORKFLOW.md](./DELIVERY_WORKFLOW.md)** @@ -44,27 +44,6 @@ ## ⚪ Backlog — Priorité 4 (P4) -### 70. Recherche sémantique — Embeddings vectoriels - -- **Effort :** 4-5 jours | **Impact :** 🟢 -- **Description :** La recherche actuelle (TF-IDF) ne trouve que les documents contenant EXACTEMENT les mots tapés. La recherche sémantique comprend le SENS de la requête et trouve des documents pertinents même s'ils utilisent des mots différents. - - **Exemple concret :** Vous cherchez « comment sauvegarder mes données ». La recherche TF-IDF ne trouvera que les documents contenant « sauvegarder » ET « données ». La recherche sémantique trouvera aussi un document titré « Stratégie de backup automatique » ou « Protection contre la perte de fichiers » parce qu'elle comprend que ces phrases parlent de la même chose. - - **Fonctionnement technique :** - - Chaque document (ou chunk de ~512 tokens) est converti en un **vecteur** (une liste de 384 nombres) par un modèle de langage léger comme `all-MiniLM-L6-v2` (80 Mo, s'exécute en ~2ms par document sur CPU). Ce vecteur capture le sens — deux phrases qui veulent dire la même chose auront des vecteurs très proches. - - Au moment de la recherche, la requête utilisateur est elle aussi convertie en vecteur. - - On calcule la **similarité cosinus** entre le vecteur de la requête et les vecteurs de tous les documents. Les documents avec la similarité la plus élevée sont retournés. - - **Recherche hybride** : on combine le score TF-IDF (pertinence par mots-clés exacts) et le score sémantique (pertinence par sens) via RRF (Reciprocal Rank Fusion) — les documents bien classés par les deux méthodes remontent en premier. - - **Stockage** : les vecteurs sont stockés avec FAISS (Facebook AI Similarity Search), une bibliothèque optimisée qui permet de chercher parmi des millions de vecteurs en quelques millisecondes. - - **Indexation** : les embeddings sont générés une fois à l'indexation du fichier (pas à chaque recherche). Un fichier modifié voit son embedding regénéré automatiquement par le watcher. -- **Pourquoi c'est important :** La recherche par mots-clés échoue dans ~30% des cas où l'utilisateur ne se souvient pas des mots exacts utilisés dans ses notes. La recherche sémantique résout ce problème. C'est particulièrement utile pour les gros vaults (500+ notes) où on ne peut pas tout parcourir manuellement. -- **Sous-tâches :** - - [ ] Génération d'embeddings : modèle `all-MiniLM-L6-v2` via `sentence-transformers` (Python) ou appel API externe - - [ ] Stockage : index vectoriel avec `numpy` + `faiss` (ou `usearch` pour performance) - - [ ] Indexation : embedding par chunk de 512 tokens avec recouvrement - - [ ] Recherche hybride : combinaison TF-IDF + similarité cosinus (RRF — Reciprocal Rank Fusion) - - [ ] UI : toggle « Recherche sémantique » dans la barre de recherche - - [ ] UI : score de similarité dans les résultats - ### 73. Synchronisation multi-appareils — Obsidian Sync compatible - **Effort :** 6-8 jours | **Impact :** 🟢 @@ -116,6 +95,7 @@ | 79 | Assistant IA — Outils (function calling) & serveur MCP | 2.3.0 | [features/ai-tools-mcp.md](./features/ai-tools-mcp.md) | | 80 | Assistant IA — Rendu Markdown, liens fichiers/paths & sessions | 2.3.0 | [features/ai-assistant-ux.md](./features/ai-assistant-ux.md) | | 69 | Éditeur mobile natif — Interface tactile optimisée | 2.3.0 | [features/mobile-editor.md](./features/mobile-editor.md) | +| 70 | Recherche sémantique — Embeddings vectoriels (hybride TF-IDF + RRF) | 2.3.0 | [features/semantic-search.md](./features/semantic-search.md) | | 77 | Application Desktop native — Tauri | 🔵 en cours | [features/desktop-tauri.md](./features/desktop-tauri.md) | --- @@ -124,10 +104,10 @@ | Priorité | Items | Effort total estimé | |---|---|---| -| ✅ Complété | #1 → #59, #61–72, #74–76, #78–80 | ~99 jours réalisés | +| ✅ Complété | #1 → #59, #61–72, #74–76, #78–80 | ~103 jours réalisés | | 🔵 P2 restant | #77 Desktop : signature code (optionnel), wizard 1er lancement (optionnel), 6 tests E2E **manuels** | ~1-2 jours | -| ⚪ P4 restant | #70 Sémantique (4-5j) · #73 Sync (6-8j) | 10-13 jours | -| **Total restant** | **3 items + finitions** | **~11-15 jours** | +| ⚪ P4 restant | #73 Sync (6-8j) | 6-8 jours | +| **Total restant** | **2 items + finitions** | **~7-10 jours** | --- diff --git a/docs/features/semantic-search.md b/docs/features/semantic-search.md new file mode 100644 index 0000000..7d5a7c9 --- /dev/null +++ b/docs/features/semantic-search.md @@ -0,0 +1,128 @@ +# #70 - Recherche sémantique - Embeddings vectoriels + +> **Statut :** ✅ Terminé — 100 % implémenté + 27 tests backend + 4 tests frontend (2026-09-12) +> **Effort :** 4-5 jours (réalisé) | **Impact :** 🟢 +> **Références :** [Roadmap](../ROADMAP.md) · [Changelog](../../CHANGELOG.md) + +- **Fichiers clés :** + - `backend/semantic_search.py` - chunking, providers d'embeddings, `VectorStore`, `SemanticIndex`, RRF, hook incrémental + - `backend/search.py` - fusion RRF dans `advanced_search(..., semantic=True)` + `_fuse_semantic_results()` + - `backend/services/search.py` - paramètre `semantic` du service + - `backend/main.py` - paramètre `semantic` de `/api/search/advanced`, schémas, `init_semantic_index()` au démarrage + - `frontend/js/search.js` - toggle `#rb-semantic`, score de similarité, raccourci `Alt+S` + - `frontend/js/state.js` - `semanticSearch`, `semanticAvailable` + - `frontend/index.html` - bouton `#rb-semantic` dans la barre de résultats + - `frontend/locales/fr.json`, `frontend/locales/en.json` - clés `search.semantic_*` + - `backend/requirements-semantic.txt` - dépendances **optionnelles** (sentence-transformers, numpy, faiss-cpu) + - `tests/test_semantic_search.py` - 27 tests backend + - `tests/frontend/semantic-search.test.mjs` - 4 tests JSDOM + +- **Description :** la recherche TF-IDF historique ne trouve que les documents contenant + **exactement** les mots tapés. La recherche sémantique comprend le **sens** de la requête et + retrouve des documents pertinents même formulés différemment (« comment sauvegarder mes + données » remonte aussi « Stratégie de backup automatique »). Les deux classements sont + fusionnés via **RRF** (Reciprocal Rank Fusion), ce qui combine précision lexicale et rappel + sémantique. + +--- + +## Architecture + +``` + ┌───────────────────────────────┐ + indexer (watcher) │ on_index_change(action, …) │ + ─────────────────► │ SemanticIndex.add/remove │ + └──────────────┬────────────────┘ + │ chunks (512 mots, recouvrement 64) + ▼ + ┌───────────────────────────────┐ + │ EmbeddingProvider │ + │ sentence-transformers │ API │ + │ │ hash (fallback sans dep) │ + └──────────────┬────────────────┘ + │ vecteurs 384 dim, L2-normalisés + ▼ + ┌───────────────────────────────┐ + │ VectorStore (numpy / faiss │ + │ / pur Python) — cosinus │ + └──────────────┬────────────────┘ + requête ──► embed ──► similarité ──┘ + │ + TF-IDF ranking ──────────► RRF ◄──┘ ──► résultats fusionnés (semantic_score) +``` + +## Backend - `backend/semantic_search.py` + +- **Chunking** : `chunk_text(text, chunk_tokens=512, overlap=64)` découpe chaque document en + fenêtres glissantes de 512 mots avec 64 mots de recouvrement, pour ne pas perdre le contexte + aux frontières. +- **Providers d'embeddings** (`EmbeddingProvider`) : + - `SentenceTransformerProvider` — modèle local `all-MiniLM-L6-v2` (384 dim), chargé + paresseusement. Actif si `sentence-transformers` est installé. + - `RemoteEmbeddingProvider` — endpoint `/embeddings` compatible OpenAI + (`OBSIGATE_EMBEDDING_API_KEY`, `OBSIGATE_EMBEDDING_BASE_URL`, `OBSIGATE_EMBEDDING_MODEL`). + - `HashEmbeddingProvider` — **repli sans aucune dépendance** : hachage signé déterministe des + unigrammes, bigrammes et trigrammes de caractères (hashing trick) + normalisation L2. Il + capture le vocabulaire partagé et les variantes morphologiques, mais pas les synonymes. + - Sélection via `OBSIGATE_EMBEDDING_PROVIDER=auto|local|remote|hash` (défaut `auto`). +- **`VectorStore`** : stocke les vecteurs par chunk. Accélération optionnelle **numpy** (produit + matriciel) puis **faiss** (`IndexFlatIP`) ; sinon cosinus pur Python. Les vecteurs étant + L2-normalisés, le produit scalaire vaut la similarité cosinus. +- **`SemanticIndex`** : singleton `get_semantic_index()`. `rebuild()` construit tout depuis + `backend.indexer.index` ; `add_document()` / `remove_document()` mettent à jour un document à + chaud (appelés par le hook `on_index_change`). `search()` regroupe les chunks par document en + gardant la meilleure similarité et filtre par vault. +- **RRF** : `rrf_fuse(rankings, k=60)` — `score = Σ 1/(k + rang)`. + +## Backend - intégration recherche + +`advanced_search(..., semantic=True)` : + +1. Le classement TF-IDF est calculé comme avant. +2. `_fuse_semantic_results()` récupère les hits sémantiques, les restreint aux mêmes filtres + (vault, tags, `title:`, `path:`, `ext:`, dates, taille, include/exclude) et fusionne les deux + classements par RRF. Les documents trouvés uniquement par le sémantique sont matérialisés + depuis l'index inversé (snippet brut, sans ``). +3. Chaque résultat porte `semantic_score` (similarité cosinus, `0.0` si absent) et la réponse + expose `semantic_available`. + +Le mode sémantique est ignoré si `regex=True` (notion purement lexicale). + +## API + +`GET /api/search/advanced?...&semantic=true` ajoute : + +- `results[].semantic_score` (float) ; +- `semantic_available` (bool) : l'index sémantique est prêt. + +## Frontend + +- Bouton `~` (`#rb-semantic`) dans la barre de résultats, actif seulement quand + `semantic_available` est vrai (sinon `disabled`). +- Raccourci clavier **Alt+S** ; l'état est conservé dans `state.semanticSearch`. +- Le badge de score affiche `score: … · sim: …` quand un score sémantique existe. +- Clés i18n `search.semantic_title`, `search.semantic_score`, `search.semantic_unavailable`. + +## Configuration + +| Variable | Défaut | Rôle | +|---|---|---| +| `OBSIGATE_EMBEDDING_PROVIDER` | `auto` | `auto`, `local`, `remote` ou `hash` | +| `OBSIGATE_EMBEDDING_MODEL` | `all-MiniLM-L6-v2` | Modèle local ou nom de modèle distant | +| `OBSIGATE_EMBEDDING_API_KEY` | — | Clé du provider distant | +| `OBSIGATE_EMBEDDING_BASE_URL` | — | URL de base du provider distant | +| `OBSIGATE_EMBEDDING_DIM` | `384` | Dimension attendue pour le provider distant | + +## Dépendances optionnelles + +`pip install -r backend/requirements-semantic.txt` (sentence-transformers + numpy + faiss-cpu). +**Sans installation, la fonctionnalité reste opérationnelle** grâce au provider `hash` et au +stockage pur Python : le CI n'installe que `backend/requirements.txt`. + +## Tests + +- `tests/test_semantic_search.py` (27) : chunking, provider hash (déterminisme, normalisation, + similarité), `VectorStore`, RRF, `SemanticIndex`, hook incrémental, `advanced_search` + sémantique, endpoint API. +- `tests/frontend/semantic-search.test.mjs` (4) : toggle désactivé tant que l'index n'est pas + prêt, bascule état + classe active, no-op si indisponible. diff --git a/frontend/index.html b/frontend/index.html index c63a8ba..6cf3edf 100644 --- a/frontend/index.html +++ b/frontend/index.html @@ -5235,6 +5235,7 @@ curl -X POST https://votre-serveur.com/webhook \ +
diff --git a/frontend/js/search.js b/frontend/js/search.js index 6477cf0..f263584 100644 --- a/frontend/js/search.js +++ b/frontend/js/search.js @@ -524,6 +524,7 @@ export function initSearch() { const caseBtn = document.getElementById("rb-case"); const wordBtn = document.getElementById("rb-word"); const regexBtn = document.getElementById("rb-regex"); + const semanticBtn = document.getElementById("rb-semantic"); const filterBtn = document.getElementById("search-filter-btn"); const clearBtn = document.getElementById("search-clear-btn"); const filterRow = document.getElementById("search-filter-row"); @@ -540,12 +541,25 @@ export function initSearch() { wordBtn.classList.toggle("active", state.searchWholeWord); regexBtn.classList.toggle("active", state.searchRegex); filterBtn.classList.toggle("active", state.searchFilterVisible); + if (semanticBtn) { + semanticBtn.classList.toggle("active", state.semanticSearch); + semanticBtn.disabled = !state.semanticAvailable; + } } // Toggle buttons caseBtn.addEventListener("click", () => { state.searchCaseSensitive = !state.searchCaseSensitive; _updateToggleUI(); _research(); }); if (wordBtn) wordBtn.addEventListener("click", () => { state.searchWholeWord = !state.searchWholeWord; _updateToggleUI(); _research(); }); if (regexBtn) regexBtn.addEventListener("click", () => { state.searchRegex = !state.searchRegex; _updateToggleUI(); _research(); }); + if (semanticBtn) { + semanticBtn.title = t("search.semantic_title"); + semanticBtn.addEventListener("click", () => { + if (!state.semanticAvailable) return; + state.semanticSearch = !state.semanticSearch; + _updateToggleUI(); + _research(); + }); + } if (filterBtn) filterBtn.addEventListener("click", () => { state.searchFilterVisible = !state.searchFilterVisible; if (filterRow) filterRow.style.display = state.searchFilterVisible ? "flex" : "none"; _updateToggleUI(); }); // ── Result navigation (up/down arrows + Enter) ── @@ -599,6 +613,7 @@ export function initSearch() { if (e.key === "c" || e.key === "C") { e.preventDefault(); caseBtn.click(); } else if (e.key === "w" || e.key === "W") { e.preventDefault(); if (wordBtn) wordBtn.click(); } else if (e.key === "r" || e.key === "R") { e.preventDefault(); if (regexBtn) regexBtn.click(); } + else if (e.key === "s" || e.key === "S") { e.preventDefault(); if (semanticBtn) semanticBtn.click(); } else if (e.key === "f" || e.key === "F") { e.preventDefault(); if (filterBtn) filterBtn.click(); input.focus(); } } }); @@ -788,6 +803,7 @@ export async function performAdvancedSearch(query, vaultFilter, tagFilter, offse if (parsed.created) url += `&created=${encodeURIComponent(parsed.created)}`; if (parsed.modified) url += `&modified=${encodeURIComponent(parsed.modified)}`; if (parsed.size) url += `&size=${encodeURIComponent(parsed.size)}`; + if (state.semanticSearch) url += "&semantic=true"; // Search timeout — abort if server takes too long const timeoutId = setTimeout( @@ -803,6 +819,12 @@ export async function performAdvancedSearch(query, vaultFilter, tagFilter, offse if (searchId !== state.currentSearchId) return; state.advancedSearchTotal = data.total; state.advancedSearchOffset = ofs; + state.semanticAvailable = !!data.semantic_available; + const semBtn = document.getElementById("rb-semantic"); + if (semBtn) { + semBtn.disabled = !state.semanticAvailable; + semBtn.classList.toggle("active", state.semanticSearch); + } // Plugin hook: filter search results data.results = await onSearchFilter(state.vault || "default", query, data.results); renderAdvancedSearchResults(data, query, tagFilter); @@ -1092,7 +1114,9 @@ export function renderAdvancedSearchResults(data, query, tagFilter) { // Score badge const scoreEl = el("span", { class: "search-result-score", style: "font-size:0.7rem;color:var(--text-muted);margin-left:8px" }); - scoreEl.textContent = `score: ${r.score}`; + let scoreText = `score: ${r.score}`; + if (r.semantic_score) scoreText += ` · ${t("search.semantic_score")}: ${r.semantic_score}`; + scoreEl.textContent = scoreText; const vaultPath = el("div", { class: "search-result-vault" }, [document.createTextNode(r.vault + " / " + r.path), scoreEl]); diff --git a/frontend/js/state.js b/frontend/js/state.js index cc55c0d..9d870de 100644 --- a/frontend/js/state.js +++ b/frontend/js/state.js @@ -24,6 +24,8 @@ export const state = { searchWholeWord: false, searchRegex: false, searchFilterVisible: false, + semanticSearch: false, + semanticAvailable: false, // Search constants SEARCH_HISTORY_KEY: "obsigate_search_history", diff --git a/frontend/locales/en.json b/frontend/locales/en.json index 87b1799..9925839 100644 --- a/frontend/locales/en.json +++ b/frontend/locales/en.json @@ -1398,6 +1398,9 @@ "search.save_search": "Save search", "search.saved": "Search saved", "search.sections_all_vaults": "All vaults", + "search.semantic_score": "sim", + "search.semantic_title": "Semantic search (Alt-S)", + "search.semantic_unavailable": "Semantic search unavailable", "search.simple_search": "Simple search", "search.smart_scoring": "Smart scoring", "search.sort_date": "Date", diff --git a/frontend/locales/fr.json b/frontend/locales/fr.json index 8c78fc1..b1beb8c 100644 --- a/frontend/locales/fr.json +++ b/frontend/locales/fr.json @@ -1398,6 +1398,9 @@ "search.save_search": "Sauvegarder la recherche", "search.saved": "Recherche sauvegardée", "search.sections_all_vaults": "Tous les vaults", + "search.semantic_score": "sim", + "search.semantic_title": "Recherche sémantique (Alt-S)", + "search.semantic_unavailable": "Recherche sémantique indisponible", "search.simple_search": "Recherche simple", "search.smart_scoring": "Scoring intelligent", "search.sort_date": "Date", diff --git a/tests/conftest.py b/tests/conftest.py index ccb427e..85b46f6 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -114,6 +114,10 @@ def app_with_vault(test_vault_dir: str): from backend.search import init_inverted_index init_inverted_index() + # Build semantic (embedding) index — uses the dependency-free hash fallback + from backend.semantic_search import init_semantic_index + init_semantic_index() + client = TestClient(app) return client @@ -196,6 +200,9 @@ def admin_client(tmp_path): from backend.search import init_inverted_index init_inverted_index() + from backend.semantic_search import init_semantic_index + init_semantic_index() + from fastapi.testclient import TestClient client = TestClient(app) yield client diff --git a/tests/frontend/semantic-search.test.mjs b/tests/frontend/semantic-search.test.mjs new file mode 100644 index 0000000..33dc4df --- /dev/null +++ b/tests/frontend/semantic-search.test.mjs @@ -0,0 +1,122 @@ +#!/usr/bin/env node +/** + * ObsiGate - JSDOM tests for the semantic search toggle (ROADMAP #70). + * + * Verifies the "Recherche sémantique" toggle wiring in frontend/js/search.js: + * - the toggle is disabled until the backend reports semantic availability + * - clicking it flips state.semanticSearch and the active class + * - clicking while unavailable is a no-op + * + * Usage: node tests/frontend/semantic-search.test.mjs + */ + +import { strict as assert } from "node:assert"; +import { JSDOM } from "jsdom"; +import { fileURLToPath, pathToFileURL } from "node:url"; +import path from "node:path"; + +const __filename = fileURLToPath(import.meta.url); +const __dirname = path.dirname(__filename); +const REPO_ROOT = path.resolve(__dirname, "..", ".."); + +// ── JSDOM bootstrap ──────────────────────────────────────────────────────── +const dom = new JSDOM( + ` + + + + + + +
+
+ `, + { url: "https://example.com/", pretendToBeVisual: true }, +); + +const w = dom.window; +globalThis.window = w; +globalThis.document = w.document; +globalThis.HTMLElement = w.HTMLElement; +globalThis.Element = w.Element; +globalThis.Node = w.Node; +globalThis.Event = w.Event; +globalThis.CustomEvent = w.CustomEvent; +globalThis.KeyboardEvent = w.KeyboardEvent; +globalThis.localStorage = w.localStorage; +Object.defineProperty(globalThis, "navigator", { + value: w.navigator, + configurable: true, + writable: true, +}); +globalThis.MutationObserver = w.MutationObserver; +globalThis.getComputedStyle = w.getComputedStyle.bind(w); +globalThis.requestAnimationFrame = (cb) => setTimeout(cb, 0); +globalThis.cancelAnimationFrame = (id) => clearTimeout(id); +globalThis.fetch = async () => ({ ok: true, json: async () => ({}) }); + +const stateMod = await import( + pathToFileURL(path.join(REPO_ROOT, "frontend", "js", "state.js")).href +); +const { state } = stateMod; +const searchMod = await import( + pathToFileURL(path.join(REPO_ROOT, "frontend", "js", "search.js")).href +); + +// ── Test harness ─────────────────────────────────────────────────────────── +let testCount = 0; +let passCount = 0; + +async function test(name, fn) { + testCount++; + try { + await fn(); + console.log(` \u2713 ${name}`); + passCount++; + } catch (e) { + console.log(` \u2717 ${name}`); + console.log(` ${e.message}`); + } +} + +searchMod.initSearch(); +const semanticBtn = document.getElementById("rb-semantic"); + +await test("semantic toggle is disabled until availability is known", () => { + assert.equal(semanticBtn.disabled, true); +}); + +await test("clicking while unavailable does not toggle", () => { + semanticBtn.click(); + assert.equal(state.semanticSearch, false); +}); + +await test("clicking while available toggles state and active class", () => { + state.semanticAvailable = true; + semanticBtn.disabled = false; + semanticBtn.click(); + assert.equal(state.semanticSearch, true); + assert.ok(semanticBtn.classList.contains("active")); + semanticBtn.click(); + assert.equal(state.semanticSearch, false); + assert.ok(!semanticBtn.classList.contains("active")); +}); + +await test("semantic state keys exist", () => { + assert.equal(state.semanticAvailable, true); + assert.equal(typeof state.semanticSearch, "boolean"); +}); + +console.log(`\n${passCount}/${testCount} tests passed`); +if (passCount !== testCount) process.exit(1); diff --git a/tests/test_semantic_search.py b/tests/test_semantic_search.py new file mode 100644 index 0000000..792e4f0 --- /dev/null +++ b/tests/test_semantic_search.py @@ -0,0 +1,243 @@ +# tests/test_semantic_search.py — Tests for semantic search (embeddings + RRF) +import math + +from backend.search import advanced_search +from backend.semantic_search import ( + EMBEDDING_DIM, + HashEmbeddingProvider, + SemanticIndex, + VectorStore, + chunk_text, + get_semantic_index, + on_index_change, + reset_semantic_index, + rrf_fuse, + semantic_status, +) + +# ═══════════════════════════════════════════════════════════════════ +# Chunking +# ═══════════════════════════════════════════════════════════════════ + +class TestChunkText: + def test_empty(self): + assert chunk_text("") == [] + assert chunk_text(" ") == [] + + def test_short_text_single_chunk(self): + chunks = chunk_text("un deux trois", chunk_tokens=512) + assert chunks == ["un deux trois"] + + def test_long_text_multiple_chunks(self): + words = " ".join(f"mot{i}" for i in range(1200)) + chunks = chunk_text(words, chunk_tokens=512, overlap=64) + assert len(chunks) >= 3 + # First chunk has 512 words + assert len(chunks[0].split()) == 512 + + def test_overlap_shared_tokens(self): + words = [f"w{i}" for i in range(600)] + chunks = chunk_text(" ".join(words), chunk_tokens=100, overlap=20) + first_tail = chunks[0].split()[-20:] + second_head = chunks[1].split()[:20] + assert first_tail == second_head + + +# ═══════════════════════════════════════════════════════════════════ +# Hash embedding provider (dependency-free fallback) +# ═══════════════════════════════════════════════════════════════════ + +class TestHashEmbeddingProvider: + def setup_method(self): + self.provider = HashEmbeddingProvider() + + def test_dimension(self): + vec = self.provider.encode_one("hello world") + assert len(vec) == EMBEDDING_DIM + + def test_l2_normalized(self): + vec = self.provider.encode_one("un texte de test assez long") + norm = math.sqrt(sum(v * v for v in vec)) + assert abs(norm - 1.0) < 1e-6 + + def test_empty_text_zero_vector(self): + vec = self.provider.encode_one("") + assert all(v == 0.0 for v in vec) + + def test_deterministic(self): + a = self.provider.encode_one("sauvegarde des données") + b = self.provider.encode_one("sauvegarde des données") + assert a == b + + def test_similar_more_similar_than_dissimilar(self): + base = self.provider.encode_one("stratégie de sauvegarde automatique des données") + close = self.provider.encode_one("sauvegarde automatique des données") + far = self.provider.encode_one("recette de cuisine au chocolat") + sim_close = sum(x * y for x, y in zip(base, close)) + sim_far = sum(x * y for x, y in zip(base, far)) + assert sim_close > sim_far + + def test_batch_matches_single(self): + texts = ["premier document", "deuxième document"] + batch = self.provider.encode(texts) + assert batch[0] == self.provider.encode_one(texts[0]) + assert batch[1] == self.provider.encode_one(texts[1]) + + +# ═══════════════════════════════════════════════════════════════════ +# Vector store +# ═══════════════════════════════════════════════════════════════════ + +class TestVectorStore: + def test_add_and_search(self): + provider = HashEmbeddingProvider() + store = VectorStore(provider.dimension) + store.add("v::a.md", "a", provider.encode_one("python programmation")) + store.add("v::b.md", "b", provider.encode_one("recette cuisine chocolat")) + hits = store.search(provider.encode_one("python"), top_k=2) + assert hits[0][0] == "v::a.md" + assert hits[0][1] > hits[1][1] + + def test_remove_document(self): + provider = HashEmbeddingProvider() + store = VectorStore(provider.dimension) + store.add("v::a.md", "a", provider.encode_one("python")) + store.add("v::a.md", "a2", provider.encode_one("code")) + store.add("v::b.md", "b", provider.encode_one("cuisine")) + assert len(store) == 3 + store.remove_document("v::a.md") + assert len(store) == 1 + hits = store.search(provider.encode_one("python"), top_k=5) + assert all(key != "v::a.md" for key, _ in hits) + + def test_empty_search(self): + store = VectorStore(EMBEDDING_DIM) + assert store.search([0.0] * EMBEDDING_DIM) == [] + + +# ═══════════════════════════════════════════════════════════════════ +# Reciprocal Rank Fusion +# ═══════════════════════════════════════════════════════════════════ + +class TestRRF: + def test_single_ranking(self): + scores = rrf_fuse([["a", "b"]]) + assert scores["a"] > scores["b"] + + def test_fusion_promotes_consensus(self): + scores = rrf_fuse([["a", "b", "c"], ["b", "c", "d"]]) + # "b" is well ranked by both methods -> best fused score + assert max(scores, key=scores.get) == "b" + + def test_dedup_within_ranking(self): + scores = rrf_fuse([["a", "a", "b"]]) + single = rrf_fuse([["a", "b"]]) + assert abs(scores["a"] - single["a"]) < 1e-9 + + +# ═══════════════════════════════════════════════════════════════════ +# SemanticIndex — unit (injected provider, manual documents) +# ═══════════════════════════════════════════════════════════════════ + +class TestSemanticIndexUnit: + def test_add_search_remove(self): + index = SemanticIndex(provider=HashEmbeddingProvider()) + index._ready = True + index.add_document("V", "backup.md", { + "path": "backup.md", + "title": "Stratégie de backup", + "content": "Protection et sauvegarde automatique des données", + "tags": [], + }) + index.add_document("V", "cuisine.md", { + "path": "cuisine.md", + "title": "Recette au chocolat", + "content": "Faire fondre le chocolat puis ajouter la farine", + "tags": [], + }) + hits = index.search("sauvegarde des données", top_k=5) + assert hits + assert hits[0][0] == "V::backup.md" + + index.remove_document("V", "backup.md") + hits = index.search("sauvegarde des données", top_k=5) + assert all(key != "V::backup.md" for key, _ in hits) + + def test_vault_filter(self): + index = SemanticIndex(provider=HashEmbeddingProvider()) + index._ready = True + index.add_document("V1", "a.md", {"path": "a.md", "title": "python", "content": "python", "tags": []}) + index.add_document("V2", "b.md", {"path": "b.md", "title": "python", "content": "python", "tags": []}) + hits = index.search("python", vault_filter="V1", top_k=5) + assert hits + assert all(key.startswith("V1::") for key, _ in hits) + + def test_not_ready_is_noop(self): + index = SemanticIndex(provider=HashEmbeddingProvider()) + index.add_document("V", "a.md", {"path": "a.md", "title": "x", "content": "x", "tags": []}) + assert len(index.store) == 0 + assert index.search("x") == [] + + +# ═══════════════════════════════════════════════════════════════════ +# Integration with the global index / advanced_search +# ═══════════════════════════════════════════════════════════════════ + +class TestSemanticIntegration: + def test_rebuild_from_global_index(self, client): + index = get_semantic_index() + assert index.is_ready() + assert len(index.doc_keys) >= 3 + + def test_on_index_change_hook(self, client): + index = get_semantic_index() + file_info = { + "path": "semantic_hook.md", + "title": "Sauvegarde", + "content": "Sauvegarde automatique des données", + "tags": [], + } + on_index_change("add", "TestVault", "semantic_hook.md", file_info) + assert "TestVault::semantic_hook.md" in index.doc_keys + on_index_change("remove", "TestVault", "semantic_hook.md", file_info) + assert "TestVault::semantic_hook.md" not in index.doc_keys + + def test_advanced_search_semantic(self, client): + result = advanced_search("python", vault_filter="all", semantic=True) + assert result["semantic_available"] is True + assert len(result["results"]) >= 1 + assert any(r.get("semantic_score", 0) > 0 for r in result["results"]) + + def test_advanced_search_semantic_field_default(self, client): + result = advanced_search("python", vault_filter="all") + for r in result["results"]: + assert "semantic_score" in r + + def test_semantic_status(self, client): + status = semantic_status() + assert status["available"] is True + assert status["documents"] >= 3 + assert status["dimension"] > 0 + + def test_reset_semantic_index(self, client): + reset_semantic_index() + assert get_semantic_index().is_ready() is False + # Restore for subsequent tests + from backend.semantic_search import init_semantic_index + init_semantic_index() + + +class TestSemanticAPI: + def test_api_semantic_flag(self, client): + resp = client.get("/api/search/advanced?q=python&vault=all&semantic=true") + assert resp.status_code == 200 + data = resp.json() + assert data["semantic_available"] is True + assert len(data["results"]) >= 1 + assert "semantic_score" in data["results"][0] + + def test_api_without_semantic(self, client): + resp = client.get("/api/search/advanced?q=python&vault=all") + assert resp.status_code == 200 + data = resp.json() + assert "semantic_available" in data