feat(search): recherche semantique - embeddings vectoriels + hybride RRF (#70)
This commit is contained in:
+10
-2
@@ -98,6 +98,7 @@ from backend.search import (
|
||||
suggest_tags,
|
||||
suggest_titles,
|
||||
)
|
||||
from backend.semantic_search import init_semantic_index
|
||||
from backend.services.backups import diff_backup as service_diff_backup
|
||||
from backend.services.backups import get_backup_dir as service_get_backup_dir
|
||||
from backend.services.backups import list_backup_files as service_list_backup_files
|
||||
@@ -281,7 +282,8 @@ class AdvancedSearchResultItem(BaseModel):
|
||||
path: str = Field(description="Relative file path")
|
||||
title: str = Field(description="File title")
|
||||
tags: list[str] = Field(description="File tags")
|
||||
score: float = Field(description="TF-IDF relevance score")
|
||||
score: float = Field(description="TF-IDF relevance score (or fused RRF score in semantic mode)")
|
||||
semantic_score: float = Field(default=0.0, description="Cosine similarity from the semantic index (0 when unavailable)")
|
||||
snippet: str = Field(description="Content excerpt with <mark> highlights")
|
||||
modified: str = Field(description="ISO 8601 modification timestamp")
|
||||
extension: str = Field(default="", description="File extension")
|
||||
@@ -301,6 +303,7 @@ class AdvancedSearchResponse(BaseModel):
|
||||
limit: int = Field(description="Page size")
|
||||
facets: SearchFacets = Field(description="Faceted counts by tag and vault")
|
||||
query_time_ms: float = Field(default=0, description="Server-side query time in milliseconds")
|
||||
semantic_available: bool = Field(default=False, description="True when the semantic (embedding) index is ready")
|
||||
|
||||
|
||||
class TitleSuggestion(BaseModel):
|
||||
@@ -694,6 +697,8 @@ async def lifespan(app: FastAPI):
|
||||
# would freeze HTTP responses if run in the async event loop.
|
||||
loop = asyncio.get_running_loop()
|
||||
await loop.run_in_executor(_search_executor, init_inverted_index)
|
||||
# Build the semantic (embedding) index in the same background thread pool.
|
||||
await loop.run_in_executor(_search_executor, init_semantic_index)
|
||||
|
||||
# Scan for plugins in all vaults
|
||||
logger.info("Scanning for plugins...")
|
||||
@@ -2589,6 +2594,7 @@ async def api_advanced_search(
|
||||
created: str | None = Query(None, description="Created date filter (>date, <date, date..date)"),
|
||||
modified: str | None = Query(None, description="Modified date filter (>date, <date, date..date, <Nd)"),
|
||||
size: str | None = Query(None, description="Size filter (>size, <size, size..size, e.g. >1MB, <10KB)"),
|
||||
semantic: bool = Query(False, description="Fuse TF-IDF with semantic embeddings (RRF)"),
|
||||
current_user=Depends(require_auth),
|
||||
):
|
||||
"""Advanced full-text search with TF-IDF scoring, facets, and pagination.
|
||||
@@ -2605,6 +2611,8 @@ async def api_advanced_search(
|
||||
- Remaining text is scored using TF-IDF with accent normalization.
|
||||
- Toggles: case_sensitive, whole_word, regex
|
||||
- Path filters: include_paths, exclude_paths (glob patterns)
|
||||
- ``semantic=true`` — fuse the TF-IDF ranking with the semantic (embedding)
|
||||
ranking via Reciprocal Rank Fusion and expose ``semantic_score`` per result.
|
||||
|
||||
Results include ``<mark>``-highlighted snippets and faceted tag/vault counts.
|
||||
"""
|
||||
@@ -2615,7 +2623,7 @@ async def api_advanced_search(
|
||||
limit=limit, offset=offset, sort=sort,
|
||||
case_sensitive=case_sensitive, whole_word=whole_word, regex=regex,
|
||||
include_paths=include_paths, exclude_paths=exclude_paths,
|
||||
created=created, modified=modified, size=size),
|
||||
created=created, modified=modified, size=size, semantic=semantic),
|
||||
)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user