- icône share-2 bleue sur les fichiers partagés dans l'arborescence
- dossier virtuel Partage/ Ã la racine du home du destinataire (lecture seule /s/{token})
- recherche plein texte couvre les documents reçus (merge search_vaults, share_token)
- cache front invalidé à la création/révocation ; +2 tests non-fuite
221 lines
6.9 KiB
Python
221 lines
6.9 KiB
Python
"""Search services shared by REST routes and the AI tool layer."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Callable
|
|
from typing import Any
|
|
|
|
|
|
def search_vaults(
|
|
q: str,
|
|
vault: str = "all",
|
|
tag: str | None = None,
|
|
limit: int = 50,
|
|
offset: int = 0,
|
|
is_allowed: Callable[[str], bool] | None = None,
|
|
username: str | None = None,
|
|
) -> dict[str, Any]:
|
|
"""Full-text search with pagination, returned as the API response payload.
|
|
|
|
``is_allowed`` (#194) filters the raw hits **before** pagination, so a
|
|
restricted vault set neither distorts ``total`` nor the returned page.
|
|
The tool layer filters on its own and leaves it as ``None``.
|
|
|
|
``username`` (#196) : les documents **reçus** par cet utilisateur via un
|
|
partage dirigé sont ajoutés aux résultats (lecture seule, sans indexer
|
|
les fichiers d'autrui — le contenu est lu à la volée et mis en cache
|
|
TF-IDF côté index, pas sur disque).
|
|
"""
|
|
from backend.search import search
|
|
|
|
all_results = search(q, vault_filter=vault, tag_filter=tag)
|
|
if is_allowed is not None:
|
|
all_results = [r for r in all_results if is_allowed(r.get("vault", ""))]
|
|
|
|
if username is not None:
|
|
all_results = all_results + _shared_results(username, q, vault)
|
|
|
|
total = len(all_results)
|
|
page = all_results[offset: offset + limit]
|
|
return {
|
|
"query": q,
|
|
"vault_filter": vault,
|
|
"tag_filter": tag,
|
|
"count": len(page),
|
|
"total": total,
|
|
"offset": offset,
|
|
"limit": limit,
|
|
"results": page,
|
|
}
|
|
|
|
|
|
def _shared_results(username: str, q: str, vault: str) -> list[dict[str, Any]]:
|
|
"""Search results for documents shared TO *username* (#196).
|
|
|
|
Reads the shared file content on the fly (read-only, source vault on
|
|
disk) and returns hits shaped exactly like ordinary search results so
|
|
the frontend can open them via the normal share page.
|
|
"""
|
|
from pathlib import Path
|
|
|
|
from backend.indexer import get_vault_data
|
|
from backend.services.paths import resolve_safe_path
|
|
from backend.share import list_shares
|
|
|
|
out: list[dict[str, Any]] = []
|
|
if not q:
|
|
return out
|
|
q_lower = q.lower()
|
|
|
|
for s in list_shares(user=username):
|
|
if s.get("created_by") == username:
|
|
continue # own share — already indexed in its own vault
|
|
if vault not in ("all", f"home-{username}"):
|
|
continue
|
|
data = get_vault_data(s["vault"])
|
|
if not data:
|
|
continue
|
|
try:
|
|
fp = resolve_safe_path(Path(data["path"]), s["path"])
|
|
if not fp.exists() or fp.suffix.lower() != ".md":
|
|
continue
|
|
raw = fp.read_text(encoding="utf-8", errors="replace")
|
|
except Exception:
|
|
continue
|
|
title = fp.stem
|
|
occurrences = raw.lower().count(q_lower)
|
|
if q_lower in title.lower():
|
|
occurrences += 1
|
|
if occurrences == 0:
|
|
continue
|
|
out.append({
|
|
"vault": f"home-{username}",
|
|
"path": f"Partage/{fp.name}",
|
|
"title": title,
|
|
"tags": [],
|
|
"score": min(occurrences, 10),
|
|
"snippet": _snippet(raw, q_lower),
|
|
"modified": None,
|
|
"share_token": s["token"],
|
|
})
|
|
return out
|
|
|
|
|
|
def _snippet(text: str, q_lower: str, width: int = 160) -> str:
|
|
"""Small excerpt around the first match (#196, shared-file search)."""
|
|
idx = text.lower().find(q_lower)
|
|
if idx < 0:
|
|
return text[:width]
|
|
start = max(0, idx - width // 2)
|
|
return "…" + text[start:start + width].replace("\n", " ") + "…"
|
|
|
|
|
|
def list_tags(vault: str | None = None) -> dict[str, int]:
|
|
"""Return tag → count, optionally restricted to a single vault."""
|
|
from backend.search import get_all_tags
|
|
|
|
return get_all_tags(vault_filter=vault)
|
|
|
|
|
|
def advanced_search_vaults(
|
|
query: str = "",
|
|
vault: str = "all",
|
|
tag: str | None = None,
|
|
limit: int = 50,
|
|
offset: int = 0,
|
|
sort: str = "relevance",
|
|
case_sensitive: bool = False,
|
|
whole_word: bool = False,
|
|
regex: bool = False,
|
|
include_paths: str | None = None,
|
|
exclude_paths: str | None = None,
|
|
created: str | None = None,
|
|
modified: str | None = None,
|
|
size: str | None = None,
|
|
semantic: bool = False,
|
|
) -> dict[str, Any]:
|
|
"""Advanced full-text search (TF-IDF, facets, operators).
|
|
|
|
When ``semantic`` is True, the TF-IDF ranking is fused with the embedding
|
|
(semantic) ranking via RRF. No permission filtering is applied: callers
|
|
that need it (the tool layer) filter the ``results`` list themselves.
|
|
"""
|
|
from backend.search import advanced_search
|
|
|
|
return advanced_search(
|
|
query,
|
|
vault_filter=vault,
|
|
tag_filter=tag,
|
|
limit=limit,
|
|
offset=offset,
|
|
sort_by=sort,
|
|
case_sensitive=case_sensitive,
|
|
whole_word=whole_word,
|
|
regex=regex,
|
|
include_paths=include_paths,
|
|
exclude_paths=exclude_paths,
|
|
created=created,
|
|
modified=modified,
|
|
size=size,
|
|
semantic=semantic,
|
|
)
|
|
|
|
|
|
def search_paths(q: str, vault: str = "all") -> dict[str, Any]:
|
|
"""Search files and directories by path substring using the path index.
|
|
|
|
No permission filtering is applied: callers that need it (the tool layer)
|
|
filter the ``results`` list themselves.
|
|
"""
|
|
from backend.indexer import path_index
|
|
|
|
if not q:
|
|
return {"query": q, "vault_filter": vault, "results": []}
|
|
|
|
query_lower = q.lower()
|
|
results: list[dict[str, Any]] = []
|
|
|
|
vaults_to_search = [vault] if vault != "all" else list(path_index.keys())
|
|
|
|
for vault_name in vaults_to_search:
|
|
for entry in path_index.get(vault_name, []):
|
|
if query_lower in entry["name"].lower() or query_lower in entry["path"].lower():
|
|
results.append({
|
|
"vault": vault_name,
|
|
"path": entry["path"],
|
|
"name": entry["name"],
|
|
"type": entry["type"],
|
|
"matched_path": entry["path"],
|
|
})
|
|
|
|
return {"query": q, "vault_filter": vault, "results": results}
|
|
|
|
|
|
def list_paths(vault: str, limit: int = 5000) -> dict[str, Any]:
|
|
"""Return a flat, capped list of every indexed path in a vault.
|
|
|
|
Backs the AI assistant ``@`` mention menu: fetching the whole path index
|
|
once lets the client filter files/directories instantly instead of issuing
|
|
a request per keystroke.
|
|
|
|
Args:
|
|
vault: Vault name.
|
|
limit: Maximum number of entries returned.
|
|
|
|
Returns:
|
|
``{"vault", "count", "results": [{vault, path, name, type}]}``.
|
|
"""
|
|
from backend.indexer import path_index
|
|
|
|
entries = path_index.get(vault, [])
|
|
results = [
|
|
{
|
|
"vault": vault,
|
|
"path": entry["path"],
|
|
"name": entry["name"],
|
|
"type": entry["type"],
|
|
}
|
|
for entry in entries[: max(0, limit)]
|
|
]
|
|
return {"vault": vault, "count": len(results), "results": results}
|