feat(assistant): #91 steps humains en direct + outils web_search/fetch_url (labels backend, SSE step, garde SSRF)
CI / lint (push) Successful in 1m22s
CI / security (push) Successful in 55s
CI / test (push) Successful in 3m0s
CI / build (push) Successful in 51s
CI / e2e (push) Successful in 10m41s

This commit is contained in:
2026-09-15 17:46:51 -04:00
parent 18b7dee2f6
commit 19d5b931c4
14 changed files with 745 additions and 32 deletions
+9 -2
View File
@@ -52,9 +52,16 @@ et [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
**discret** au-dessus de chaque réponse rapporte le fournisseur et le modèle
réellement utilisés (SSE `provider`/`model`) ; une **barre d'actions** au survol
sous chaque bloc : « Copier » (texte brut + toast) et, pour les réponses,
« Ajouter » (insertion dans le document ouvert dans l'éditeur Forge). Fiche :
« Ajouter » (insertion dans le document ouvert dans l'éditeur Forge).
Complément : la section « N étapes » affiche désormais des libellés humains
produits par le backend (`backend/tools/labels.py`, événements SSE en direct
pendant l'exécution + ligne « Réflexion »), et deux nouveaux outils web
principaux sont disponibles en mode agent : `web_search` (SearXNG
auto-hébergé) et `fetch_url` (page publique → texte, garde SSRF). Le reste
des catégories d'outils Notion est documenté pour le futur dans la fiche.
Fiche :
[docs/features/ai-assistant-conversation-ux.md](./docs/features/ai-assistant-conversation-ux.md)
Tests : `tests/frontend/ai.test.mjs` (74).
Tests : `tests/frontend/ai.test.mjs` (76), `tests/test_tool_labels.py` (8), `tests/test_web_tools.py` (8).
### Corrigé
+33 -2
View File
@@ -30,6 +30,7 @@ from backend.tools.api import (
call_tool,
get_tool_schemas,
)
from backend.tools.labels import thought_step_label, tool_step_label
logger = logging.getLogger("obsigate.agent.loop")
@@ -54,6 +55,9 @@ class ToolCallRecord:
arguments: dict[str, Any]
ok: bool
result: Any
# Human-readable « step » label for the Notion-style UI
# ({key, params} — see backend.tools.labels).
step: dict[str, Any] = field(default_factory=dict)
@dataclass
@@ -63,6 +67,9 @@ class AgentResult:
content: str = ""
messages: list[dict[str, Any]] = field(default_factory=list)
tool_calls: list[ToolCallRecord] = field(default_factory=list)
# Ordered Notion-style step descriptors ({key, params}); tool steps and
# intermediate reasoning notes interleaved by execution order.
steps: list[dict[str, Any]] = field(default_factory=list)
iterations: int = 0
stopped: str = STOP_DONE
pending: dict[str, Any] | None = None
@@ -142,7 +149,10 @@ def _execute_confirmed(
payload = e.to_dict()
ok = False
record = ToolCallRecord(name=name, arguments=arguments, ok=ok, result=payload)
record = ToolCallRecord(
name=name, arguments=arguments, ok=ok, result=payload,
step=tool_step_label(name, arguments),
)
executed.append(record)
if on_tool_call is not None:
on_tool_call(record)
@@ -164,6 +174,7 @@ async def run_agent(
max_iterations: int = DEFAULT_MAX_ITERATIONS,
max_tool_calls: int | None = None,
on_tool_call: Callable[[ToolCallRecord], None] | None = None,
on_thought: Callable[[dict[str, Any]], None] | None = None,
resume_messages: list[dict[str, Any]] | None = None,
confirm_pending: dict[str, Any] | None = None,
) -> AgentResult:
@@ -195,6 +206,15 @@ async def run_agent(
if tools is None:
tools = get_tool_schemas(scope=ToolScope.IN_APP)
quota = DEFAULT_MAX_TOOL_CALLS if max_tool_calls is None else max_tool_calls
steps: list[dict[str, Any]] = []
def _emit_note(text: str) -> None:
"""Record an intermediate reasoning note as a visible step."""
note = thought_step_label(text)
if note["params"]["value"]:
steps.append(note)
if on_thought is not None:
on_thought(note)
convo = [dict(m) for m in (resume_messages if resume_messages is not None else messages)]
executed: list[ToolCallRecord] = []
@@ -205,6 +225,7 @@ async def run_agent(
content="",
messages=convo,
tool_calls=executed,
steps=steps,
iterations=0,
stopped=STOP_QUOTA_EXCEEDED,
)
@@ -218,10 +239,13 @@ async def run_agent(
content=response.content or "",
messages=convo,
tool_calls=executed,
steps=steps,
iterations=iteration,
stopped=STOP_DONE,
)
# Intermediate reasoning shown alongside tool calls → a "thought" step.
_emit_note(response.content or "")
convo.append(_assistant_tool_message(response.content, response.tool_calls))
for call in response.tool_calls:
@@ -231,6 +255,7 @@ async def run_agent(
content=response.content or "",
messages=convo,
tool_calls=executed,
steps=steps,
iterations=iteration,
stopped=STOP_QUOTA_EXCEEDED,
)
@@ -247,6 +272,7 @@ async def run_agent(
content=response.content or "",
messages=convo,
tool_calls=executed,
steps=steps,
iterations=iteration,
stopped=STOP_CONFIRMATION_REQUIRED,
pending=pending,
@@ -255,8 +281,12 @@ async def run_agent(
payload = e.to_dict()
ok = False
record = ToolCallRecord(name=call.name, arguments=call.arguments, ok=ok, result=payload)
record = ToolCallRecord(
name=call.name, arguments=call.arguments, ok=ok, result=payload,
step=tool_step_label(call.name, call.arguments),
)
executed.append(record)
steps.append(record.step)
if on_tool_call is not None:
on_tool_call(record)
@@ -272,6 +302,7 @@ async def run_agent(
content="",
messages=convo,
tool_calls=executed,
steps=steps,
iterations=max_iterations,
stopped=STOP_MAX_ITERATIONS,
)
+48 -8
View File
@@ -317,6 +317,21 @@ def _resolve_provider_name(requested: str | None) -> str | None:
return None
def _tool_event_sse(rec) -> str:
"""Serialize one executed tool call as an SSE ``tool`` event."""
payload = json.dumps(
{"name": rec.name, "ok": rec.ok, "arguments": rec.arguments, "step": rec.step},
ensure_ascii=False,
)
return f"event: tool\ndata: {payload}\n\n"
def _thought_event_sse(note: dict) -> str:
"""Serialize one intermediate reasoning note as an SSE ``step`` event."""
payload = json.dumps({"step": note}, ensure_ascii=False)
return f"event: step\ndata: {payload}\n\n"
# ── Endpoints ──
@@ -456,6 +471,8 @@ async def api_bookslm_agent(
)
async def generate_sse():
import asyncio
try:
cfg_name = _resolve_provider_name(req.provider)
if cfg_name is None:
@@ -466,20 +483,43 @@ async def api_bookslm_agent(
yield f"event: error\ndata: {error_data}\n\n"
return
result = await run_agent(
# Stream tool events live: each executed step is pushed on the
# queue by the loop callback and emitted as soon as it happens,
# so the UI can grow its « N steps » block while thinking.
queue: asyncio.Queue = asyncio.Queue()
def _on_tool(rec) -> None:
queue.put_nowait(("tool", rec))
run_task = asyncio.create_task(run_agent(
messages,
ctx=ctx,
llm=_llm,
resume_messages=req.confirm_messages,
confirm_pending=req.confirm,
)
on_tool_call=_on_tool,
on_thought=lambda note: queue.put_nowait(("thought", note)),
))
for rec in result.tool_calls:
payload = json.dumps(
{"name": rec.name, "ok": rec.ok, "arguments": rec.arguments},
ensure_ascii=False,
)
yield f"event: tool\ndata: {payload}\n\n"
# Drain every completed step as soon as it lands, while the agent
# keeps running in the background. If the client disconnects, the
# generator is cancelled: release the run so it cannot orphan.
try:
while True:
try:
kind, item = await asyncio.wait_for(queue.get(), timeout=0.25)
except asyncio.TimeoutError:
if run_task.done():
break
continue
yield _tool_event_sse(item) if kind == "tool" else _thought_event_sse(item)
while not queue.empty():
kind, item = queue.get_nowait()
yield _tool_event_sse(item) if kind == "tool" else _thought_event_sse(item)
result = run_task.result()
finally:
if not run_task.done():
run_task.cancel()
if result.stopped == "confirmation_required":
pending = json.dumps(
+1
View File
@@ -10,6 +10,7 @@ which ``.gitignore`` excludes via ``_*.py``), hence this explicit facade.
"""
from backend.tools import service as _service # noqa: F401 (registers tools)
from backend.tools import web as _web # noqa: F401 (registers web tools)
from backend.tools.context import (
ToolConfirmationRequired,
ToolContext,
+89
View File
@@ -0,0 +1,89 @@
"""Human-readable labels for tool calls (Notion-style “steps” section).
The assistant's conversation window shows a collapsible « N steps » block
above each answer. Raw tool names (``search_fulltext``) are developer-centric;
this module maps every registered tool to an i18n message key plus the
argument worth surfacing (path, query, …), so the UI can render sentences like
« Recherche dans le vault : pizza » or « Fichier lu : notes/x.md ».
The label KEY is emitted with each ``tool`` SSE event; the client resolves it
against its locale dictionaries (``ai.step.<key>``). Unknown tools fall back to
a generic key with the raw name prettified, so a new tool never breaks the UI.
"""
from __future__ import annotations
from typing import Any
# tool name -> (message-key, primary argument name or None)
# NOTE: argument names MUST match the pydantic input models in
# backend/tools/schemas.py (verified by tests/test_tool_labels.py).
_STEP_LABELS: dict[str, tuple[str, str | None]] = {
"list_vaults": ("vaults", None),
"list_directory": ("directory", "path"),
"list_all_files": ("files", "dir"),
"read_file": ("file_read", "path"),
"read_file_raw": ("file_read", "path"),
"get_backlinks": ("backlinks", "path"),
"list_backups": ("backups", "path"),
"diff_backup": ("backup_diff", "path"),
"get_graph": ("graph", None),
"search_fulltext": ("search", "q"),
"search_advanced": ("search", "q"),
"search_paths": ("search_paths", "q"),
"list_tags": ("tags", None),
"suggest_tags": ("tags_suggest", "q"),
"list_recent": ("recent", None),
"create_file": ("file_create", "path"),
"create_directory": ("dir_create", "path"),
"edit_file": ("file_edit", "path"),
"append_to_file": ("file_append", "path"),
"rename_file": ("file_rename", "path"),
"rename_directory": ("dir_rename", "path"),
"move_path": ("move", "source_path"),
"replace_in_files": ("replace", "find"),
"delete_file": ("file_delete", "path"),
"delete_directory": ("dir_delete", "path"),
"restore_backup": ("backup_restore", "path"),
"web_search": ("web_search", "query"),
"fetch_url": ("fetch_url", "url"),
}
GENERIC_KEY = "generic"
def _prettify(name: str) -> str:
return name.replace("_", " ").strip()
def _primary_value(arguments: dict[str, Any], arg: str | None) -> str | None:
if not arg:
return None
value = arguments.get(arg)
if isinstance(value, str) and value.strip():
return value.strip()
return None
def tool_step_label(name: str, arguments: dict[str, Any] | None = None) -> dict[str, Any]:
"""Return ``{key, params}`` describing one tool call in human terms.
``key`` resolves client-side to ``ai.step.<key>``; ``params`` carries the
optional ``{value}`` placeholder (path, query, …). For an unknown tool the
generic key is used with the prettified name as the value.
"""
arguments = arguments or {}
key, arg = _STEP_LABELS.get(name, (GENERIC_KEY, None))
if key == GENERIC_KEY:
return {"key": GENERIC_KEY, "params": {"tool": _prettify(name)}}
value = _primary_value(arguments, arg)
params: dict[str, str] = {}
if value is not None:
params["value"] = value
return {"key": key, "params": params}
def thought_step_label(text: str) -> dict[str, Any]:
"""Step descriptor for one intermediate reasoning note of the model."""
cleaned = " ".join((text or "").split())
return {"key": "thought", "params": {"value": cleaned[:400]}}
+16
View File
@@ -237,6 +237,22 @@ class RestoreBackupInput(BaseModel):
version: int = Field(..., description="Backup timestamp to restore")
class WebSearchInput(BaseModel):
"""Search the public web through the configured meta-search engine."""
query: str = Field(..., description="Search terms")
max_results: int = Field(5, ge=1, le=10, description="Number of results to return")
category: str = Field("", description="Optional engine category: general, news, it, science")
language: str = Field("", description="Optional language code, e.g. 'fr'")
page: int = Field(1, ge=1, le=10, description="Result page number")
class FetchUrlInput(BaseModel):
"""Fetch one public web page and return its readable text."""
url: str = Field(..., description="Absolute http(s) URL of a public page")
class ToolResult(BaseModel):
"""Uniform result returned by :func:`backend.tools.registry.call_tool`."""
+217
View File
@@ -0,0 +1,217 @@
"""Web tools for the assistant (Notion-style "research" capabilities).
Phase 1 of the documented web-toolset roadmap:
* ``web_search`` — query the self-hosted SearXNG instance (no API key).
* ``fetch_url`` — retrieve a public web page and return readable text.
Both are READ-risk tools (no confirmation), rate-limited through the shared
registry, SSRF-guarded (scheme + private-address rejection), and size-capped.
Configuration (environment):
* ``OBSIGATE_SEARXNG_URL`` — defaults to https://search.dracodev.net
* ``OBSIGATE_WEB_TIMEOUT`` — seconds, default 10
"""
from __future__ import annotations
import html as html_lib
import ipaddress
import logging
import os
import re
import socket
from typing import Any
from urllib.parse import urlparse
import httpx
from backend.tools.context import ToolError, ToolRisk, ToolScope
from backend.tools.registry import tool
from backend.tools.schemas import FetchUrlInput, WebSearchInput
logger = logging.getLogger("obsigate.tools.web")
SEARXNG_URL = os.environ.get("OBSIGATE_SEARXNG_URL", "https://search.dracodev.net")
WEB_TIMEOUT = float(os.environ.get("OBSIGATE_WEB_TIMEOUT", "10"))
USER_AGENT = "ObsiGateAssistant/1.0 (+self-hosted vault AI)"
MAX_FETCH_BYTES = 1_500_000
MAX_TEXT_CHARS = 20_000
_BLOCKED_TAGS_RE = re.compile(
r"<(script|style|noscript|template|svg)\b.*?</\1>", re.IGNORECASE | re.DOTALL
)
_TAG_RE = re.compile(r"<[^>]+>")
_BLOCK_SPLIT_RE = re.compile(
r"</?(?:p|div|br|li|h[1-6]|tr|table|ul|ol|section|article|header|footer)\b[^>]*>",
re.IGNORECASE,
)
class SSRFError(ToolError):
"""Raised for a URL whose host is private/loopback or scheme unsupported."""
def _assert_public_http_url(url: str) -> str:
"""Reject non-http(s) schemes and private/loopback/link-local targets."""
try:
parsed = urlparse(url)
except ValueError as e:
raise SSRFError("URL invalide", code="invalid_url") from e
if parsed.scheme not in ("http", "https"):
raise SSRFError("Seuls les schémas http/https sont autorisés", code="invalid_scheme")
host = parsed.hostname
if not host:
raise SSRFError("URL sans hôte", code="invalid_url")
# Resolve the host so DNS-rebinding to internal IPs is also caught.
try:
infos = socket.getaddrinfo(host, None)
except socket.gaierror as e:
raise SSRFError(f"Hôte introuvable: {host}", code="dns_error") from e
for info in infos:
ip = ipaddress.ip_address(info[4][0])
if (
ip.is_private
or ip.is_loopback
or ip.is_link_local
or ip.is_reserved
or ip.is_multicast
or ip.is_unspecified
):
raise SSRFError("Accès aux adresses internes interdit", code="ssrf_blocked")
return url
def _html_to_text(raw: str) -> str:
"""Cheap HTML → readable text: strip scripts/styles, tags, then compress.
Comments (which may carry script-like payloads) are removed first.
"""
text = re.sub(r"<!--.*?-->", " ", raw, flags=re.DOTALL)
text = _BLOCKED_TAGS_RE.sub(" ", text)
# Keep block boundaries as newlines before dropping the remaining tags.
text = _BLOCK_SPLIT_RE.sub("\n", text)
text = _TAG_RE.sub("", text)
text = html_lib.unescape(text)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r" ?\n ?", "\n", text)
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
@tool(
name="web_search",
description=(
"Search the public web for current information and return ranked results "
"(title, url, snippet). Use for facts outside the vault: weather, news, "
"documentation, versions, prices, anything that needs live sources."
),
input_model=WebSearchInput,
risk=ToolRisk.READ,
scopes=(ToolScope.IN_APP,),
)
def web_search(ctx, params: WebSearchInput) -> dict[str, Any]:
"""Query the self-hosted SearXNG instance and return trimmed results."""
query = params.query.strip()
if not query:
raise ToolError("Requête vide", code="invalid_arguments")
url = SEARXNG_URL.rstrip("/") + "/search"
try:
resp = httpx.get(
url,
params={
"q": query,
"format": "json",
"categories": params.category or "general",
"pageno": max(1, params.page),
**({"language": params.language} if params.language else {}),
"safesearch": "1",
},
headers={"User-Agent": USER_AGENT},
timeout=WEB_TIMEOUT,
follow_redirects=False,
)
resp.raise_for_status()
data = resp.json()
except httpx.HTTPError as e:
logger.warning("web_search failed: %s", e)
raise ToolError(
"Le moteur de recherche web est momentanément indisponible.",
code="web_search_unavailable",
) from e
results: list[dict[str, Any]] = []
for item in (data.get("results") or [])[: params.max_results]:
results.append(
{
"title": (item.get("title") or "")[:300],
"url": item.get("url") or "",
"snippet": (item.get("content") or "")[:600],
"published": item.get("publishedDate"),
"score": item.get("score"),
}
)
return {
"query": query,
"engine": "searxng",
"results": results,
"count": len(results),
}
@tool(
name="fetch_url",
description=(
"Fetch a public web page (http/https) and return its readable text. "
"Use after web_search to read a promising result in detail. HTML is "
"converted to plain text; binary pages are rejected."
),
input_model=FetchUrlInput,
risk=ToolRisk.READ,
scopes=(ToolScope.IN_APP,),
)
def fetch_url(ctx, params: FetchUrlInput) -> dict[str, Any]:
"""Retrieve one page, guard against SSRF, and extract its text."""
url = _assert_public_http_url(params.url.strip())
try:
# Follow redirects manually so every hop is re-checked against the
# private-address SSRF guard (a public page can redirect to 127.0.0.1).
resp = None
for _hop in range(5):
resp = httpx.get(
url,
headers={"User-Agent": USER_AGENT, "Accept": "text/html,application/xhtml+xml,*/*"},
timeout=WEB_TIMEOUT,
follow_redirects=False,
)
if resp.status_code in (301, 302, 303, 307, 308):
location = resp.headers.get("location") or ""
if not location:
break
url = str(httpx.URL(url).join(location))
url = _assert_public_http_url(url)
continue
break
assert resp is not None
resp.raise_for_status()
except SSRFError:
raise
except httpx.HTTPError as e:
logger.warning("fetch_url failed for %s: %s", url, e)
raise ToolError("Impossible de récupérer la page.", code="fetch_unavailable") from e
ctype = (resp.headers.get("content-type") or "").lower()
if not any(t in ctype for t in ("html", "xml", "text", "json", "markdown")):
raise ToolError(
f"Type de contenu non pris en charge: {ctype.split(';')[0] or 'inconnu'}",
code="unsupported_content_type",
)
raw = (resp.content[:MAX_FETCH_BYTES]).decode(resp.encoding or "utf-8", errors="replace")
title_match = re.search(r"<title[^>]*>(.*?)</title>", raw, re.IGNORECASE | re.DOTALL)
title = html_lib.unescape(title_match.group(1)).strip()[:300] if title_match else ""
text = _html_to_text(raw)[:MAX_TEXT_CHARS]
return {
"url": str(resp.url),
"status": resp.status_code,
"title": title,
"text": text,
"truncated": len(raw) > MAX_TEXT_CHARS,
}
+37 -1
View File
@@ -80,4 +80,40 @@
- L'ancien épinglage flex `order: -1` (première itération de #91) est **remplacé** :
l'ordre est redevenu chronologique, l'ancrage est fait par défilement explicite.
- Les pouces haut/bas Notion ne sont **pas** repris : aucun signal n'existe
côté backend ; « Ajouter » (insertion éditeur réelle) les remplace utilement.
côté backend ; « Ajouter » (insertion éditeur réelle) les remplace utilement.
## G. Section « steps » enrichie + outils web (complément #91) — ✅ livré
- [x] **G1.** Chaque étape est décrite par le **backend** (`backend/tools/labels.py`) :
événement SSE `tool` avec `step {key, params}` ; le frontend résout
`ai.step.<key>` (FR/EN) → phrases humaines : « Recherche dans le vault : pizza »,
« Fichier lu : notes/a.md », etc. Outil inconnu → libellé générique (jamais cassé).
- [x] **G2.** Événements en **direct** : les steps sont poussés sur une file
`asyncio.Queue` par la boucle agent et émis dès leur exécution — le bloc
« N étapes » grandit pendant que l'assistant travaille (plus de liste fin de run).
- [x] **G3.** Ligne « Réflexion » (`ai.step.thought`) : quand le modèle émet un
texte intermédiaire avec ses appels d'outils, il est publié comme event SSE
`step` et affiché dans la liste (`. thought` Notion).
- [x] **G4.** Nouveaux outils principaux côté **web** (`backend/tools/web.py`,
risques READ, scope IN_APP, SSRF-guard + limite de taille) :
`web_search` (SearXNG auto-hébergé, configurable via `OBSIGATE_SEARXNG_URL`)
et `fetch_url` (lecture d'une page publique, HTML → texte).
Steps affichés : « Recherche sur le web : … » / « Page web consultée : … ».
### Outils restants — documentés pour le futur (hors #91)
La catégorie Notion « étapes » peut s'étendre ; chaque futur outil devra être un
tool du registre (`@tool`) + un libellé dans `labels.py` + deux clés i18n.
Priorités proposées (à transformer en items `#NN` quand implémentés) :
- **Créer/supprimer un fichier depuis une réponse** : déjà couvert par les cartes
d'action `create_file` / l'outil `create_file` — exposer une action « Créer la
note » dans la barre d'actions (post-traitement du texte copié).
- **Insérer dans le document courant** : l'équivalent « Ajouter » côté assistant
est livré (G) ; l'export vers une sélection précise de l'éditeur reste possible.
- **Convertir un format / tableur / document** : outils `convert_format`,
`create_spreadsheet` (csv/xlsx via backend) — risque WRITE (confirmation).
- **Calendrier / tâches** : intégrer l'API existante `n8n-automation` ou un
serveur MCP externe (la porte MCP est déjà ouverte côté ObsiGate).
- **Courriel / messagerie** : via le serveur MCP externe (Gmail/IMPT SMTP) —
jamais de clé en dur : passer par Infisical.
- **Sources connectées (Slack, GitHub, Drive)** : uniquement par **MCP externe**
(`docs/features/ai-tools-mcp.md`), pas d'outils natifs — l'attaque surface
reste dans le registre + rate-limit + audit.
+41 -16
View File
@@ -2058,22 +2058,35 @@ class BooksLM {
// ── Tool activity & confirmations (agent mode) ──────────────────────
/** Agent tool calls collapse into a discreet “N steps” line (Notion-style). */
_renderToolActivity(toolCalls) {
const wrap = document.createElement('details');
wrap.className = 'bookslm-tool-trace';
const summary = document.createElement('summary');
summary.textContent = t('ai.steps_count', { count: toolCalls.length });
wrap.appendChild(summary);
for (const call of toolCalls) {
const line = document.createElement('div');
line.className = 'bookslm-tool-line' + (call.ok === false ? ' failed' : '');
line.textContent = `${call.ok === false ? '⚠' : '🔧'} ${t('ai.tool_call', { name: call.name })}`;
if (call.ok === false) line.title = t('ai.tool_call_failed', { name: call.name });
wrap.appendChild(line);
/** Human one-liner for a recorded step (backend label key or legacy name). */
_stepText(call) {
const step = call.step;
if (step && step.key) {
const key = `ai.step.${step.key}`;
const label = t(key, step.params || {});
if (label && label !== key) return label;
}
return t('ai.tool_call', { name: call.name });
}
/** Agent tool calls collapse into a discreet “N steps” line (Notion-style). */
_renderToolActivity(toolCalls) {
const wrap = document.createElement('details');
wrap.className = 'bookslm-tool-trace';
const summary = document.createElement('summary');
const countKey = toolCalls.length > 1 ? 'ai.steps_count_plural' : 'ai.steps_count';
summary.textContent = t(countKey, { count: toolCalls.length });
wrap.appendChild(summary);
for (const call of toolCalls) {
const line = document.createElement('div');
line.className = 'bookslm-tool-line' + (call.ok === false ? ' failed' : '');
line.textContent = `${call.ok === false ? '⚠' : '·'} ${this._stepText(call)}`;
if (call.ok === false) line.title = t('ai.tool_call_failed', { name: call.name });
else if (call.step && call.step.params && call.step.params.value) line.title = call.step.params.value;
wrap.appendChild(line);
}
return wrap;
}
return wrap;
}
_renderConfirmationCard(msg) {
const conf = msg.confirmation;
@@ -2393,10 +2406,22 @@ class BooksLM {
}
if (currentEvent === 'tool') {
assistantMsg.toolCalls.push({ name: data.name, ok: data.ok !== false, arguments: data.arguments || {} });
assistantMsg.toolCalls.push({
name: data.name,
ok: data.ok !== false,
arguments: data.arguments || {},
step: data.step || null,
});
this._setActivity('working', t('ai.activity_tool', { name: data.name }));
return;
}
// Intermediate reasoning note streamed by the agent (Notion-style
// ". thought" line inside the steps block).
if (currentEvent === 'step' && data.step) {
assistantMsg.toolCalls.push({ name: '', ok: true, arguments: {}, step: data.step });
this._setActivity('working', t('ai.activity_thinking'));
return;
}
if (currentEvent === 'confirmation') {
assistantMsg.confirmation = {
pending: data.pending || data,
+31 -1
View File
@@ -1780,7 +1780,37 @@
"bookslm.insert_hint": "Append the answer to the document open in the editor",
"bookslm.inserted": "Answer added to the document",
"bookslm.insert_no_editor": "No document open in the editor",
"ai.steps_count": "{count} steps",
"ai.steps_count": "{count} step",
"ai.steps_count_plural": "{count} steps",
"ai.activity_thinking": "Thinking…",
"ai.step.thought": "Thought: {value}",
"ai.step.backlinks": "Analyzed backlinks",
"ai.step.backup_diff": "Compared backup: {value}",
"ai.step.backup_restore": "Restored backup: {value}",
"ai.step.backups": "Listed backups",
"ai.step.dir_create": "Proposed directory: {value}",
"ai.step.dir_delete": "Deleted directory: {value}",
"ai.step.dir_rename": "Renamed directory: {value}",
"ai.step.directory": "Explored directory: {value}",
"ai.step.file_append": "Appended to file: {value}",
"ai.step.file_create": "Proposed file: {value}",
"ai.step.file_delete": "Deleted file: {value}",
"ai.step.file_edit": "Edited file: {value}",
"ai.step.file_read": "Read file: {value}",
"ai.step.file_rename": "Renamed file: {value}",
"ai.step.files": "Listed files",
"ai.step.generic": "Used tool: {tool}",
"ai.step.graph": "Consulted the vault graph",
"ai.step.move": "Moved: {value}",
"ai.step.recent": "Listed recent files",
"ai.step.replace": "Replacements: {value}",
"ai.step.search": "Searched the vault: {value}",
"ai.step.search_paths": "Searched paths: {value}",
"ai.step.tags": "Listed tags",
"ai.step.tags_suggest": "Suggested tags for {value}",
"ai.step.vaults": "Listed the vaults",
"ai.step.fetch_url": "Opened a web page: {value}",
"ai.step.web_search": "Searched the web: {value}",
"bookslm.copied": "Copied to clipboard",
"bookslm.error": "AI service error",
"bookslm.regenerate": "Regenerate",
+31 -1
View File
@@ -1780,7 +1780,37 @@
"bookslm.insert_hint": "Ajouter la réponse au document ouvert dans l'éditeur",
"bookslm.inserted": "Réponse ajoutée au document",
"bookslm.insert_no_editor": "Aucun document ouvert dans l'éditeur",
"ai.steps_count": "{count} étapes",
"ai.steps_count": "{count} étape",
"ai.steps_count_plural": "{count} étapes",
"ai.activity_thinking": "Réflexion…",
"ai.step.thought": "Réflexion : {value}",
"ai.step.backlinks": "Backlinks analysés",
"ai.step.backup_diff": "Comparaison de sauvegarde : {value}",
"ai.step.backup_restore": "Sauvegarde restaurée : {value}",
"ai.step.backups": "Sauvegardes consultées",
"ai.step.dir_create": "Dossier proposé : {value}",
"ai.step.dir_delete": "Dossier supprimé : {value}",
"ai.step.dir_rename": "Dossier renommé : {value}",
"ai.step.directory": "Répertoire exploré : {value}",
"ai.step.file_append": "Fichier complété : {value}",
"ai.step.file_create": "Fichier proposé : {value}",
"ai.step.file_delete": "Fichier supprimé : {value}",
"ai.step.file_edit": "Fichier modifié : {value}",
"ai.step.file_read": "Fichier lu : {value}",
"ai.step.file_rename": "Fichier renommé : {value}",
"ai.step.files": "Fichiers listés",
"ai.step.generic": "Outil utilisé : {tool}",
"ai.step.graph": "Graphe du vault consulté",
"ai.step.move": "Élément déplacé : {value}",
"ai.step.recent": "Fichiers récents consultés",
"ai.step.replace": "Remplacements : {value}",
"ai.step.search": "Recherche dans le vault : {value}",
"ai.step.search_paths": "Chemins recherchés : {value}",
"ai.step.tags": "Tags consultés",
"ai.step.tags_suggest": "Tags suggérés pour {value}",
"ai.step.vaults": "Liste des vaults consultée",
"ai.step.fetch_url": "Page web consultée : {value}",
"ai.step.web_search": "Recherche sur le web : {value}",
"bookslm.copied": "Réponse copiée dans le presse-papiers",
"bookslm.error": "Erreur du service AI",
"bookslm.regenerate": "Régénérer",
+13 -1
View File
@@ -335,7 +335,7 @@ async function main() {
await test("agent tool calls collapse into a discreet steps block", () => {
const b = new BooksLM();
const trace = b._renderToolActivity([
{ name: "read_file", ok: true },
{ name: "read_file", ok: true, step: { key: "file_read", params: { value: "a.md" } } },
{ name: "search", ok: true },
]);
assert.equal(trace.tagName, "DETAILS", "collapsible details element");
@@ -345,6 +345,18 @@ async function main() {
assert.equal(trace.querySelectorAll(".bookslm-tool-line").length, 2, "each step listed inside");
});
await test("_stepText uses the backend label key, falls back to tool name", () => {
const b = new BooksLM();
// No locales loaded in the JSDOM harness → t() returns the key, so the
// fallback path (legacy name) is what surfaces; assert it never throws
// and produces a non-empty string for both shapes.
const withStep = b._stepText({ name: "read_file", step: { key: "file_read", params: { value: "a.md" } } });
const legacy = b._stepText({ name: "read_file" });
assert.equal(typeof withStep, "string");
assert.equal(typeof legacy, "string");
assert.ok(legacy.length > 0);
});
// ── 5. Source guards: no window.prompt() and authenticated BooksLM ──
await test("ai.js no longer uses window.prompt()", async () => {
const { readFileSync } = await import("node:fs");
+67
View File
@@ -0,0 +1,67 @@
"""Unit tests for the Notion-style step labels (backend.tools.labels)."""
from backend.agent.loop import ToolCallRecord
from backend.tools.labels import GENERIC_KEY, _STEP_LABELS, tool_step_label
from backend.tools.registry import list_tools
from backend.tools.context import ToolScope
class TestToolStepLabel:
def test_every_in_app_tool_has_a_label(self):
# A registered tool without a mapping still degrades gracefully to the
# generic key, but the curated coverage must stay complete for the UI.
names = {spec.name for spec in list_tools(scope=ToolScope.IN_APP)}
assert names.issubset(set(_STEP_LABELS)), names - set(_STEP_LABELS)
def test_label_argument_names_match_input_models(self):
# A mapped primary argument that the tool's input model does not
# actually declare would silently render "{value}" placeholders —
# verify every (name, arg) pair against the real schemas.
from backend.tools.registry import get_tool
for name, (_key, arg) in _STEP_LABELS.items():
if arg is None:
continue
spec = get_tool(name)
assert spec is not None, name
fields = spec.input_model.model_fields
assert arg in fields, f"{name}: input model has no field '{arg}'"
def test_named_tool_returns_key_and_value(self):
label = tool_step_label("read_file", {"vault": "V", "path": "notes/a.md"})
assert label["key"] == "file_read"
assert label["params"]["value"] == "notes/a.md"
def test_search_tool_surfaces_the_query(self):
label = tool_step_label("search_fulltext", {"q": "pizza", "vault": "V"})
assert label["key"] == "search"
assert label["params"]["value"] == "pizza"
def test_no_argument_tools_have_empty_params(self):
label = tool_step_label("list_vaults", {})
assert label["key"] == "vaults"
assert label["params"] == {}
def test_missing_argument_degrades_to_key_only(self):
label = tool_step_label("read_file", {})
assert label["key"] == "file_read"
assert "value" not in label["params"]
def test_unknown_tool_falls_back_to_generic(self):
label = tool_step_label("future_tool_name", {"x": 1})
assert label["key"] == GENERIC_KEY
assert label["params"]["tool"] == "future tool name"
def test_arguments_none_is_accepted(self):
assert tool_step_label("list_tags", None)["key"] == "tags"
def test_tool_call_record_carries_step(self):
rec = ToolCallRecord(
name="read_file",
arguments={"path": "a.md"},
ok=True,
result={},
step=tool_step_label("read_file", {"path": "a.md"}),
)
assert rec.step["key"] == "file_read"
assert rec.step["params"]["value"] == "a.md"
+112
View File
@@ -0,0 +1,112 @@
"""Unit tests for the web tools (backend.tools.web): web_search + fetch_url."""
from typing import Any
import pytest
import backend.tools.web as web
from backend.tools.api import get_tool_schemas
from backend.tools.context import ToolContext, ToolError, ToolMode, ToolRisk
from backend.tools.registry import get_tool
class FakeResponse:
def __init__(self, payload: Any = None, json_data: Any = None, status_code: int = 200,
content: bytes = b"", headers: dict | None = None, url: str = "https://example.com/x"):
self._payload = payload
self._json = json_data
self.status_code = status_code
self.content = content
self.headers = headers or {}
self.url = url
self.encoding = "utf-8"
def json(self):
return self._json
def raise_for_status(self):
if self.status_code >= 400:
raise web.httpx.HTTPStatusError("boom", request=None, response=self) # type: ignore[arg-type]
def _ctx() -> ToolContext:
return ToolContext(user={"username": "tester", "vaults": []}, mode=ToolMode.IN_APP)
class TestRegistration:
def test_tools_registered_read_only_in_app(self):
for name in ("web_search", "fetch_url"):
spec = get_tool(name)
assert spec is not None, name
assert spec.risk == ToolRisk.READ
assert name in [s["function"]["name"] for s in get_tool_schemas()]
class TestWebSearch:
def test_returns_trimmed_results(self, monkeypatch):
captured = {}
def fake_get(url, params=None, **kw):
captured["url"] = url
captured["params"] = params
return FakeResponse(json_data={
"results": [
{"title": "A", "url": "https://a.dev", "content": "x" * 900, "score": 1.0,
"publishedDate": None},
] * 12
})
monkeypatch.setattr(web.httpx, "get", fake_get)
out = web.web_search(_ctx(), web.WebSearchInput(query="pizza", max_results=3))
assert out["engine"] == "searxng"
assert out["count"] == 3
assert len(out["results"][0]["snippet"]) <= 600
assert captured["params"]["q"] == "pizza"
def test_empty_query_rejected(self):
with pytest.raises(ToolError):
web.web_search(_ctx(), web.WebSearchInput(query=" "))
def test_engine_unavailable_maps_to_tool_error(self, monkeypatch):
def boom(*a, **kw):
raise web.httpx.ConnectError("down")
monkeypatch.setattr(web.httpx, "get", boom)
with pytest.raises(ToolError) as ei:
web.web_search(_ctx(), web.WebSearchInput(query="x"))
assert ei.value.code == "web_search_unavailable"
class TestFetchUrl:
def test_html_converted_to_text(self, monkeypatch):
html = (b"<html><head><title>T&</title><style>b{}</style>"
b"<script>evil()</script></head><body><p>hello</p><ul>"
b"<li>one</li><li>two</li></ul></body></html>")
monkeypatch.setattr(web.httpx, "get",
lambda *a, **k: FakeResponse(content=html,
headers={"content-type": "text/html; charset=utf-8"}))
out = web.fetch_url(_ctx(), web.FetchUrlInput(url="https://example.com/x"))
assert out["title"] == "T&"
assert "hello" in out["text"]
assert "one" in out["text"] and "two" in out["text"]
assert "evil" not in out["text"]
assert out["status"] == 200
def test_private_address_rejected(self):
for url in ("http://127.0.0.1/admin", "http://169.254.1.1/x", "http://localhost:8000/api"):
with pytest.raises(ToolError) as ei:
web.fetch_url(_ctx(), web.FetchUrlInput(url=url))
assert ei.value.code in ("ssrf_blocked", "dns_error", "invalid_scheme"), url
def test_non_http_scheme_rejected(self):
with pytest.raises(ToolError) as ei:
web.fetch_url(_ctx(), web.FetchUrlInput(url="file:///etc/passwd"))
assert ei.value.code == "invalid_scheme"
def test_binary_content_rejected(self, monkeypatch):
monkeypatch.setattr(web.httpx, "get",
lambda *a, **k: FakeResponse(content=b"%PDF-1.4...",
headers={"content-type": "application/pdf"}))
with pytest.raises(ToolError) as ei:
web.fetch_url(_ctx(), web.FetchUrlInput(url="https://example.com/f.pdf"))
assert ei.value.code == "unsupported_content_type"