137 lines
5.7 KiB
Python
137 lines
5.7 KiB
Python
"""Unit tests for the web tools (backend.tools.web): web_search + fetch_url."""
|
|
|
|
from typing import Any
|
|
|
|
import pytest
|
|
|
|
import backend.tools.web as web
|
|
from backend.tools.api import get_tool_schemas
|
|
from backend.tools.context import ToolContext, ToolError, ToolMode, ToolRisk
|
|
from backend.tools.registry import get_tool
|
|
|
|
|
|
class FakeResponse:
|
|
def __init__(self, payload: Any = None, json_data: Any = None, status_code: int = 200,
|
|
content: bytes = b"", headers: dict | None = None, url: str = "https://example.com/x"):
|
|
self._payload = payload
|
|
self._json = json_data
|
|
self.status_code = status_code
|
|
self.content = content
|
|
self.headers = headers or {}
|
|
self.url = url
|
|
self.encoding = "utf-8"
|
|
|
|
def json(self):
|
|
return self._json
|
|
|
|
def raise_for_status(self):
|
|
if self.status_code >= 400:
|
|
raise web.httpx.HTTPStatusError("boom", request=None, response=self) # type: ignore[arg-type]
|
|
|
|
|
|
def _ctx() -> ToolContext:
|
|
return ToolContext(user={"username": "tester", "vaults": []}, mode=ToolMode.IN_APP)
|
|
|
|
|
|
class TestRegistration:
|
|
def test_tools_registered_read_only_in_app(self):
|
|
for name in ("web_search", "fetch_url"):
|
|
spec = get_tool(name)
|
|
assert spec is not None, name
|
|
assert spec.risk == ToolRisk.READ
|
|
assert name in [s["function"]["name"] for s in get_tool_schemas()]
|
|
|
|
|
|
class TestWebSearch:
|
|
def test_returns_trimmed_results(self, monkeypatch):
|
|
captured = {}
|
|
|
|
def fake_get(url, params=None, **kw):
|
|
captured["url"] = url
|
|
captured["params"] = params
|
|
return FakeResponse(json_data={
|
|
"results": [
|
|
{"title": "A", "url": "https://a.dev", "content": "x" * 900, "score": 1.0,
|
|
"publishedDate": None},
|
|
] * 12
|
|
})
|
|
|
|
monkeypatch.setattr(web.httpx, "get", fake_get)
|
|
out = web.web_search(_ctx(), web.WebSearchInput(query="pizza", max_results=3))
|
|
assert out["engine"] == "searxng"
|
|
assert out["count"] == 3
|
|
assert len(out["results"][0]["snippet"]) <= 600
|
|
assert captured["params"]["q"] == "pizza"
|
|
|
|
def test_empty_query_rejected(self):
|
|
with pytest.raises(ToolError):
|
|
web.web_search(_ctx(), web.WebSearchInput(query=" "))
|
|
|
|
def test_empty_result_set_warns_about_blocked_engines(self, monkeypatch):
|
|
"""An all-blocked instance answers 200 with no results: the model must
|
|
be told instead of retrying the same search until the quota burns."""
|
|
monkeypatch.setattr(web.httpx, "get", lambda *a, **kw: FakeResponse(json_data={
|
|
"results": [],
|
|
"number_of_results": 0,
|
|
"unresponsive_engines": [["duckduckgo", "CAPTCHA"], ["google", "access denied"]],
|
|
}))
|
|
out = web.web_search(_ctx(), web.WebSearchInput(query="meteo montreal"))
|
|
assert out["count"] == 0
|
|
assert out["unresponsive_engines"] == ["duckduckgo", "google"]
|
|
assert "warning" in out
|
|
assert "indisponibles" in out["warning"]
|
|
|
|
def test_results_carry_no_warning(self, monkeypatch):
|
|
monkeypatch.setattr(web.httpx, "get", lambda *a, **kw: FakeResponse(json_data={
|
|
"results": [{"title": "A", "url": "https://a.dev", "content": "x"}],
|
|
"unresponsive_engines": [["brave", "rate limited"]],
|
|
}))
|
|
out = web.web_search(_ctx(), web.WebSearchInput(query="pizza"))
|
|
assert out["count"] == 1
|
|
assert "warning" not in out
|
|
assert out["unresponsive_engines"] == ["brave"]
|
|
|
|
def test_engine_unavailable_maps_to_tool_error(self, monkeypatch):
|
|
def boom(*a, **kw):
|
|
raise web.httpx.ConnectError("down")
|
|
|
|
monkeypatch.setattr(web.httpx, "get", boom)
|
|
with pytest.raises(ToolError) as ei:
|
|
web.web_search(_ctx(), web.WebSearchInput(query="x"))
|
|
assert ei.value.code == "web_search_unavailable"
|
|
|
|
|
|
class TestFetchUrl:
|
|
def test_html_converted_to_text(self, monkeypatch):
|
|
html = (b"<html><head><title>T&</title><style>b{}</style>"
|
|
b"<script>evil()</script></head><body><p>hello</p><ul>"
|
|
b"<li>one</li><li>two</li></ul></body></html>")
|
|
monkeypatch.setattr(web.httpx, "get",
|
|
lambda *a, **k: FakeResponse(content=html,
|
|
headers={"content-type": "text/html; charset=utf-8"}))
|
|
out = web.fetch_url(_ctx(), web.FetchUrlInput(url="https://example.com/x"))
|
|
assert out["title"] == "T&"
|
|
assert "hello" in out["text"]
|
|
assert "one" in out["text"] and "two" in out["text"]
|
|
assert "evil" not in out["text"]
|
|
assert out["status"] == 200
|
|
|
|
def test_private_address_rejected(self):
|
|
for url in ("http://127.0.0.1/admin", "http://169.254.1.1/x", "http://localhost:8000/api"):
|
|
with pytest.raises(ToolError) as ei:
|
|
web.fetch_url(_ctx(), web.FetchUrlInput(url=url))
|
|
assert ei.value.code in ("ssrf_blocked", "dns_error", "invalid_scheme"), url
|
|
|
|
def test_non_http_scheme_rejected(self):
|
|
with pytest.raises(ToolError) as ei:
|
|
web.fetch_url(_ctx(), web.FetchUrlInput(url="file:///etc/passwd"))
|
|
assert ei.value.code == "invalid_scheme"
|
|
|
|
def test_binary_content_rejected(self, monkeypatch):
|
|
monkeypatch.setattr(web.httpx, "get",
|
|
lambda *a, **k: FakeResponse(content=b"%PDF-1.4...",
|
|
headers={"content-type": "application/pdf"}))
|
|
with pytest.raises(ToolError) as ei:
|
|
web.fetch_url(_ctx(), web.FetchUrlInput(url="https://example.com/f.pdf"))
|
|
assert ei.value.code == "unsupported_content_type"
|