Deux causes indépendantes faisaient échouer le job test de la CI (et seulement en CI : ni .env ni les mêmes ordres de chargement en local) : - test_agent_web_tools.mock_http patche http_client.shared_client pendant sa fenêtre d'exécution. Le premier import d'importers/url_fetch dans cette fenêtre fige la factory moquée dans le namespace du module → tous les appels fetch_url suivants du processus passaient par le handler mocké de l'autre test (« assert '…/post' == '…/page' » dans test_v56_import). url_fetch résout désormais le client à l'appel, et le mock_http restaure la vraie factory (référence figée au chargement du module) y compris sur url_fetch ; - test_agent.py posait RATE_LIMIT_ENABLED=false en env var, sans effet sur le singleton Settings déjà instancié : sous pytest -n auto, le worker dépassait le quota de 60 req/min et 8 tests recevaient des 429 (KeyError 'id' au passage). Le fixture désactive maintenant le limiter sur le singleton, comme les autres fichiers de tests.
443 lines
17 KiB
Python
443 lines
17 KiB
Python
"""FlowDeck — v7.46.0 : web tools de l'agent (web_search / fetch_url / search_code).
|
|
|
|
Ces tests ne touchent JAMAIS le réseau : la couche httpx est injectée via le
|
|
paramètre ``transport`` des services, et les tools sont exercés à travers le
|
|
registre (``ToolRegistry.execute``) comme le ferait ``AgentEngine``.
|
|
|
|
Points critiques couverts :
|
|
* le registre expose bien les 3 tools et leurs schémas ;
|
|
* chaque preset de la galerie ne référence que des tools existants ;
|
|
* ``web_search`` : provider Exa, repli DuckDuckGo, et filtrage des URLs
|
|
anti-bot du moteur ;
|
|
* ``fetch_url`` : markdown extrait, troncature, et **refus SSRF** (localhost,
|
|
IP privée, schémas `file:`/`ftp:`, lien vers metadata cloud à travers une
|
|
redirection) ;
|
|
* ``search_code`` : repos / code / issues, et message clair sur quota.
|
|
|
|
Convention du projet : les tests async passent par ``asyncio.run`` (idem
|
|
``tests/test_agent.py``), pas de marqueur pytest-asyncio.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import contextlib
|
|
import ipaddress
|
|
import json
|
|
import socket
|
|
|
|
import httpx
|
|
import pytest
|
|
|
|
from app.config import settings
|
|
from app.services import http_client
|
|
from app.services import tool_registry as tr
|
|
from app.services import web_search as ws
|
|
from app.services.skill_gallery import GALLERY, export_skill, parse_payload
|
|
from app.services.tool_registry import ToolRegistry
|
|
|
|
WEB_TOOLS = ("web_search", "fetch_url", "search_code")
|
|
|
|
#: IP publique factice : les tests ne doivent dépendre ni du réseau HTTP ni du DNS.
|
|
PUBLIC_IP = "93.184.216.34"
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def stub_dns(monkeypatch):
|
|
"""Neutralise le DNS pour les noms d'hôtes publics.
|
|
|
|
Le garde-fou SSRF (`_is_public_host`) résout l'hôte pour décider si l'IP
|
|
est publique : sans ce stub, chaque test `fetch_url` sur `example.com`
|
|
ferait une vraie résolution DNS — lente, et susceptible d'échouer sous
|
|
`pytest -n auto`, ce qui rendrait ces tests dépendants de l'environnement.
|
|
Les IP littérales (`127.0.0.1`, `169.254.169.254`, `::1`) gardent la
|
|
résolution réelle : le refus des adresses privées reste donc testé pour de
|
|
vrai, y compris sur les redirections.
|
|
"""
|
|
real = socket.getaddrinfo
|
|
|
|
def fake_getaddrinfo(host, *args, **kwargs):
|
|
try:
|
|
ipaddress.ip_address(host)
|
|
except ValueError:
|
|
return [(socket.AF_INET, socket.SOCK_STREAM, 6, "", (PUBLIC_IP, 0))]
|
|
return real(host, *args, **kwargs)
|
|
|
|
monkeypatch.setattr(socket, "getaddrinfo", fake_getaddrinfo)
|
|
|
|
|
|
@pytest.fixture
|
|
def registry() -> ToolRegistry:
|
|
return ToolRegistry()
|
|
|
|
|
|
#: Vraie factory du client partagé, figée AU CHARGEMENT du module (donc avant
|
|
#: tout patch). La restauration repart de CETTE référence : si un import tardif
|
|
#: d'un module consommateur a déjà figé une factory moquée, le `finally` la
|
|
#: répare au lieu de perpétuer la fuite.
|
|
_REAL_SHARED_CLIENT = http_client.shared_client
|
|
|
|
|
|
@contextlib.contextmanager
|
|
def mock_http(handler):
|
|
"""Injecte un transport httpx dans le client partagé des services.
|
|
|
|
``web_search`` importe ``shared_client`` au niveau module, l'outillage
|
|
``fetch_url`` fait un import local : on patche les deux références pour
|
|
qu'aucun test n'atteigne jamais le réseau, même sans ``transport=``.
|
|
"""
|
|
original = http_client.shared_client
|
|
|
|
def build(**kwargs):
|
|
kwargs.pop("transport", None)
|
|
kwargs["transport"] = httpx.MockTransport(handler)
|
|
return original(**kwargs)
|
|
|
|
targets = (http_client, ws, tr)
|
|
for module in targets:
|
|
monkey = getattr(module, "shared_client", None)
|
|
if monkey is not None:
|
|
module.shared_client = build
|
|
try:
|
|
yield
|
|
finally:
|
|
for module in targets:
|
|
if getattr(module, "shared_client", None) is build:
|
|
module.shared_client = _REAL_SHARED_CLIENT
|
|
# Un consommateur importé PENDANT la fenêtre patchée a figé `build`
|
|
# dans son propre namespace : on le remet sur la vraie factory.
|
|
from app.services.importers import url_fetch
|
|
if getattr(url_fetch, "shared_client", None) is build:
|
|
url_fetch.shared_client = _REAL_SHARED_CLIENT
|
|
|
|
|
|
def run(coro):
|
|
return asyncio.run(coro)
|
|
|
|
|
|
# ── Registre ───────────────────────────────────────────────────────────────
|
|
|
|
def test_registry_exposes_web_tools(registry):
|
|
for name in WEB_TOOLS:
|
|
assert name in registry.tools, name
|
|
assert registry.tools[name].description
|
|
|
|
|
|
def test_registry_schema_includes_web_tools(registry):
|
|
names = [t["name"] for t in registry.schema()]
|
|
for name in WEB_TOOLS:
|
|
assert name in names
|
|
# Un tool hors périmètre ne doit jamais fuiter dans le schéma.
|
|
scoped = [t["name"] for t in registry.schema({"tools": ["search_workspace"]})]
|
|
assert scoped == ["search_workspace"]
|
|
|
|
|
|
def test_registry_rejects_unknown_tool(registry):
|
|
res = run(registry.execute("web_search_does_not_exist", {}))
|
|
assert res.status == "error"
|
|
|
|
|
|
# ── Cohérence galerie / registre ───────────────────────────────────────────
|
|
|
|
def test_gallery_allowed_tools_all_exist():
|
|
tools = set(ToolRegistry().tools)
|
|
for slug, preset in GALLERY.items():
|
|
unknown = [t for t in preset["allowed_tools"] if t not in tools]
|
|
assert not unknown, f"{slug} référence des tools inexistants : {unknown}"
|
|
|
|
|
|
def test_gallery_presets_are_complete():
|
|
for slug, preset in GALLERY.items():
|
|
assert preset["name"].strip(), slug
|
|
assert preset["description"].strip(), slug
|
|
assert preset["prompt_template"].strip(), slug
|
|
assert preset["allowed_tools"], slug
|
|
|
|
|
|
def test_web_presets_survive_export_import():
|
|
"""Chaque nouveau preset doit survivre au cycle export → import."""
|
|
for slug in ("recherche-marche", "veille-techno", "debug-web"):
|
|
preset = GALLERY[slug]
|
|
payload = export_skill({
|
|
"name": preset["name"],
|
|
"description": preset["description"],
|
|
"prompt_template": preset["prompt_template"],
|
|
"allowed_tools_json": json.dumps(preset["allowed_tools"]),
|
|
})
|
|
fields = parse_payload(payload)
|
|
assert fields["name"] == preset["name"]
|
|
assert sorted(fields["allowed_tools"]) == sorted(preset["allowed_tools"])
|
|
|
|
|
|
# ── web_search ─────────────────────────────────────────────────────────────
|
|
|
|
EXA_BODY = {
|
|
"results": [
|
|
{"title": "FastAPI docs", "url": "https://fastapi.tiangolo.com/",
|
|
"text": "FastAPI framework", "publishedDate": "2026-01-02T00:00:00Z"},
|
|
{"title": "Spam", "url": "https://duckduckgo.com/y.js", "text": "à filtrer"},
|
|
],
|
|
}
|
|
|
|
DDG_HTML = """
|
|
<table><tr>
|
|
<td><a class="result-link" href="/l/?uddg=https%3A%2F%2Fexample.com%2Fa&rut=1">Résultat A</a></td>
|
|
<td class="result-snippet">Extrait du <b>résultat</b> A</td>
|
|
</tr><tr>
|
|
<td><a class="result-link" href="https://example.org/b">Résultat B</a></td>
|
|
<td class="result-snippet">Extrait B</td>
|
|
</tr></table>
|
|
"""
|
|
|
|
|
|
def test_web_search_exa(monkeypatch):
|
|
monkeypatch.setattr(settings, "web_search_provider", "exa")
|
|
monkeypatch.setattr(settings, "exa_api_key", "test-key")
|
|
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
assert request.headers["x-api-key"] == "test-key"
|
|
return httpx.Response(200, json=EXA_BODY)
|
|
|
|
with mock_http(handler):
|
|
results, provider = run(ws.search_web("fastapi", transport=httpx.MockTransport(handler)))
|
|
|
|
assert provider == "exa"
|
|
assert [r["url"] for r in results] == ["https://fastapi.tiangolo.com/"]
|
|
assert results[0]["source"] == "exa"
|
|
assert results[0]["title"] == "FastAPI docs"
|
|
assert results[0]["published"] == "2026-01-02"
|
|
|
|
|
|
def test_web_search_filters_search_engine_urls(monkeypatch):
|
|
"""Une URL du moteur lui-même ne doit jamais être renvoyée à l'IA."""
|
|
monkeypatch.setattr(settings, "web_search_provider", "exa")
|
|
monkeypatch.setattr(settings, "exa_api_key", "k")
|
|
handler = lambda r: httpx.Response(200, json=EXA_BODY) # noqa: E731
|
|
with mock_http(handler):
|
|
results, _ = run(ws.search_web("x", transport=httpx.MockTransport(handler)))
|
|
assert all("duckduckgo.com" not in r["url"] for r in results)
|
|
|
|
|
|
def test_web_search_falls_back_to_duckduckgo(monkeypatch):
|
|
"""Sans clé Exa, le tool bascule sur le repli sans compte."""
|
|
monkeypatch.setattr(settings, "web_search_provider", "exa")
|
|
monkeypatch.setattr(settings, "exa_api_key", "")
|
|
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
assert "duckduckgo" in str(request.url)
|
|
return httpx.Response(200, text=DDG_HTML,
|
|
headers={"content-type": "text/html"})
|
|
|
|
with mock_http(handler):
|
|
results, provider = run(ws.search_web("test", transport=httpx.MockTransport(handler)))
|
|
|
|
assert provider == "duckduckgo"
|
|
assert [r["url"] for r in results] == ["https://example.com/a", "https://example.org/b"]
|
|
assert results[0]["title"] == "Résultat A"
|
|
assert results[0]["snippet"] == "Extrait du résultat A"
|
|
|
|
|
|
def test_web_search_empty_query_returns_nothing(monkeypatch):
|
|
monkeypatch.setattr(settings, "exa_api_key", "k")
|
|
results, provider = run(ws.search_web(" "))
|
|
assert results == []
|
|
assert provider == ""
|
|
|
|
|
|
def test_available_providers_reflects_config(monkeypatch):
|
|
monkeypatch.setattr(settings, "exa_api_key", "")
|
|
assert ws.available_providers()["exa"] is False
|
|
monkeypatch.setattr(settings, "exa_api_key", "k")
|
|
assert ws.available_providers()["exa"] is True
|
|
|
|
|
|
def test_web_search_tool_returns_actionable_error(registry, monkeypatch):
|
|
"""Aucun provider ne répond → le tool dit comment corriger."""
|
|
monkeypatch.setattr(settings, "exa_api_key", "")
|
|
handler = lambda r: httpx.Response(503, text="") # noqa: E731
|
|
with mock_http(handler):
|
|
res = run(registry.execute("web_search", {"query": "test"}))
|
|
assert res.status == "error"
|
|
assert "EXA_API_KEY" in res.message
|
|
|
|
|
|
def test_web_search_tool_success(registry, monkeypatch):
|
|
monkeypatch.setattr(settings, "web_search_provider", "exa")
|
|
monkeypatch.setattr(settings, "exa_api_key", "k")
|
|
handler = lambda r: httpx.Response(200, json=EXA_BODY) # noqa: E731
|
|
with mock_http(handler):
|
|
res = run(registry.execute("web_search", {"query": "fastapi", "num_results": 3}))
|
|
assert res.status == "success"
|
|
assert res.data["count"] == 1
|
|
assert res.data["provider"] == "exa"
|
|
|
|
|
|
# ── fetch_url ──────────────────────────────────────────────────────────────
|
|
|
|
HTML_PAGE = """
|
|
<html><head><title>Titre de la page</title>
|
|
<meta property="og:description" content="Description courte"></head>
|
|
<body><article>
|
|
<h1>Titre de la page</h1>
|
|
<p>Premier paragraphe avec du <b>gras</b>.</p>
|
|
<pre><code>print("hello")</code></pre>
|
|
<ul><li>un</li><li>deux</li></ul>
|
|
</article></body></html>
|
|
"""
|
|
|
|
|
|
def test_fetch_url_returns_markdown(registry):
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
assert str(request.url) == "https://example.com/page"
|
|
return httpx.Response(200, text=HTML_PAGE,
|
|
headers={"content-type": "text/html; charset=utf-8"})
|
|
|
|
with mock_http(handler):
|
|
res = run(registry.execute("fetch_url", {"url": "https://example.com/page"}))
|
|
|
|
assert res.status == "success", res.message
|
|
assert "Premier paragraphe" in res.data["markdown"]
|
|
assert "print" in res.data["markdown"]
|
|
assert res.data["title"] == "Titre de la page"
|
|
assert res.data["truncated"] is False
|
|
|
|
|
|
@pytest.mark.parametrize("url", [
|
|
"http://localhost:8080/api/health",
|
|
"http://127.0.0.1/admin",
|
|
"http://169.254.169.254/latest/meta-data/",
|
|
"http://[::1]/",
|
|
"file:///etc/passwd",
|
|
"ftp://example.com/x",
|
|
])
|
|
def test_fetch_url_blocks_ssrf_targets(registry, url):
|
|
res = run(registry.execute("fetch_url", {"url": url}))
|
|
assert res.status == "error", f"{url} aurait dû être refusé"
|
|
|
|
|
|
def test_fetch_url_revalidates_redirect_target(registry):
|
|
"""Une URL publique redirigeant vers le metadata cloud doit être refusée."""
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
return httpx.Response(302, headers={"location": "http://169.254.169.254/latest/"})
|
|
|
|
with mock_http(handler):
|
|
res = run(registry.execute("fetch_url", {"url": "https://example.com/"}))
|
|
|
|
assert res.status == "error"
|
|
assert "non autorisé" in res.message
|
|
|
|
|
|
def test_fetch_url_follows_public_redirect(registry):
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
if request.url.path == "/old":
|
|
return httpx.Response(301, headers={"location": "https://example.com/new"})
|
|
return httpx.Response(200, text=HTML_PAGE,
|
|
headers={"content-type": "text/html"})
|
|
|
|
with mock_http(handler):
|
|
res = run(registry.execute("fetch_url", {"url": "https://example.com/old"}))
|
|
|
|
assert res.status == "success", res.message
|
|
assert res.data["url"] == "https://example.com/new"
|
|
|
|
|
|
def test_fetch_url_truncates(registry):
|
|
long_html = "<html><body><p>" + ("a" * 50000) + "</p></body></html>"
|
|
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
return httpx.Response(200, text=long_html,
|
|
headers={"content-type": "text/html"})
|
|
|
|
with mock_http(handler):
|
|
res = run(registry.execute(
|
|
"fetch_url", {"url": "https://example.com/long", "max_chars": 1000}))
|
|
|
|
assert res.status == "success"
|
|
assert res.data["truncated"] is True
|
|
assert len(res.data["markdown"]) < 1200
|
|
|
|
|
|
def test_fetch_url_empty_url(registry):
|
|
assert run(registry.execute("fetch_url", {"url": " "})).status == "error"
|
|
|
|
|
|
# ── search_code ────────────────────────────────────────────────────────────
|
|
|
|
GH_REPOS = {
|
|
"total_count": 2,
|
|
"items": [
|
|
{"full_name": "tiangolo/fastapi", "html_url": "https://github.com/tiangolo/fastapi",
|
|
"description": "FastAPI framework", "stargazers_count": 80000},
|
|
{"full_name": "someone/fastapi-clone", "html_url": "https://github.com/someone/fastapi-clone",
|
|
"description": "A clone", "stargazers_count": 3},
|
|
],
|
|
}
|
|
|
|
GH_ISSUES = {
|
|
"total_count": 1,
|
|
"items": [
|
|
{"number": 42, "title": "Crash on startup", "state": "closed",
|
|
"html_url": "https://github.com/o/r/issues/42", "body": "It crashes"},
|
|
],
|
|
}
|
|
|
|
|
|
def test_search_code_repos(monkeypatch):
|
|
monkeypatch.setattr(settings, "github_token", "")
|
|
handler = lambda r: httpx.Response(200, json=GH_REPOS) # noqa: E731
|
|
results = run(ws.search_github("fastapi", "repositories", 5,
|
|
transport=httpx.MockTransport(handler)))
|
|
assert [r["title"] for r in results] == ["tiangolo/fastapi", "someone/fastapi-clone"]
|
|
assert results[0]["stars"] == 80000
|
|
assert results[0]["source"] == "github/repos"
|
|
|
|
|
|
def test_search_code_issues_sends_token(monkeypatch):
|
|
monkeypatch.setattr(settings, "github_token", "ghp_test")
|
|
|
|
def handler(request: httpx.Request) -> httpx.Response:
|
|
assert request.headers["Authorization"] == "Bearer ghp_test"
|
|
assert request.url.path.endswith("/search/issues")
|
|
return httpx.Response(200, json=GH_ISSUES)
|
|
|
|
results = run(ws.search_github("crash", "issues", 5,
|
|
transport=httpx.MockTransport(handler)))
|
|
assert results[0]["title"] == "Crash on startup"
|
|
assert results[0]["state"] == "closed"
|
|
assert results[0]["url"].endswith("/issues/42")
|
|
|
|
|
|
def test_search_code_quota_message(monkeypatch):
|
|
monkeypatch.setattr(settings, "github_token", "")
|
|
handler = lambda r: httpx.Response(403, json={}) # noqa: E731
|
|
with pytest.raises(RuntimeError) as err:
|
|
run(ws.search_github("x", "repos", 5, transport=httpx.MockTransport(handler)))
|
|
assert "GITHUB_TOKEN" in str(err.value)
|
|
|
|
|
|
def test_search_code_tool_wraps_errors(registry, monkeypatch):
|
|
"""Le tool doit transformer une erreur GitHub en message lisible, sans réseau."""
|
|
async def _boom(*a, **kw):
|
|
raise RuntimeError("GitHub: requête de recherche invalide (422)")
|
|
|
|
# L'outillage importe ``search_github`` au moment de l'appel : c'est
|
|
# l'attribut du module qu'il faut patcher, pas celui du registre.
|
|
monkeypatch.setattr(ws, "search_github", _boom)
|
|
res = run(registry.execute("search_code", {"query": "!!!"}))
|
|
assert res.status == "error"
|
|
assert "GitHub" in res.message
|
|
|
|
|
|
def test_search_code_tool_success(registry, monkeypatch):
|
|
monkeypatch.setattr(settings, "github_token", "")
|
|
handler = lambda r: httpx.Response(200, json=GH_REPOS) # noqa: E731
|
|
real_search = ws.search_github
|
|
|
|
async def _fake(query, kind="repositories", limit=5, transport=None):
|
|
return await real_search(query, kind, limit, transport=httpx.MockTransport(handler))
|
|
|
|
monkeypatch.setattr(ws, "search_github", _fake)
|
|
res = run(registry.execute("search_code", {"query": "fastapi", "kind": "repositories"}))
|
|
assert res.status == "success"
|
|
assert res.data["count"] == 2
|
|
assert res.data["results"][0]["title"] == "tiangolo/fastapi"
|