Editeur: - sync() lisait el.textContent qui supprime les <br> : ajout de gt() (innerText) utilise par sync(), tableaux (toggle/columns) et _resultText. - Scission sur Entree + limites haut/bas de bloc via splitCaret() (marqueur temporaire au curseur), plus de perte de fin de bloc multi-lignes. Agent: - fetch_provider_models() valide la liste nvidia : filtre des modeles non-chat (nom) puis probe reelle /chat/completions (6 paralleles, 6s), seuls les modeles 2xx restent. 429 conserve (route, donc utilisable). - Repli sur la liste filtree si toutes les validations echouent. - tests/test_llm_config.py : 5 cas (catalogue mock, 404 chat, autres providers non valides, chute de securite, offline). 276 tests passent.
139 lines
5.5 KiB
Python
139 lines
5.5 KiB
Python
"""FlowDeck — tests for app.services.llm_config.fetch_provider_models.
|
|
|
|
Verifies the live model-list fetching and the chat-capability validation used
|
|
for noisy providers (NVIDIA lists its whole catalog on /v1/models, most of
|
|
which answers 404 on /v1/chat/completions).
|
|
"""
|
|
import asyncio
|
|
import json
|
|
import os
|
|
import threading
|
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
|
|
import pytest
|
|
|
|
# Pinned BEFORE any app import: app.config.settings is a process-wide singleton
|
|
# frozen at first import, and pytest imports test modules during collection —
|
|
# before the `client` fixtures of other test files set these env vars.
|
|
os.environ["RATE_LIMIT_ENABLED"] = "false"
|
|
os.environ["LLM_PROVIDER"] = "offline"
|
|
os.environ["DATABASE_URL"] = "sqlite:///:memory:"
|
|
os.environ["APP_SECRET_KEY"] = "test-secret-for-tests"
|
|
|
|
from app.services.llm_config import _is_likely_chat, fetch_provider_models
|
|
|
|
# Models returned by the mock {"id": ...} list.
|
|
MOCK_CATALOG = [
|
|
"nvidia/nemotron-3-super-120b-a12b", # chat-capable
|
|
"meta/llama-3.1-70b-instruct", # chat-capable
|
|
"nvidia/embed-qa-4", # embeddings only → 404 on chat
|
|
"nvidia/rerank-qa-mistral-4b", # reranker only → 404 on chat
|
|
"black-forest-labs/flux-1-schnell", # image gen → 404 on chat
|
|
]
|
|
|
|
|
|
def _make_server(reject_non_chat: bool = True, accept: set | None = None,
|
|
fail_all_chat: bool = False):
|
|
"""HTTP server mimicking an OpenAI-compatible /models + /chat/completions.
|
|
|
|
With ``reject_non_chat`` the chat endpoint returns 404 for any model whose
|
|
id contains embed/rerank/flux, exactly like NVIDIA's public API.
|
|
With ``fail_all_chat`` every chat call returns 404 (probe failure case).
|
|
``accept`` is an optional allow-list; when set, only those models answer 200.
|
|
"""
|
|
accept_models = accept if accept is not None else set(MOCK_CATALOG[:2])
|
|
|
|
class Handler(BaseHTTPRequestHandler):
|
|
def log_message(self, *args): # silence test noise
|
|
pass
|
|
|
|
def do_GET(self):
|
|
if self.path.endswith("/models"):
|
|
body = json.dumps({"data": [{"id": m} for m in MOCK_CATALOG]}).encode()
|
|
self.send_response(200)
|
|
self.send_header("Content-Type", "application/json")
|
|
self.send_header("Content-Length", str(len(body)))
|
|
self.end_headers()
|
|
self.wfile.write(body)
|
|
else:
|
|
self.send_response(404)
|
|
self.send_header("Content-Length", "0")
|
|
self.end_headers()
|
|
|
|
def do_POST(self):
|
|
if self.path.endswith("/chat/completions"):
|
|
n = int(self.headers.get("Content-Length", 0) or 0)
|
|
try:
|
|
req = json.loads(self.rfile.read(n) or b"{}")
|
|
except ValueError:
|
|
req = {}
|
|
model = req.get("model", "")
|
|
works = model in accept_models
|
|
if reject_non_chat and any(x in model for x in ("embed", "rerank", "flux")):
|
|
works = False
|
|
if fail_all_chat:
|
|
works = False
|
|
self.send_response(200 if works else 404)
|
|
self.send_header("Content-Length", "0")
|
|
self.end_headers()
|
|
else:
|
|
self.send_response(404)
|
|
self.send_header("Content-Length", "0")
|
|
self.end_headers()
|
|
|
|
server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
|
|
threading.Thread(target=server.serve_forever, daemon=True).start()
|
|
return server
|
|
|
|
|
|
@pytest.fixture
|
|
def mock_server():
|
|
srv = _make_server()
|
|
yield srv
|
|
srv.shutdown()
|
|
|
|
|
|
def test_is_likely_chat_filters_obvious_non_llm():
|
|
assert _is_likely_chat("nvidia/nemotron-3-super-120b-a12b")
|
|
assert _is_likely_chat("meta/llama-3.1-70b-instruct")
|
|
assert not _is_likely_chat("nvidia/embed-qa-4")
|
|
assert not _is_likely_chat("snowflake/arctic-embed-l")
|
|
assert not _is_likely_chat("nvidia/rerank-qa-mistral-4b")
|
|
assert not _is_likely_chat("black-forest-labs/flux-1-schnell")
|
|
assert not _is_likely_chat("nvidia/tts")
|
|
|
|
|
|
def test_nvidia_only_chat_models_survive(mock_server):
|
|
base = f"http://127.0.0.1:{mock_server.server_port}/v1"
|
|
models = asyncio.run(fetch_provider_models(
|
|
"nvidia", api_key="nvapi-test", api_base=base, timeout=2)
|
|
)
|
|
assert models == ["nvidia/nemotron-3-super-120b-a12b", "meta/llama-3.1-70b-instruct"]
|
|
|
|
|
|
def test_validation_falls_back_to_name_filter_when_all_fail():
|
|
# Everything answers 404 → validated list empty → keep the name-filtered
|
|
# candidates instead of wiping the selector.
|
|
srv = _make_server(fail_all_chat=True)
|
|
try:
|
|
base = f"http://127.0.0.1:{srv.server_port}/v1"
|
|
models = asyncio.run(fetch_provider_models(
|
|
"nvidia", api_key="nvapi-test", api_base=base, timeout=2,
|
|
))
|
|
finally:
|
|
srv.shutdown()
|
|
assert set(models) == {"nvidia/nemotron-3-super-120b-a12b", "meta/llama-3.1-70b-instruct"}
|
|
|
|
|
|
def test_other_providers_keep_full_list(mock_server):
|
|
# Provider not in _CHAT_VALIDATED_PROVIDERS (openai) is returned verbatim:
|
|
# no chat probes are sent (server returns 404 on POST, would break otherwise).
|
|
base = f"http://127.0.0.1:{mock_server.server_port}/v1"
|
|
models = asyncio.run(fetch_provider_models(
|
|
"openai", api_key="sk-test", api_base=base, timeout=2)
|
|
)
|
|
assert models == MOCK_CATALOG
|
|
|
|
|
|
def test_offline_returns_empty_list():
|
|
assert asyncio.run(fetch_provider_models("offline", api_base="", timeout=2)) == [] |