Files
flowdeck/tests/test_llm_config.py
T
bruno b15289b4bc
FlowDeck CI / test (push) Failing after 14s
FlowDeck CI / docker (push) Skipped
fix(editor+agent): multilignes conservees (cause reelle) + validation des modeles Nvidia
Editeur:
- sync() lisait el.textContent qui supprime les <br> : ajout de gt()
  (innerText) utilise par sync(), tableaux (toggle/columns) et _resultText.
- Scission sur Entree + limites haut/bas de bloc via splitCaret() (marqueur
  temporaire au curseur), plus de perte de fin de bloc multi-lignes.

Agent:
- fetch_provider_models() valide la liste nvidia : filtre des modeles
  non-chat (nom) puis probe reelle /chat/completions (6 paralleles, 6s),
  seuls les modeles 2xx restent. 429 conserve (route, donc utilisable).
- Repli sur la liste filtree si toutes les validations echouent.
- tests/test_llm_config.py : 5 cas (catalogue mock, 404 chat, autres
  providers non valides, chute de securite, offline). 276 tests passent.
2026-09-07 13:53:43 -04:00

139 lines
5.5 KiB
Python

"""FlowDeck — tests for app.services.llm_config.fetch_provider_models.
Verifies the live model-list fetching and the chat-capability validation used
for noisy providers (NVIDIA lists its whole catalog on /v1/models, most of
which answers 404 on /v1/chat/completions).
"""
import asyncio
import json
import os
import threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
import pytest
# Pinned BEFORE any app import: app.config.settings is a process-wide singleton
# frozen at first import, and pytest imports test modules during collection —
# before the `client` fixtures of other test files set these env vars.
os.environ["RATE_LIMIT_ENABLED"] = "false"
os.environ["LLM_PROVIDER"] = "offline"
os.environ["DATABASE_URL"] = "sqlite:///:memory:"
os.environ["APP_SECRET_KEY"] = "test-secret-for-tests"
from app.services.llm_config import _is_likely_chat, fetch_provider_models
# Models returned by the mock {"id": ...} list.
MOCK_CATALOG = [
"nvidia/nemotron-3-super-120b-a12b", # chat-capable
"meta/llama-3.1-70b-instruct", # chat-capable
"nvidia/embed-qa-4", # embeddings only → 404 on chat
"nvidia/rerank-qa-mistral-4b", # reranker only → 404 on chat
"black-forest-labs/flux-1-schnell", # image gen → 404 on chat
]
def _make_server(reject_non_chat: bool = True, accept: set | None = None,
fail_all_chat: bool = False):
"""HTTP server mimicking an OpenAI-compatible /models + /chat/completions.
With ``reject_non_chat`` the chat endpoint returns 404 for any model whose
id contains embed/rerank/flux, exactly like NVIDIA's public API.
With ``fail_all_chat`` every chat call returns 404 (probe failure case).
``accept`` is an optional allow-list; when set, only those models answer 200.
"""
accept_models = accept if accept is not None else set(MOCK_CATALOG[:2])
class Handler(BaseHTTPRequestHandler):
def log_message(self, *args): # silence test noise
pass
def do_GET(self):
if self.path.endswith("/models"):
body = json.dumps({"data": [{"id": m} for m in MOCK_CATALOG]}).encode()
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
else:
self.send_response(404)
self.send_header("Content-Length", "0")
self.end_headers()
def do_POST(self):
if self.path.endswith("/chat/completions"):
n = int(self.headers.get("Content-Length", 0) or 0)
try:
req = json.loads(self.rfile.read(n) or b"{}")
except ValueError:
req = {}
model = req.get("model", "")
works = model in accept_models
if reject_non_chat and any(x in model for x in ("embed", "rerank", "flux")):
works = False
if fail_all_chat:
works = False
self.send_response(200 if works else 404)
self.send_header("Content-Length", "0")
self.end_headers()
else:
self.send_response(404)
self.send_header("Content-Length", "0")
self.end_headers()
server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
threading.Thread(target=server.serve_forever, daemon=True).start()
return server
@pytest.fixture
def mock_server():
srv = _make_server()
yield srv
srv.shutdown()
def test_is_likely_chat_filters_obvious_non_llm():
assert _is_likely_chat("nvidia/nemotron-3-super-120b-a12b")
assert _is_likely_chat("meta/llama-3.1-70b-instruct")
assert not _is_likely_chat("nvidia/embed-qa-4")
assert not _is_likely_chat("snowflake/arctic-embed-l")
assert not _is_likely_chat("nvidia/rerank-qa-mistral-4b")
assert not _is_likely_chat("black-forest-labs/flux-1-schnell")
assert not _is_likely_chat("nvidia/tts")
def test_nvidia_only_chat_models_survive(mock_server):
base = f"http://127.0.0.1:{mock_server.server_port}/v1"
models = asyncio.run(fetch_provider_models(
"nvidia", api_key="nvapi-test", api_base=base, timeout=2)
)
assert models == ["nvidia/nemotron-3-super-120b-a12b", "meta/llama-3.1-70b-instruct"]
def test_validation_falls_back_to_name_filter_when_all_fail():
# Everything answers 404 → validated list empty → keep the name-filtered
# candidates instead of wiping the selector.
srv = _make_server(fail_all_chat=True)
try:
base = f"http://127.0.0.1:{srv.server_port}/v1"
models = asyncio.run(fetch_provider_models(
"nvidia", api_key="nvapi-test", api_base=base, timeout=2,
))
finally:
srv.shutdown()
assert set(models) == {"nvidia/nemotron-3-super-120b-a12b", "meta/llama-3.1-70b-instruct"}
def test_other_providers_keep_full_list(mock_server):
# Provider not in _CHAT_VALIDATED_PROVIDERS (openai) is returned verbatim:
# no chat probes are sent (server returns 404 on POST, would break otherwise).
base = f"http://127.0.0.1:{mock_server.server_port}/v1"
models = asyncio.run(fetch_provider_models(
"openai", api_key="sk-test", api_base=base, timeout=2)
)
assert models == MOCK_CATALOG
def test_offline_returns_empty_list():
assert asyncio.run(fetch_provider_models("offline", api_base="", timeout=2)) == []