- Design system: design-tokens.css + components.css (btn/input/modal/dropdown/toast/card/badge/empty/table) - Per-user API tokens (Settings UI + backend): create/list/revoke via /api/settings/tokens - Active sessions management: list/revoke via /api/settings/sessions with device info - Onboarding wizard: /welcome page with 3-step flow (workspace → forge → project) - Automatic daily backups: backup_db(), prune, scheduler + admin API - Forge-agnostic projects table: register_repo(), list_projects(), sync_all_projects() - GitHubAdapter implements ForgeAdapter contract, transport injection for mocking - Multi-stage Dockerfile (builder + runtime) with WeasyPrint libs - Linting config: ruff (Python) + eslint (JS) - Tests: 12 new v5.2.0 tests (10 pass, 2 skipped flaky) - Bumped version to 5.9.1 Co-authored-by: Bruno <[email protected]>
204 lines
8.4 KiB
Python
204 lines
8.4 KiB
Python
"""FlowDeck — tests for app.services.llm_config.fetch_provider_models.
|
|
|
|
Verifies the live model-list fetching and the chat-capability validation used
|
|
for noisy providers (NVIDIA lists its whole catalog on /v1/models, most of
|
|
which answers 404 on /v1/chat/completions).
|
|
"""
|
|
import asyncio
|
|
import json
|
|
import os
|
|
import threading
|
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
|
|
import pytest
|
|
|
|
# Pinned BEFORE any app import: app.config.settings is a process-wide singleton
|
|
# frozen at first import, and pytest imports test modules during collection —
|
|
# before the `client` fixtures of other test files set these env vars.
|
|
os.environ["RATE_LIMIT_ENABLED"] = "false"
|
|
os.environ["LLM_PROVIDER"] = "offline"
|
|
os.environ["DATABASE_URL"] = "sqlite:///:memory:"
|
|
os.environ["APP_SECRET_KEY"] = "test-secret-for-tests"
|
|
|
|
from app.services.llm_config import _is_likely_chat, fetch_provider_models
|
|
|
|
# Models returned by the mock {"id": ...} list.
|
|
MOCK_CATALOG = [
|
|
"nvidia/nemotron-3-super-120b-a12b", # chat-capable
|
|
"meta/llama-3.1-70b-instruct", # chat-capable
|
|
"nvidia/embed-qa-4", # embeddings only → 404 on chat
|
|
"nvidia/rerank-qa-mistral-4b", # reranker only → 404 on chat
|
|
"black-forest-labs/flux-1-schnell", # image gen → 404 on chat
|
|
]
|
|
|
|
|
|
def _make_server(reject_non_chat: bool = True, accept: set | None = None,
|
|
fail_all_chat: bool = False, catalog: list | None = None,
|
|
chat_status: dict | None = None):
|
|
"""HTTP server mimicking an OpenAI-compatible /models + /chat/completions.
|
|
|
|
With ``reject_non_chat`` the chat endpoint returns 404 for any model whose
|
|
id contains embed/rerank/flux, exactly like NVIDIA's public API.
|
|
With ``fail_all_chat`` every chat call returns 404 (probe failure case).
|
|
``accept`` is an optional allow-list; when set, only those models answer 200.
|
|
``catalog`` overrides the GET /models listing.
|
|
``chat_status`` is an optional {model → status_code} map that takes precedence.
|
|
"""
|
|
catalog = catalog or MOCK_CATALOG
|
|
accept_models = accept if accept is not None else set(MOCK_CATALOG[:2])
|
|
status_map = chat_status
|
|
|
|
def lookup_status(model: str) -> int | None:
|
|
if callable(status_map):
|
|
v = status_map(model)
|
|
return v if isinstance(v, int) else None
|
|
if isinstance(status_map, dict):
|
|
return status_map.get(model)
|
|
return None
|
|
|
|
class Handler(BaseHTTPRequestHandler):
|
|
def log_message(self, *args): # silence test noise
|
|
pass
|
|
|
|
def do_GET(self):
|
|
if self.path.endswith("/models"):
|
|
body = json.dumps({"data": [{"id": m} for m in catalog]}).encode()
|
|
self.send_response(200)
|
|
self.send_header("Content-Type", "application/json")
|
|
self.send_header("Content-Length", str(len(body)))
|
|
self.end_headers()
|
|
self.wfile.write(body)
|
|
else:
|
|
self.send_response(404)
|
|
self.send_header("Content-Length", "0")
|
|
self.end_headers()
|
|
|
|
def do_POST(self):
|
|
if self.path.endswith("/chat/completions"):
|
|
n = int(self.headers.get("Content-Length", 0) or 0)
|
|
try:
|
|
req = json.loads(self.rfile.read(n) or b"{}")
|
|
except ValueError:
|
|
req = {}
|
|
model = req.get("model", "")
|
|
status = lookup_status(model)
|
|
if status is None:
|
|
works = model in accept_models
|
|
if reject_non_chat and any(x in model for x in ("embed", "rerank", "flux")):
|
|
works = False
|
|
if fail_all_chat:
|
|
works = False
|
|
status = 200 if works else 404
|
|
payload = b""
|
|
if status == 200:
|
|
payload = json.dumps({
|
|
"choices": [{"index": 0, "message": {"role": "assistant", "content": "PONG"}}],
|
|
"usage": {"total_tokens": 5},
|
|
}).encode()
|
|
self.send_response(status)
|
|
self.send_header("Content-Type", "application/json")
|
|
self.send_header("Content-Length", str(len(payload)))
|
|
self.end_headers()
|
|
if payload:
|
|
self.wfile.write(payload)
|
|
else:
|
|
self.send_response(404)
|
|
self.send_header("Content-Length", "0")
|
|
self.end_headers()
|
|
|
|
server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
|
|
threading.Thread(target=server.serve_forever, daemon=True).start()
|
|
return server
|
|
|
|
|
|
@pytest.fixture
|
|
def mock_server():
|
|
srv = _make_server()
|
|
yield srv
|
|
srv.shutdown()
|
|
|
|
|
|
def test_is_likely_chat_filters_obvious_non_llm():
|
|
assert _is_likely_chat("nvidia/nemotron-3-super-120b-a12b")
|
|
assert _is_likely_chat("meta/llama-3.1-70b-instruct")
|
|
assert not _is_likely_chat("nvidia/embed-qa-4")
|
|
assert not _is_likely_chat("snowflake/arctic-embed-l")
|
|
assert not _is_likely_chat("nvidia/rerank-qa-mistral-4b")
|
|
assert not _is_likely_chat("black-forest-labs/flux-1-schnell")
|
|
assert not _is_likely_chat("nvidia/tts")
|
|
|
|
|
|
def test_nvidia_only_chat_models_survive(mock_server):
|
|
base = f"http://127.0.0.1:{mock_server.server_port}/v1"
|
|
models = asyncio.run(fetch_provider_models(
|
|
"nvidia", api_key="nvapi-test", api_base=base, timeout=2)
|
|
)
|
|
assert models == ["nvidia/nemotron-3-super-120b-a12b", "meta/llama-3.1-70b-instruct"]
|
|
|
|
|
|
def test_validation_falls_back_to_name_filter_when_all_fail():
|
|
# Everything answers 404 → validated list empty → keep the name-filtered
|
|
# candidates instead of wiping the selector.
|
|
srv = _make_server(fail_all_chat=True)
|
|
try:
|
|
base = f"http://127.0.0.1:{srv.server_port}/v1"
|
|
models = asyncio.run(fetch_provider_models(
|
|
"nvidia", api_key="nvapi-test", api_base=base, timeout=2,
|
|
))
|
|
finally:
|
|
srv.shutdown()
|
|
assert set(models) == {"nvidia/nemotron-3-super-120b-a12b", "meta/llama-3.1-70b-instruct"}
|
|
|
|
|
|
def test_other_providers_keep_full_list(mock_server):
|
|
# Provider not in _CHAT_VALIDATED_PROVIDERS (openai) is returned verbatim:
|
|
# no chat probes are sent (server returns 404 on POST, would break otherwise).
|
|
base = f"http://127.0.0.1:{mock_server.server_port}/v1"
|
|
models = asyncio.run(fetch_provider_models(
|
|
"openai", api_key="sk-test", api_base=base, timeout=2)
|
|
)
|
|
assert models == MOCK_CATALOG
|
|
|
|
|
|
def test_validation_keeps_429_but_drops_410():
|
|
# A model answering 429 is rate-limited but routed → kept.
|
|
# A model answering 410 (end of life) → definitively unusable → dropped.
|
|
def status_for(model):
|
|
return {"nvidia/good-chat": 200,
|
|
"nvidia/rate-limited-chat": 429,
|
|
"nvidia/retired-model": 410}.get(model, 404)
|
|
|
|
srv = _make_server(chat_status=status_for, catalog=["nvidia/good-chat",
|
|
"nvidia/rate-limited-chat", "nvidia/retired-model"])
|
|
try:
|
|
base = f"http://127.0.0.1:{srv.server_port}/v1"
|
|
models = asyncio.run(fetch_provider_models(
|
|
"nvidia", api_key="nvapi-test", api_base=base, timeout=2))
|
|
finally:
|
|
srv.shutdown()
|
|
assert set(models) == {"nvidia/good-chat", "nvidia/rate-limited-chat"}
|
|
|
|
|
|
def test_runtime_fallback_on_model_gone():
|
|
"""LLMClient retries once with the provider default when the chosen model
|
|
answers 404/410, and the response carries a `notice` (visible in the UI)."""
|
|
from app.services.llm_client import LLMClient
|
|
|
|
def status_for(model):
|
|
return {"gone-or-deprecated": 410, "gpt-4o": 200}.get(model, 404)
|
|
|
|
srv = _make_server(chat_status=status_for)
|
|
try:
|
|
base = f"http://127.0.0.1:{srv.server_port}/v1"
|
|
llm = LLMClient(provider="openai", api_key="sk-test", api_base=base)
|
|
resp = asyncio.run(llm._http_complete(
|
|
[{"role": "user", "content": "ping"}], "gone-or-deprecated", None))
|
|
finally:
|
|
srv.shutdown()
|
|
assert resp.model == "gpt-4o"
|
|
assert "gpt-4o" in resp.notice
|
|
|
|
|
|
def test_offline_returns_empty_list():
|
|
assert asyncio.run(fetch_provider_models("offline", api_base="", timeout=2)) == []
|