diff --git a/CHANGELOG.md b/CHANGELOG.md index a66b4ce..5493482 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,21 @@ # Changelog — FlowDeck +## v4.15.5 (2026-09-06) — Correctif test de connexion provider (nvidia) + +### Corrections +- **Test de connexion multi-provider** : le modèle configuré globalement (ex. `deepseek-v4-flash`) + n'est plus envoyé lorsqu'on teste un **autre** provider (ex. nvidia) — c'était la cause d'un + HTTP 404 « model not found ». Chaque provider utilise désormais son modèle par défaut tant que + le modèle global ne le concerne pas. +- **Presets NVIDIA actualisés** : les anciens modèles (`nvidia/llama-3.1-70b-instruct`, + `nvidia/nemotron-4-340b-instruct`) étaient retirés de la plateforme ou soumis à abonnement ; + liste remplacée par des modèles disponibles (`nvidia/nemotron-3-super-120b-a12b`, + `meta/llama-3.1-70b-instruct`, `deepseek-ai/deepseek-v4-pro`, `z-ai/glm-5.2`, …). + +> Note : un 404 persistant après cette correction vient de NVIDIA lui-même (code +> « Function … : Not found for account ») — cause fréquente : le droit « Public API Endpoints » +> pas activé sur le compte/clé pour le modèle choisi. + ## v4.15.4 (2026-09-06) — Correctifs listes agent (@ et /) ### Corrections diff --git a/VERSION b/VERSION index fe9a0c2..5c56ff0 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -4.15.4 \ No newline at end of file +4.15.5 \ No newline at end of file diff --git a/app/main.py b/app/main.py index 2456173..10e6af4 100644 --- a/app/main.py +++ b/app/main.py @@ -60,7 +60,7 @@ async def lifespan(_app: FastAPI): app = FastAPI( title="FlowDeck", - version="4.15.4", + version="4.15.5", docs_url="/docs" if settings.log_level == "DEBUG" else None, redoc_url=None, lifespan=lifespan, diff --git a/app/services/llm_client.py b/app/services/llm_client.py index ea4dc94..5cf6631 100644 --- a/app/services/llm_client.py +++ b/app/services/llm_client.py @@ -34,7 +34,7 @@ PROVIDERS = { "google": ("https://generativelanguage.googleapis.com/v1beta", "gemini-2.0-pro"), "deepseek": ("https://api.deepseek.com/v1", "deepseek-chat"), "qwencloud": ("https://dashscope.aliyuncs.com/compatible-mode/v1", "qwen-max"), - "nvidia": ("https://integrate.api.nvidia.com/v1", "nvidia/llama-3.1-70b-instruct"), + "nvidia": ("https://integrate.api.nvidia.com/v1", "nvidia/nemotron-3-super-120b-a12b"), "openrouter": ("https://openrouter.ai/api/v1", "meta-llama/llama-3.3-70b-instruct"), "ollama": ("http://localhost:11434/v1", "llama3.1"), "offline": (None, None), @@ -47,7 +47,9 @@ PROVIDER_MODELS: dict[str, list[str]] = { "google": ["gemini-2.0-pro", "gemini-2.0-flash", "gemini-1.5-pro", "gemini-1.5-flash"], "deepseek": ["deepseek-chat", "deepseek-reasoner"], "qwencloud": ["qwen-max", "qwen-plus", "qwen-turbo", "qwen-long"], - "nvidia": ["nvidia/llama-3.1-70b-instruct", "nvidia/nemotron-4-340b-instruct"], + "nvidia": ["nvidia/nemotron-3-super-120b-a12b", "nvidia/nemotron-3-nano-30b-a3b", + "meta/llama-3.1-70b-instruct", "nvidia/llama-3.3-nemotron-super-49b-v1.5", + "deepseek-ai/deepseek-v4-pro", "z-ai/glm-5.2"], "openrouter": ["meta-llama/llama-3.3-70b-instruct", "anthropic/claude-3.5-sonnet", "openai/gpt-4o", "mistralai/mistral-large"], "ollama": ["llama3.1", "llama3", "mistral", "qwen2.5", "gemma2", "mixtral"], @@ -98,7 +100,12 @@ class LLMClient: self.api_base = api_base if api_base is not None else cfg["api_base"] base, model = PROVIDERS.get(self.provider, (None, None)) self.api_base = self.api_base or base - self.default_model = cfg["model"] or model or "gpt-4o" + # Le modèle global configuré n'est valable que pour le provider global : + # tester un autre provider (ex. nvidia alors que deepseek est actif) ne doit + # PAS lui envoyer le modèle du provider actif (sinon « model not found »). + cfg_provider = (cfg.get("provider") or "offline").lower() + global_model = (cfg.get("model") or "") if self.provider == cfg_provider else "" + self.default_model = global_model or model or "gpt-4o" # ── Public API ── diff --git a/tests/test_agent.py b/tests/test_agent.py index 50f3001..67f354a 100644 --- a/tests/test_agent.py +++ b/tests/test_agent.py @@ -695,6 +695,27 @@ def test_router_patch_providers_requires_admin(client): assert r.status_code == 403 +def test_llm_client_default_model_provider_scoped(client): + # Le modèle global ne doit pas fuir vers un autre provider testé (ex. nvidia + # alors que deepseek est actif) sinon le provider reçoit un modèle inconnu + # (« model not found », HTTP 404). + from app.services.llm_config import set_llm_config + from app.services.llm_client import LLMClient + + set_llm_config(provider="deepseek", model="deepseek-v4-flash", + api_key="sk-global", api_base="https://api.deepseek.com/v1") + + llm = LLMClient(provider="nvidia", api_key="nvapi-xxx") + assert llm.provider == "nvidia" + assert llm.default_model == "nvidia/nemotron-3-super-120b-a12b" + assert llm.default_model != "deepseek-v4-flash" + + # le provider global conserve bien son modèle + llm2 = LLMClient() + assert llm2.provider == "deepseek" + assert llm2.default_model == "deepseek-v4-flash" + + def test_router_providers_test_offline(client): r = client.post("/api/agent/providers/test", json={"provider": "offline"}) assert r.status_code == 200