Files
flowdeck/app/services/llm_client.py
T
bruno 3cb9edc1f3
FlowDeck CI / test (push) Failing after 7s
FlowDeck CI / docker (push) Skipped
feat(agent): v4.10.1 - config LLM dans l'UI (sélecteur provider/modèle, config runtime, admin + test connexion)
- Panneau agent : selects provider/modèles connus persistés par conversation
  (colonnes agent_conversations.provider/model, migration idempotente)
- GET /api/agent/providers enrichi (liste providers + modèles + requires_key)
- PATCH /api/agent/providers (admin) : config runtime DB-backed (table llm_config)
- POST /api/agent/providers/test : test connexion via LLMClient.ping() sans fallback mock
- Settings > Admin > Agent & IA : provider, modèle, clé API, URL API, save + test
- VERSION -> 4.10.1, CHANGELOG + ROADMAP mis à jour, 232 tests verts
2026-09-05 10:58:09 -04:00

291 lines
12 KiB
Python

"""FlowDeck — LLM client abstraction (v4.10.0).
Abstraction over multiple LLM providers so the agent never talks to the DB
directly — it emits *tool intentions* (function calls) that AgentEngine turns
into guarded internal actions.
Supported providers (OpenAI-compatible chat-completions JSON response):
openai, anthropic*, google*, deepseek, qwencloud, nvidia, openrouter, ollama.
(* routed through an OpenAI-compatible gateway / any configured api_base)
When no API key is configured (or provider == "offline") the client falls back
to a deterministic, dependency-free *mock planner*. This keeps the whole agent
functional — and fully testable — with zero external calls, which is what the
local deployment and the test-suite rely on.
"""
from __future__ import annotations
import asyncio
import json
import logging
import re
from dataclasses import dataclass, field
import httpx
from app.config import settings
logger = logging.getLogger(__name__)
# Provider → default model + base URL when llm_model/api_base are empty.
PROVIDERS = {
"openai": ("https://api.openai.com/v1", "gpt-4o"),
"anthropic": ("https://api.anthropic.com/v1", "claude-opus-4-8"),
"google": ("https://generativelanguage.googleapis.com/v1beta", "gemini-2.0-pro"),
"deepseek": ("https://api.deepseek.com/v1", "deepseek-chat"),
"qwencloud": ("https://dashscope.aliyuncs.com/compatible-mode/v1", "qwen-max"),
"nvidia": ("https://integrate.api.nvidia.com/v1", "nvidia/llama-3.1-70b-instruct"),
"openrouter": ("https://openrouter.ai/api/v1", "meta-llama/llama-3.3-70b-instruct"),
"ollama": ("http://localhost:11434/v1", "llama3.1"),
"offline": (None, None),
}
# Curated model presets surfaced by /api/agent/providers for the UI selectors.
PROVIDER_MODELS: dict[str, list[str]] = {
"openai": ["gpt-4o", "gpt-4o-mini", "gpt-4.1", "gpt-4.1-mini", "o3-mini", "gpt-4-turbo"],
"anthropic": ["claude-opus-4-8", "claude-sonnet-4-5", "claude-3-5-sonnet", "claude-haiku-4-5"],
"google": ["gemini-2.0-pro", "gemini-2.0-flash", "gemini-1.5-pro", "gemini-1.5-flash"],
"deepseek": ["deepseek-chat", "deepseek-reasoner"],
"qwencloud": ["qwen-max", "qwen-plus", "qwen-turbo", "qwen-long"],
"nvidia": ["nvidia/llama-3.1-70b-instruct", "nvidia/nemotron-4-340b-instruct"],
"openrouter": ["meta-llama/llama-3.3-70b-instruct", "anthropic/claude-3.5-sonnet",
"openai/gpt-4o", "mistralai/mistral-large"],
"ollama": ["llama3.1", "llama3", "mistral", "qwen2.5", "gemma2", "mixtral"],
"offline": [],
}
# Llama-style / ChatML tool markers used by the mock planner.
_CREATE_PATTERNS = [
(re.compile(r"cr[eéé]er\s+(?:une\s+)?collection[:\s]+[\"']?([A-Za-zÀ-ÿ0-9 _\-]+)"),
lambda m: ("create_collection", {"name": m.group(1).strip()})),
(re.compile(r"create\s+collection\s+[\"']?([A-Za-z0-9 _\-]+)"),
lambda m: ("create_collection", {"name": m.group(1).strip()})),
(re.compile(r"create\s+a\s+page\s+[\"']?([A-Za-z0-9 _\-]+)"),
lambda m: ("create_page", {"title": m.group(1).strip()})),
]
_SEARCH_PATTERNS = [
(re.compile(r"(?:recherche|search|trouve|find)\s+[\"']?([A-Za-z0-9 _\-]+)"),
lambda m: ("search_workspace", {"query": m.group(1).strip()})),
]
# Loose fallback: "collection <Name>" → create_collection (covers "crée une collection X",
# "créer la collection X", "create collection X", etc.)
_COLLECTION_LINE = re.compile(
r"\bcollection\b[:\s]+(?:nomm[ée]e\s+)?([A-Za-zÀ-ÿ0-9_][^,.\n()]*[A-Za-zÀ-ÿ0-9_])",
re.IGNORECASE,
)
@dataclass
class LLMResponse:
"""Normalized completion: either a final text or one or more tool calls."""
text: str = ""
tool_calls: list[dict] = field(default_factory=list)
model: str = ""
usage: dict = field(default_factory=dict)
class LLMClient:
"""Multi-provider chat client with tool-calling support and offline mock."""
def __init__(self, provider: str | None = None, api_key: str | None = None,
api_base: str | None = None):
from .llm_config import get_llm_config # local import avoids a cycle
cfg = get_llm_config()
self.provider = (provider or cfg["provider"] or "offline").lower()
self.api_key = api_key if api_key is not None else cfg["api_key"]
self.api_base = api_base if api_base is not None else cfg["api_base"]
base, model = PROVIDERS.get(self.provider, (None, None))
self.api_base = self.api_base or base
self.default_model = cfg["model"] or model or "gpt-4o"
# ── Public API ──
async def complete(self, messages: list[dict], *, model: str | None = None,
tools: list[dict] | None = None,
stream: bool = False) -> LLMResponse:
"""Send a chat completion. Returns text and/or tool_calls."""
model = model or self.default_model
if self.provider == "offline" or not self._has_credentials():
return await self._mock_complete(messages, model, tools)
try:
return await asyncio.wait_for(
self._http_complete(messages, model, tools),
timeout=settings.agent_run_timeout_seconds,
)
except Exception as exc: # noqa: BLE001 — degrade gracefully to mock
logger.warning("LLM provider '%s' failed (%s); falling back to offline", self.provider, exc)
return await self._mock_complete(messages, model, tools)
async def is_available(self) -> bool:
"""True when a real provider is configured."""
return self.provider != "offline" and self._has_credentials()
async def ping(self, *, model: str | None = None) -> LLMResponse:
"""Reach the provider without mock fallback (used by the "Test connection"
UI). Raises on any real error so the caller can surface it."""
model = model or self.default_model
if self.provider == "offline":
return LLMResponse(text="Mode hors-ligne (mock) — aucun appel réseau nécessaire.", model=model)
if not self._has_credentials():
raise PermissionError(f"Clé API manquante pour le provider « {self.provider} »")
return await asyncio.wait_for(
self._http_complete(
[{"role": "user", "content": "Réponds uniquement par le mot : PONG"}],
model,
None,
),
timeout=settings.agent_run_timeout_seconds,
)
# ── Helpers ──
def _has_credentials(self) -> bool:
if self.provider == "ollama":
return True # local, no key required
return bool(self.api_key)
def _endpoint(self) -> str:
return f"{self.api_base.rstrip('/')}/chat/completions"
async def _http_complete(self, messages, model, tools) -> LLMResponse:
payload: dict = {
"model": model,
"messages": messages,
"temperature": 0.2,
}
if tools:
payload["tools"] = [{"type": "function", "function": t} for t in tools]
payload["tool_choice"] = "auto"
headers = {"Content-Type": "application/json"}
if self.api_key:
headers["Authorization"] = f"Bearer {self.api_key}"
async with httpx.AsyncClient(timeout=settings.agent_run_timeout_seconds) as client:
resp = await client.post(self._endpoint(), json=payload, headers=headers)
resp.raise_for_status()
data = resp.json()
choice = data["choices"][0]["message"]
text = choice.get("content") or ""
tool_calls = []
for tc in choice.get("tool_calls") or []:
try:
args = json.loads(tc["function"].get("arguments") or "{}")
except json.JSONDecodeError:
args = {}
tool_calls.append({"name": tc["function"]["name"], "arguments": args})
return LLMResponse(
text=text,
tool_calls=tool_calls,
model=model,
usage=data.get("usage", {}),
)
# ── Offline mock planner (deterministic, no network) ──
async def _mock_complete(self, messages, model, tools) -> LLMResponse:
user_content = self._last_user_content(messages)
sys_content = self._system_content(messages)
# Once tool results are already in the conversation, we have acted:
# stop issuing new tool calls and conclude.
if any(m.get("role") == "tool" for m in messages):
return LLMResponse(
text="Objectif traité — actions enregistrées dans le journal d'audit.",
model=model,
)
# Skill-driven: if the objective names a known skill, mirror its template.
skill_hint = self._extract_skill_hint(user_content)
if skill_hint == "sprint":
return LLMResponse(
tool_calls=[
{"name": "read_gitea_issues", "arguments": {"owner": "bruno", "repo": "flowdeck", "state": "open"}},
{"name": "create_collection", "arguments": {"name": "Sprint"}},
{"name": "add_property", "arguments": {"collection_id": 0, "name": "Status", "prop_type": "select", "options": ["Todo", "In Progress", "Done"]}},
{"name": "create_view", "arguments": {"collection_id": 0, "view_type": "board"}},
],
text="Plan: analyze open issues, then build a sprint board.",
model=model,
)
# Exact keyword → tool intent resolution.
for regex, builder in _CREATE_PATTERNS:
m = regex.search(user_content)
if m:
return LLMResponse(
tool_calls=[dict(name=name, arguments=self._bind_placeholders(args, sys_content)) for name, args in [builder(m)]],
text=f"Plan: running {builder(m)[0]}.",
model=model,
)
for regex, builder in _SEARCH_PATTERNS:
m = regex.search(user_content)
if m:
return LLMResponse(
tool_calls=[dict(name=name, arguments=args) for name, args in [builder(m)]],
text=f"Plan: searching '{m.group(1)}'.",
model=model,
)
# Loose "collection <X>" detection → treat as a create intent.
low = user_content.lower()
if "collection" in low:
m = _COLLECTION_LINE.search(user_content)
if m:
name = m.group(1).strip()
return LLMResponse(
tool_calls=[{"name": "create_collection",
"arguments": self._bind_placeholders({"name": name}, sys_content)}],
text=f"Plan: create collection '{name}'.",
model=model,
)
# Plain conversational objective → final answer (no tool).
return LLMResponse(
text=self._summarize(user_content),
model=model,
)
def _bind_placeholders(self, args: dict, sys_content: str) -> dict:
"""Inject a collection id from the context when the planner left it as 0."""
args = dict(args)
if args.get("collection_id") == 0:
match = re.search(r"Collection IDs?:\s*([0-9,\s]+)", sys_content)
if match:
ids = [int(x) for x in re.split(r"[,\s]+", match.group(1).strip()) if x.isdigit()]
if ids:
args["collection_id"] = ids[0]
return args
@staticmethod
def _system_content(messages) -> str:
return "\n".join(m.get("content", "") for m in messages if m.get("role") == "system")
@staticmethod
def _last_user_content(messages) -> str:
for m in reversed(messages):
if m.get("role") == "user":
c = m.get("content", "")
if isinstance(c, list):
return " ".join(p.get("text", "") for p in c if isinstance(p, dict))
return str(c)
return ""
@staticmethod
def _extract_skill_hint(content: str) -> str | None:
low = content.lower()
if "sprint" in low or "préparation de sprint" in low:
return "sprint"
return None
@staticmethod
def _summarize(content: str) -> str:
"""Produce a terse final summary from a conversational objective."""
content = content.split("\n# Contexte")[0]
return (content[:600] + "…" if len(content) > 600 else content)