Files
flowdeck/app/services/llm_client.py
T
bruno 113f880863
FlowDeck CI / test (push) Failing after 11s
FlowDeck CI / docker (push) Skipped
fix(agent): v4.15.1 - protocole tool_calls conforme + plus de repli silencieux sur le mock (echos)
- L'engine envoie le message assistant avec ses tool_calls (id) et chaque resultat d'outil avec son tool_call_id, y compris en cas de refus; la passe suivante n'est plus refusee par l'API
- Un fournisseur reel configure qui echoue remonte maintenant une erreur (SSE error) au lieu de repeter la question via le mock hors-ligne (mock reserve a offline/sans cle)
- llm_client conserve id + arguments_raw des tool_calls
- test de regression test_engine_tool_protocol_messages
2026-09-06 13:06:41 -04:00

465 lines
20 KiB
Python

"""FlowDeck — LLM client abstraction (v4.10.0).
Abstraction over multiple LLM providers so the agent never talks to the DB
directly — it emits *tool intentions* (function calls) that AgentEngine turns
into guarded internal actions.
Supported providers (OpenAI-compatible chat-completions JSON response):
openai, anthropic*, google*, deepseek, qwencloud, nvidia, openrouter, ollama.
(* routed through an OpenAI-compatible gateway / any configured api_base)
When no API key is configured (or provider == "offline") the client falls back
to a deterministic, dependency-free *mock planner*. This keeps the whole agent
functional — and fully testable — with zero external calls, which is what the
local deployment and the test-suite rely on.
"""
from __future__ import annotations
import asyncio
import json
import logging
import re
from dataclasses import dataclass, field
import httpx
from app.config import settings
logger = logging.getLogger(__name__)
# Provider → default model + base URL when llm_model/api_base are empty.
PROVIDERS = {
"openai": ("https://api.openai.com/v1", "gpt-4o"),
"anthropic": ("https://api.anthropic.com/v1", "claude-opus-4-8"),
"google": ("https://generativelanguage.googleapis.com/v1beta", "gemini-2.0-pro"),
"deepseek": ("https://api.deepseek.com/v1", "deepseek-chat"),
"qwencloud": ("https://dashscope.aliyuncs.com/compatible-mode/v1", "qwen-max"),
"nvidia": ("https://integrate.api.nvidia.com/v1", "nvidia/llama-3.1-70b-instruct"),
"openrouter": ("https://openrouter.ai/api/v1", "meta-llama/llama-3.3-70b-instruct"),
"ollama": ("http://localhost:11434/v1", "llama3.1"),
"offline": (None, None),
}
# Curated model presets surfaced by /api/agent/providers for the UI selectors.
PROVIDER_MODELS: dict[str, list[str]] = {
"openai": ["gpt-4o", "gpt-4o-mini", "gpt-4.1", "gpt-4.1-mini", "o3-mini", "gpt-4-turbo"],
"anthropic": ["claude-opus-4-8", "claude-sonnet-4-5", "claude-3-5-sonnet", "claude-haiku-4-5"],
"google": ["gemini-2.0-pro", "gemini-2.0-flash", "gemini-1.5-pro", "gemini-1.5-flash"],
"deepseek": ["deepseek-chat", "deepseek-reasoner"],
"qwencloud": ["qwen-max", "qwen-plus", "qwen-turbo", "qwen-long"],
"nvidia": ["nvidia/llama-3.1-70b-instruct", "nvidia/nemotron-4-340b-instruct"],
"openrouter": ["meta-llama/llama-3.3-70b-instruct", "anthropic/claude-3.5-sonnet",
"openai/gpt-4o", "mistralai/mistral-large"],
"ollama": ["llama3.1", "llama3", "mistral", "qwen2.5", "gemma2", "mixtral"],
"offline": [],
}
# Llama-style / ChatML tool markers used by the mock planner.
_CREATE_PATTERNS = [
(re.compile(r"cr[eéé]er\s+(?:une\s+)?collection[:\s]+[\"']?([A-Za-zÀ-ÿ0-9 _\-]+)"),
lambda m: ("create_collection", {"name": m.group(1).strip()})),
(re.compile(r"create\s+collection\s+[\"']?([A-Za-z0-9 _\-]+)"),
lambda m: ("create_collection", {"name": m.group(1).strip()})),
(re.compile(r"create\s+a\s+page\s+[\"']?([A-Za-z0-9 _\-]+)"),
lambda m: ("create_page", {"title": m.group(1).strip()})),
]
_SEARCH_PATTERNS = [
(re.compile(r"(?:recherche|search|trouve|find)\s+[\"']?([A-Za-z0-9 _\-]+)"),
lambda m: ("search_workspace", {"query": m.group(1).strip()})),
]
# Loose fallback: "collection <Name>" → create_collection (covers "crée une collection X",
# "créer la collection X", "create collection X", etc.)
_COLLECTION_LINE = re.compile(
r"\bcollection\b[:\s]+(?:nomm[ée]e\s+)?([A-Za-zÀ-ÿ0-9_][^,.\n()]*[A-Za-zÀ-ÿ0-9_])",
re.IGNORECASE,
)
@dataclass
class LLMResponse:
"""Normalized completion: either a final text or one or more tool calls."""
text: str = ""
tool_calls: list[dict] = field(default_factory=list)
model: str = ""
usage: dict = field(default_factory=dict)
class LLMClient:
"""Multi-provider chat client with tool-calling support and offline mock."""
def __init__(self, provider: str | None = None, api_key: str | None = None,
api_base: str | None = None):
from .llm_config import get_llm_config # local import avoids a cycle
cfg = get_llm_config()
self.provider = (provider or cfg["provider"] or "offline").lower()
self.api_key = api_key if api_key is not None else cfg["api_key"]
self.api_base = api_base if api_base is not None else cfg["api_base"]
base, model = PROVIDERS.get(self.provider, (None, None))
self.api_base = self.api_base or base
self.default_model = cfg["model"] or model or "gpt-4o"
# ── Public API ──
async def complete(self, messages: list[dict], *, model: str | None = None,
tools: list[dict] | None = None,
stream: bool = False) -> LLMResponse:
"""Send a chat completion. Returns text and/or tool_calls."""
model = model or self.default_model
if self.provider == "offline" or not self._has_credentials():
return await self._mock_complete(messages, model, tools)
try:
return await asyncio.wait_for(
self._http_complete(messages, model, tools),
timeout=settings.agent_run_timeout_seconds,
)
except Exception as exc: # noqa: BLE001 — never mask a real-provider failure
# On NE retombe PAS silencieusement sur le mock quand un fournisseur
# réel est configuré : l'erreur doit remonter (SSE "error") pour que
# l'utilisateur voie pourquoi rien n'a été généré.
logger.warning("LLM provider '%s' failed (%s)", self.provider, exc)
raise
async def is_available(self) -> bool:
"""True when a real provider is configured."""
return self.provider != "offline" and self._has_credentials()
async def ping(self, *, model: str | None = None) -> LLMResponse:
"""Reach the provider without mock fallback (used by the "Test connection"
UI). Raises on any real error so the caller can surface it."""
model = model or self.default_model
if self.provider == "offline":
return LLMResponse(text="Mode hors-ligne (mock) — aucun appel réseau nécessaire.", model=model)
if not self._has_credentials():
raise PermissionError(f"Clé API manquante pour le provider « {self.provider} »")
return await asyncio.wait_for(
self._http_complete(
[{"role": "user", "content": "Réponds uniquement par le mot : PONG"}],
model,
None,
),
timeout=settings.agent_run_timeout_seconds,
)
# ── Helpers ──
def _has_credentials(self) -> bool:
if self.provider == "ollama":
return True # local, no key required
return bool(self.api_key)
def _endpoint(self) -> str:
return f"{self.api_base.rstrip('/')}/chat/completions"
async def _http_complete(self, messages, model, tools) -> LLMResponse:
payload: dict = {
"model": model,
"messages": messages,
"temperature": 0.2,
}
if tools:
payload["tools"] = [{"type": "function", "function": t} for t in tools]
payload["tool_choice"] = "auto"
headers = {"Content-Type": "application/json"}
if self.api_key:
headers["Authorization"] = f"Bearer {self.api_key}"
async with httpx.AsyncClient(timeout=settings.agent_run_timeout_seconds) as client:
resp = await client.post(self._endpoint(), json=payload, headers=headers)
resp.raise_for_status()
data = resp.json()
choice = data["choices"][0]["message"]
text = choice.get("content") or ""
tool_calls = []
for tc in choice.get("tool_calls") or []:
fn = tc.get("function") or {}
try:
args = json.loads(fn.get("arguments") or "{}")
except json.JSONDecodeError:
args = {}
tool_calls.append({
"id": tc.get("id") or "",
"name": fn.get("name"),
"arguments": args,
"arguments_raw": fn.get("arguments") or "",
})
return LLMResponse(
text=text,
tool_calls=tool_calls,
model=model,
usage=data.get("usage", {}),
)
# ── Offline mock planner (deterministic, no network) ──
async def _mock_complete(self, messages, model, tools) -> LLMResponse:
user_content = self._last_user_content(messages)
sys_content = self._system_content(messages)
# Only the user's objective drives the planner. The engine appends the
# workspace/document snapshot under "# Contexte"; that text must never
# trigger keyword heuristics (a doc mentioning "recherche"/"collection"
# used to misroute content requests into tool calls).
objective = user_content.split("\n# Contexte")[0]
# Once tool results are already in the conversation, we have acted:
# stop issuing new tool calls and conclude.
if any(m.get("role") == "tool" for m in messages):
return LLMResponse(
text="Objectif traité — actions enregistrées dans le journal d'audit.",
model=model,
)
# Skill-driven: if the objective names a known skill, mirror its template.
skill_hint = self._extract_skill_hint(objective)
if skill_hint == "sprint":
return LLMResponse(
tool_calls=[
{"name": "read_gitea_issues", "arguments": {"owner": "bruno", "repo": "flowdeck", "state": "open"}},
{"name": "create_collection", "arguments": {"name": "Sprint"}},
{"name": "add_property", "arguments": {"collection_id": 0, "name": "Status", "prop_type": "select", "options": ["Todo", "In Progress", "Done"]}},
{"name": "create_view", "arguments": {"collection_id": 0, "view_type": "board"}},
],
text="Plan: analyze open issues, then build a sprint board.",
model=model,
)
# Inline content-generation ("Ask AI" / "AI meeting note") — answered
# before the tool-intent heuristics and scoped to the user objective
# only, so an injected "# Contexte" that happens to mention "collection"
# can't misroute a writing request into a create-collection action.
draft = self._draft_reply(objective)
if draft:
return LLMResponse(text=draft, model=model)
# Documents / espaces de travail (offline): unambiguous intents resolved
# from the objective — create a document (optionally in a named
# workspace) or list the accessible workspaces.
doc_args = self._document_create_args(objective)
if doc_args is not None:
return LLMResponse(
tool_calls=[{"name": "create_document", "arguments": doc_args}],
text="Plan: création d'un document.",
model=model,
)
if self._is_workspaces_request(objective):
return LLMResponse(
tool_calls=[{"name": "read_workspaces", "arguments": {}}],
text="Plan: lister les espaces de travail.",
model=model,
)
# Exact keyword → tool intent resolution (objective only).
for regex, builder in _CREATE_PATTERNS:
m = regex.search(objective)
if m:
return LLMResponse(
tool_calls=[dict(name=name, arguments=self._bind_placeholders(args, sys_content)) for name, args in [builder(m)]],
text=f"Plan: running {builder(m)[0]}.",
model=model,
)
for regex, builder in _SEARCH_PATTERNS:
m = regex.search(objective)
if m:
return LLMResponse(
tool_calls=[dict(name=name, arguments=args) for name, args in [builder(m)]],
text=f"Plan: searching '{m.group(1)}'.",
model=model,
)
# Loose "collection <X>" detection → treat as a create intent.
low = objective.lower()
if "collection" in low:
m = _COLLECTION_LINE.search(objective)
if m:
name = m.group(1).strip()
return LLMResponse(
tool_calls=[{"name": "create_collection",
"arguments": self._bind_placeholders({"name": name}, sys_content)}],
text=f"Plan: create collection '{name}'.",
model=model,
)
# Plain conversational objective → final answer (no tool).
return LLMResponse(
text=self._summarize(objective),
model=model,
)
def _bind_placeholders(self, args: dict, sys_content: str) -> dict:
"""Inject a collection id from the context when the planner left it as 0."""
args = dict(args)
if args.get("collection_id") == 0:
match = re.search(r"Collection IDs?:\s*([0-9,\s]+)", sys_content)
if match:
ids = [int(x) for x in re.split(r"[,\s]+", match.group(1).strip()) if x.isdigit()]
if ids:
args["collection_id"] = ids[0]
return args
@staticmethod
def _document_create_args(content: str) -> dict | None:
"""Deterministic `create_document` intent for the offline mock.
Only fires when the user clearly asks to *create* a document (a content
rewrite such as « résume / traduis ce document » is left to `_draft_reply`).
Returns None when the message is not a create-document intent.
"""
low = content.lower()
if "document" not in low:
return None
if not any(k in low for k in ("création", "créer", "crée", "crées", "create",
"nouveau document", "nouvelle page", "faire un")):
return None
quotes = re.findall(r'[«"]([^«»"]{1,80})[»"]', content)
title = quotes[0].strip() if quotes else None
if not title:
m = re.search(
r"\bdocument\b\s*(?:nomm[ée]e?\s+|intitul[ée]e?\s+|appel[ée]e?\s+)?"
r'[«"]?\s*([A-Za-zÀ-ÿ0-9][A-Za-zÀ-ÿ0-9_ \-]{1,60})',
content, re.IGNORECASE,
)
if m:
title = m.group(1).strip()
if not title:
return None
args = {"title": title}
if len(quotes) > 1 and re.search(r"\b(workspace|espace de travail)\b", low):
args["workspace_name"] = quotes[-1].strip()
return args
@staticmethod
def _is_workspaces_request(content: str) -> bool:
"""True when the user asks to list / locate the workspaces."""
low = content.lower()
has_ws = any(w in low for w in ("workspace", "espace de travail", "espaces de travail"))
has_verb = any(v in low for v in ("liste", "lister", "list", "quels", "montre",
"affiche", "mes espaces", "ou sont", "où sont"))
return has_ws and has_verb
@staticmethod
def _system_content(messages) -> str:
return "\n".join(m.get("content", "") for m in messages if m.get("role") == "system")
@staticmethod
def _last_user_content(messages) -> str:
for m in reversed(messages):
if m.get("role") == "user":
c = m.get("content", "")
if isinstance(c, list):
return " ".join(p.get("text", "") for p in c if isinstance(p, dict))
return str(c)
return ""
@staticmethod
def _extract_skill_hint(content: str) -> str | None:
low = content.lower()
if "sprint" in low or "préparation de sprint" in low:
return "sprint"
return None
@staticmethod
def _summarize(content: str) -> str:
"""Produce a terse final summary from a conversational objective."""
content = content.split("\n# Contexte")[0]
return (content[:600] + "…" if len(content) > 600 else content)
def _draft_reply(self, content: str) -> str | None:
"""Generate usable structured copy for content/meeting requests when no
real LLM is configured (offline mock). Returns None when the message is
not a clear content-generation intent so other paths keep their behavior.
"""
from datetime import date
low = content.lower()
title = None
m = re.search(r"intitul[ée]e\s*[«\"']([^»\"']+)[»\"']", content)
if m:
title = m.group(1).strip()
today = date.today().isoformat()
# ── AI meeting note template ──
if any(k in low for k in ("ai meeting note", "meeting note",
"compte-rendu", "compte rendu",
"notes de réunion", "réunion")):
return (
"📅 AI Meeting Note\n"
f"Date : {today} · Participants : (à renseigner)\n"
"\n"
"## Résumé\n"
"Point central de la discussion et contexte (à compléter).\n"
"\n"
"## Décisions\n"
"• Décision 1 — valider le périmètre et les responsables.\n"
"• Décision 2 — définir la prochaine échéance.\n"
"\n"
"## Action items\n"
"☐ Action 1 — responsable : …, échéance : …\n"
"☐ Action 2 — responsable : …, échéance : …\n"
"\n"
"## Prochaines étapes\n"
"• Planifier le suivi et archiver ce compte-rendu.\n"
)
# ── Traduction / analyse du document (hors-ligne) ──
if re.search(r"traduis|traduit|translate", low):
return (
"⚠️ **Traduction non disponible en mode hors-ligne** (aucun modèle d'IA "
"connecté).\n\n"
"Connectez un fournisseur dans **Paramètres → Agent & IA**, puis relancez "
"« Traduire cette page » : le document traduit apparaîtra ici, avec un aperçu "
"à approuver ou à rejeter avant application."
)
if re.search(r"r[ée]sum|am[ée]lior|sugg[èe]re des|propose des", low):
return (
"⚠️ **Cette action nécessite un modèle d'IA connecté** pour analyser le "
"document.\n\n"
"Configurez une clé API dans **Paramètres → Agent & IA**, puis relancez "
"l'action : l'agent générera la proposition ici, avec un aperçu à approuver "
"ou à rejeter avant application."
)
# ── Page / document draft (contextual "Ask AI") ──
if title:
t = title[:80]
return (
f"{t}\n"
"\n"
f"Présentation générale du sujet « {t} » : objectif, contexte et "
"public visé en quelques phrases. (Document généré hors-ligne — "
"connectez une clé API pour une rédaction complète.)\n"
"\n"
"## Objectif\n"
"• Clarifier le besoin couvert par ce document.\n"
"• Lister les livrables attendus.\n"
"\n"
"## Points clés\n"
"• Idée principale 1 avec les arguments associés.\n"
"• Idée principale 2 et les exemples concrets.\n"
"\n"
"## Prochaines étapes\n"
"• Relire, compléter et mettre en forme ce contenu.\n"
)
# ── Generic drafting verb, no title (typing directly in the chat) ──
if re.search(r"^(r[ée]dige|[ée]cris|[ée]crire|g[ée]n[èe]re|produis|d[ée]veloppe|"
r"[ée]cris\s+un|g[ée]n[èe]re\s+un|r[ée]dige\s+un)\b", low):
return (
"## Introduction\n"
"Contexte et objectif de ce texte, en une à deux phrases.\n"
"\n"
"## Développement\n"
"• Premier argument structuré avec un exemple.\n"
"• Deuxième argument appuyé par une donnée ou un fait.\n"
"\n"
"## Conclusion\n"
"Synthèse et prochaine étape recommandée.\n"
)
return None