"""FlowDeck — LLM client abstraction (v4.10.0). Abstraction over multiple LLM providers so the agent never talks to the DB directly — it emits *tool intentions* (function calls) that AgentEngine turns into guarded internal actions. Supported providers (OpenAI-compatible chat-completions JSON response): openai, anthropic*, google*, deepseek, qwencloud, nvidia, openrouter, ollama. (* routed through an OpenAI-compatible gateway / any configured api_base) When no API key is configured (or provider == "offline") the client falls back to a deterministic, dependency-free *mock planner*. This keeps the whole agent functional — and fully testable — with zero external calls, which is what the local deployment and the test-suite rely on. """ from __future__ import annotations import asyncio import json import logging import re from dataclasses import dataclass, field import httpx from app.config import settings logger = logging.getLogger(__name__) # Provider → default model + base URL when llm_model/api_base are empty. PROVIDERS = { "openai": ("https://api.openai.com/v1", "gpt-4o"), "anthropic": ("https://api.anthropic.com/v1", "claude-opus-4-8"), "google": ("https://generativelanguage.googleapis.com/v1beta", "gemini-2.0-pro"), "deepseek": ("https://api.deepseek.com/v1", "deepseek-chat"), "qwencloud": ("https://dashscope.aliyuncs.com/compatible-mode/v1", "qwen-max"), "nvidia": ("https://integrate.api.nvidia.com/v1", "nvidia/nemotron-3-super-120b-a12b"), "openrouter": ("https://openrouter.ai/api/v1", "meta-llama/llama-3.3-70b-instruct"), "ollama": ("http://localhost:11434/v1", "llama3.1"), "offline": (None, None), } # Curated model presets surfaced by /api/agent/providers for the UI selectors. PROVIDER_MODELS: dict[str, list[str]] = { "openai": ["gpt-4o", "gpt-4o-mini", "gpt-4.1", "gpt-4.1-mini", "o3-mini", "gpt-4-turbo"], "anthropic": ["claude-opus-4-8", "claude-sonnet-4-5", "claude-3-5-sonnet", "claude-haiku-4-5"], "google": ["gemini-2.0-pro", "gemini-2.0-flash", "gemini-1.5-pro", "gemini-1.5-flash"], "deepseek": ["deepseek-chat", "deepseek-reasoner"], "qwencloud": ["qwen-max", "qwen-plus", "qwen-turbo", "qwen-long"], "nvidia": ["nvidia/nemotron-3-super-120b-a12b", "nvidia/nemotron-3-nano-30b-a3b", "meta/llama-3.1-70b-instruct", "nvidia/llama-3.3-nemotron-super-49b-v1.5", "deepseek-ai/deepseek-v4-pro", "z-ai/glm-5.2"], "openrouter": ["meta-llama/llama-3.3-70b-instruct", "anthropic/claude-3.5-sonnet", "openai/gpt-4o", "mistralai/mistral-large"], "ollama": ["llama3.1", "llama3", "mistral", "qwen2.5", "gemma2", "mixtral"], "offline": [], } # Llama-style / ChatML tool markers used by the mock planner. _CREATE_PATTERNS = [ (re.compile(r"cr[eéé]er\s+(?:une\s+)?collection[:\s]+[\"']?([A-Za-zÀ-ÿ0-9 _\-]+)"), lambda m: ("create_collection", {"name": m.group(1).strip()})), (re.compile(r"create\s+collection\s+[\"']?([A-Za-z0-9 _\-]+)"), lambda m: ("create_collection", {"name": m.group(1).strip()})), (re.compile(r"create\s+a\s+page\s+[\"']?([A-Za-z0-9 _\-]+)"), lambda m: ("create_page", {"title": m.group(1).strip()})), ] _SEARCH_PATTERNS = [ (re.compile(r"(?:recherche|search|trouve|find)\s+[\"']?([A-Za-z0-9 _\-]+)"), lambda m: ("search_workspace", {"query": m.group(1).strip()})), ] # Loose fallback: "collection " → create_collection (covers "crée une collection X", # "créer la collection X", "create collection X", etc.) _COLLECTION_LINE = re.compile( r"\bcollection\b[:\s]+(?:nomm[ée]e\s+)?([A-Za-zÀ-ÿ0-9_][^,.\n()]*[A-Za-zÀ-ÿ0-9_])", re.IGNORECASE, ) @dataclass class LLMResponse: """Normalized completion: either a final text or one or more tool calls.""" text: str = "" tool_calls: list[dict] = field(default_factory=list) model: str = "" usage: dict = field(default_factory=dict) notice: str = "" class LLMClient: """Multi-provider chat client with tool-calling support and offline mock.""" def __init__(self, provider: str | None = None, api_key: str | None = None, api_base: str | None = None): from .llm_config import get_llm_config # local import avoids a cycle cfg = get_llm_config() self.provider = (provider or cfg["provider"] or "offline").lower() self.api_key = api_key if api_key is not None else cfg["api_key"] self.api_base = api_base if api_base is not None else cfg["api_base"] base, model = PROVIDERS.get(self.provider, (None, None)) self.api_base = self.api_base or base # Le modèle global configuré n'est valable que pour le provider global : # tester un autre provider (ex. nvidia alors que deepseek est actif) ne doit # PAS lui envoyer le modèle du provider actif (sinon « model not found »). cfg_provider = (cfg.get("provider") or "offline").lower() global_model = (cfg.get("model") or "") if self.provider == cfg_provider else "" self.default_model = global_model or model or "gpt-4o" # ── Public API ── async def complete(self, messages: list[dict], *, model: str | None = None, tools: list[dict] | None = None, stream: bool = False) -> LLMResponse: """Send a chat completion. Returns text and/or tool_calls.""" model = model or self.default_model if self.provider == "offline" or not self._has_credentials(): return await self._mock_complete(messages, model, tools) try: return await asyncio.wait_for( self._http_complete(messages, model, tools), timeout=settings.agent_run_timeout_seconds, ) except Exception as exc: # noqa: BLE001 — never mask a real-provider failure # On NE retombe PAS silencieusement sur le mock quand un fournisseur # réel est configuré : l'erreur doit remonter (SSE "error") pour que # l'utilisateur voie pourquoi rien n'a été généré. logger.warning("LLM provider '%s' failed (%s)", self.provider, exc) raise async def is_available(self) -> bool: """True when a real provider is configured.""" return self.provider != "offline" and self._has_credentials() async def ping(self, *, model: str | None = None) -> LLMResponse: """Reach the provider without mock fallback (used by the "Test connection" UI). Raises on any real error so the caller can surface it.""" model = model or self.default_model if self.provider == "offline": return LLMResponse(text="Mode hors-ligne (mock) — aucun appel réseau nécessaire.", model=model) if not self._has_credentials(): raise PermissionError(f"Clé API manquante pour le provider « {self.provider} »") return await asyncio.wait_for( self._http_complete( [{"role": "user", "content": "Réponds uniquement par le mot : PONG"}], model, None, ), timeout=settings.agent_run_timeout_seconds, ) # ── Helpers ── def _has_credentials(self) -> bool: if self.provider == "ollama": return True # local, no key required return bool(self.api_key) def _endpoint(self) -> str: return f"{self.api_base.rstrip('/')}/chat/completions" async def _http_complete(self, messages, model, tools, *, _noticer: str = "") -> LLMResponse: payload: dict = { "model": model, "messages": messages, "temperature": 0.2, } if tools: payload["tools"] = [{"type": "function", "function": t} for t in tools] payload["tool_choice"] = "auto" headers = {"Content-Type": "application/json"} if self.api_key: headers["Authorization"] = f"Bearer {self.api_key}" try: async with httpx.AsyncClient(timeout=settings.agent_run_timeout_seconds) as client: resp = await client.post(self._endpoint(), json=payload, headers=headers) resp.raise_for_status() data = resp.json() except httpx.HTTPStatusError as exc: # Repli robuste : le modèle choisi a été retiré / n'existe plus # (404 « model not found » / 410 « has reached its end of life »). # Au lieu d'échouer, on retente UNE fois avec le modèle par défaut du # provider et on signale le basculement — la liste validée peut avoir # vieilli (modèle déprécié entre deux rafraîchissements). if exc.response.status_code in (404, 410) \ and model and self.default_model and model != self.default_model: logger.warning( "Model '%s' unavailable (%s) on %s — retrying with default '%s'", model, exc.response.status_code, self.provider, self.default_model, ) return await self._http_complete(messages, self.default_model, tools, _noticer=( f"Le modèle « {model} » n'est plus disponible ({exc.response.status_code}). " f"Réponse générée avec « {self.default_model} » à la place." )) raise response = self._parse_response(data, model) response.notice = _noticer or "" return response def _parse_response(self, data: dict, model: str) -> LLMResponse: choice = data["choices"][0]["message"] text = choice.get("content") or "" tool_calls = [] for tc in choice.get("tool_calls") or []: fn = tc.get("function") or {} try: args = json.loads(fn.get("arguments") or "{}") except json.JSONDecodeError: args = {} tool_calls.append({ "id": tc.get("id") or "", "name": fn.get("name"), "arguments": args, "arguments_raw": fn.get("arguments") or "", }) return LLMResponse( text=text, tool_calls=tool_calls, model=model, usage=data.get("usage", {}), ) # ── Offline mock planner (deterministic, no network) ── async def _mock_complete(self, messages, model, tools) -> LLMResponse: user_content = self._last_user_content(messages) sys_content = self._system_content(messages) # Only the user's objective drives the planner. The engine appends the # workspace/document snapshot under "# Contexte"; that text must never # trigger keyword heuristics (a doc mentioning "recherche"/"collection" # used to misroute content requests into tool calls). objective = user_content.split("\n# Contexte")[0] # Once tool results are already in the conversation, we have acted: # stop issuing new tool calls and conclude. if any(m.get("role") == "tool" for m in messages): return LLMResponse( text="Objectif traité — actions enregistrées dans le journal d'audit.", model=model, ) # Skill-driven: if the objective names a known skill, mirror its template. skill_hint = self._extract_skill_hint(objective) if skill_hint == "sprint": return LLMResponse( tool_calls=[ {"name": "read_gitea_issues", "arguments": {"owner": "bruno", "repo": "flowdeck", "state": "open"}}, {"name": "create_collection", "arguments": {"name": "Sprint"}}, {"name": "add_property", "arguments": {"collection_id": 0, "name": "Status", "prop_type": "select", "options": ["Todo", "In Progress", "Done"]}}, {"name": "create_view", "arguments": {"collection_id": 0, "view_type": "board"}}, ], text="Plan: analyze open issues, then build a sprint board.", model=model, ) # Inline content-generation ("Ask AI" / "AI meeting note") — answered # before the tool-intent heuristics and scoped to the user objective # only, so an injected "# Contexte" that happens to mention "collection" # can't misroute a writing request into a create-collection action. draft = self._draft_reply(objective) if draft: return LLMResponse(text=draft, model=model) # Documents / espaces de travail (offline): unambiguous intents resolved # from the objective — create a document (optionally in a named # workspace) or list the accessible workspaces. doc_args = self._document_create_args(objective) if doc_args is not None: return LLMResponse( tool_calls=[{"name": "create_document", "arguments": doc_args}], text="Plan: création d'un document.", model=model, ) if self._is_workspaces_request(objective): return LLMResponse( tool_calls=[{"name": "read_workspaces", "arguments": {}}], text="Plan: lister les espaces de travail.", model=model, ) # Exact keyword → tool intent resolution (objective only). for regex, builder in _CREATE_PATTERNS: m = regex.search(objective) if m: return LLMResponse( tool_calls=[dict(name=name, arguments=self._bind_placeholders(args, sys_content)) for name, args in [builder(m)]], text=f"Plan: running {builder(m)[0]}.", model=model, ) for regex, builder in _SEARCH_PATTERNS: m = regex.search(objective) if m: return LLMResponse( tool_calls=[dict(name=name, arguments=args) for name, args in [builder(m)]], text=f"Plan: searching '{m.group(1)}'.", model=model, ) # Loose "collection " detection → treat as a create intent. low = objective.lower() if "collection" in low: m = _COLLECTION_LINE.search(objective) if m: name = m.group(1).strip() return LLMResponse( tool_calls=[{"name": "create_collection", "arguments": self._bind_placeholders({"name": name}, sys_content)}], text=f"Plan: create collection '{name}'.", model=model, ) # Plain conversational objective → final answer (no tool). return LLMResponse( text=self._summarize(objective), model=model, ) def _bind_placeholders(self, args: dict, sys_content: str) -> dict: """Inject a collection id from the context when the planner left it as 0.""" args = dict(args) if args.get("collection_id") == 0: match = re.search(r"Collection IDs?:\s*([0-9,\s]+)", sys_content) if match: ids = [int(x) for x in re.split(r"[,\s]+", match.group(1).strip()) if x.isdigit()] if ids: args["collection_id"] = ids[0] return args @staticmethod def _document_create_args(content: str) -> dict | None: """Deterministic `create_document` intent for the offline mock. Only fires when the user clearly asks to *create* a document (a content rewrite such as « résume / traduis ce document » is left to `_draft_reply`). Returns None when the message is not a create-document intent. """ low = content.lower() if "document" not in low: return None if not any(k in low for k in ("création", "créer", "crée", "crées", "create", "nouveau document", "nouvelle page", "faire un")): return None quotes = re.findall(r'[«"]([^«»"]{1,80})[»"]', content) title = quotes[0].strip() if quotes else None if not title: m = re.search( r"\bdocument\b\s*(?:nomm[ée]e?\s+|intitul[ée]e?\s+|appel[ée]e?\s+)?" r'[«"]?\s*([A-Za-zÀ-ÿ0-9][A-Za-zÀ-ÿ0-9_ \-]{1,60})', content, re.IGNORECASE, ) if m: title = m.group(1).strip() if not title: return None args = {"title": title} if len(quotes) > 1 and re.search(r"\b(workspace|espace de travail)\b", low): args["workspace_name"] = quotes[-1].strip() return args @staticmethod def _is_workspaces_request(content: str) -> bool: """True when the user asks to list / locate the workspaces.""" low = content.lower() has_ws = any(w in low for w in ("workspace", "espace de travail", "espaces de travail")) has_verb = any(v in low for v in ("liste", "lister", "list", "quels", "montre", "affiche", "mes espaces", "ou sont", "où sont")) return has_ws and has_verb @staticmethod def _system_content(messages) -> str: return "\n".join(m.get("content", "") for m in messages if m.get("role") == "system") @staticmethod def _last_user_content(messages) -> str: for m in reversed(messages): if m.get("role") == "user": c = m.get("content", "") if isinstance(c, list): return " ".join(p.get("text", "") for p in c if isinstance(p, dict)) return str(c) return "" @staticmethod def _extract_skill_hint(content: str) -> str | None: low = content.lower() if "sprint" in low or "préparation de sprint" in low: return "sprint" return None @staticmethod def _summarize(content: str) -> str: """Produce a terse final summary from a conversational objective.""" content = content.split("\n# Contexte")[0] return (content[:600] + "…" if len(content) > 600 else content) def _draft_reply(self, content: str) -> str | None: """Generate usable structured copy for content/meeting requests when no real LLM is configured (offline mock). Returns None when the message is not a clear content-generation intent so other paths keep their behavior. """ from datetime import date low = content.lower() title = None m = re.search(r"intitul[ée]e\s*[«\"']([^»\"']+)[»\"']", content) if m: title = m.group(1).strip() today = date.today().isoformat() # ── AI meeting note template ── if any(k in low for k in ("ai meeting note", "meeting note", "compte-rendu", "compte rendu", "notes de réunion", "réunion")): return ( "📅 AI Meeting Note\n" f"Date : {today} · Participants : (à renseigner)\n" "\n" "## Résumé\n" "Point central de la discussion et contexte (à compléter).\n" "\n" "## Décisions\n" "• Décision 1 — valider le périmètre et les responsables.\n" "• Décision 2 — définir la prochaine échéance.\n" "\n" "## Action items\n" "☐ Action 1 — responsable : …, échéance : …\n" "☐ Action 2 — responsable : …, échéance : …\n" "\n" "## Prochaines étapes\n" "• Planifier le suivi et archiver ce compte-rendu.\n" ) # ── Traduction / analyse du document (hors-ligne) ── if re.search(r"traduis|traduit|translate", low): return ( "⚠️ **Traduction non disponible en mode hors-ligne** (aucun modèle d'IA " "connecté).\n\n" "Connectez un fournisseur dans **Paramètres → Agent & IA**, puis relancez " "« Traduire cette page » : le document traduit apparaîtra ici, avec un aperçu " "à approuver ou à rejeter avant application." ) if re.search(r"r[ée]sum|am[ée]lior|sugg[èe]re des|propose des", low): return ( "⚠️ **Cette action nécessite un modèle d'IA connecté** pour analyser le " "document.\n\n" "Configurez une clé API dans **Paramètres → Agent & IA**, puis relancez " "l'action : l'agent générera la proposition ici, avec un aperçu à approuver " "ou à rejeter avant application." ) # ── Page / document draft (contextual "Ask AI") ── if title: t = title[:80] return ( f"{t}\n" "\n" f"Présentation générale du sujet « {t} » : objectif, contexte et " "public visé en quelques phrases. (Document généré hors-ligne — " "connectez une clé API pour une rédaction complète.)\n" "\n" "## Objectif\n" "• Clarifier le besoin couvert par ce document.\n" "• Lister les livrables attendus.\n" "\n" "## Points clés\n" "• Idée principale 1 avec les arguments associés.\n" "• Idée principale 2 et les exemples concrets.\n" "\n" "## Prochaines étapes\n" "• Relire, compléter et mettre en forme ce contenu.\n" ) # ── Generic drafting verb, no title (typing directly in the chat) ── if re.search(r"^(r[ée]dige|[ée]cris|[ée]crire|g[ée]n[èe]re|produis|d[ée]veloppe|" r"[ée]cris\s+un|g[ée]n[èe]re\s+un|r[ée]dige\s+un)\b", low): return ( "## Introduction\n" "Contexte et objectif de ce texte, en une à deux phrases.\n" "\n" "## Développement\n" "• Premier argument structuré avec un exemple.\n" "• Deuxième argument appuyé par une donnée ou un fait.\n" "\n" "## Conclusion\n" "Synthèse et prochaine étape recommandée.\n" ) return None