fix(ai): garantir une reponse finale quand la boucle d'agent epuise son budget (BUG-052)
This commit is contained in:
+85
-17
@@ -40,6 +40,15 @@ MAX_TOOL_RESULT_CHARS = 100_000
|
||||
# Quota: maximum tool calls executed per agent run (``BOOKSLM_MAX_TOOL_CALLS``).
|
||||
DEFAULT_MAX_TOOL_CALLS = int(os.environ.get("BOOKSLM_MAX_TOOL_CALLS", "25"))
|
||||
|
||||
# Sent as a last user turn when the loop stopped before the model produced an
|
||||
# answer (iteration/quota budget exhausted while it was still calling tools).
|
||||
_FINALIZE_INSTRUCTION = (
|
||||
"N'appelle plus aucun outil. Réponds maintenant directement à l'utilisateur, "
|
||||
"en français, à partir des informations déjà recueillies ci-dessus. "
|
||||
"Structure la réponse en Markdown, cite les liens sources utiles, et si les "
|
||||
"informations sont insuffisantes, dis-le explicitement."
|
||||
)
|
||||
|
||||
# Stopping reasons
|
||||
STOP_DONE = "done"
|
||||
STOP_MAX_ITERATIONS = "max_iterations"
|
||||
@@ -109,8 +118,8 @@ def _assistant_tool_message(content: str | None, tool_calls: list[Any]) -> dict[
|
||||
}
|
||||
|
||||
|
||||
def _deferred_tool_message(call: Any) -> dict[str, Any]:
|
||||
"""Answer a tool call that was not reached because the run paused.
|
||||
def _deferred_tool_message(call: Any, reason: str | None = None) -> dict[str, Any]:
|
||||
"""Answer a tool call that was not reached because the run stopped early.
|
||||
|
||||
A single LLM response may carry several tool calls. When one of them is
|
||||
mutating and pauses the run for confirmation, the assistant message already
|
||||
@@ -125,7 +134,7 @@ def _deferred_tool_message(call: Any) -> dict[str, Any]:
|
||||
"name": call.name,
|
||||
"content": json.dumps({
|
||||
"status": "deferred",
|
||||
"reason": (
|
||||
"reason": reason or (
|
||||
"Not executed: the run paused to confirm an earlier tool call. "
|
||||
"Re-issue this call if it is still needed."
|
||||
),
|
||||
@@ -133,6 +142,69 @@ def _deferred_tool_message(call: Any) -> dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
def _fallback_summary(executed: list[ToolCallRecord]) -> str:
|
||||
"""Deterministic non-empty answer built from the gathered tool results.
|
||||
|
||||
Used only if the final synthesis call fails or returns nothing, so a turn
|
||||
never ends on an empty message (BUG-052).
|
||||
"""
|
||||
lines: list[str] = []
|
||||
for record in executed:
|
||||
data = record.result
|
||||
if not isinstance(data, dict):
|
||||
continue
|
||||
for item in (data.get("results") or [])[:5]:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
title = item.get("title") or item.get("url") or ""
|
||||
url = item.get("url") or ""
|
||||
lines.append(f"- [{title}]({url})" if url else f"- {title}")
|
||||
if data.get("url") and data.get("text"):
|
||||
title = data.get("title") or data["url"]
|
||||
lines.append(f"- [{title}]({data['url']})")
|
||||
if not lines:
|
||||
return "Je n'ai pas pu produire de réponse à partir des résultats obtenus."
|
||||
unique = list(dict.fromkeys(lines))
|
||||
return "Voici les sources pertinentes trouvées :\n" + "\n".join(unique)
|
||||
|
||||
|
||||
async def _finalize_answer(
|
||||
llm: Callable[..., Any],
|
||||
convo: list[dict[str, Any]],
|
||||
executed: list[ToolCallRecord],
|
||||
steps: list[dict[str, Any]],
|
||||
iterations: int,
|
||||
stopped: str,
|
||||
) -> AgentResult:
|
||||
"""Guarantee a textual answer when the loop stopped before producing one.
|
||||
|
||||
Web research often exhausts the iteration budget while the model is still
|
||||
calling tools; returning ``content=""`` left the conversation with steps and
|
||||
sources but no answer. One final tool-less call asks the model to synthesize
|
||||
the gathered results, and a deterministic source list is used as a last
|
||||
resort (BUG-052).
|
||||
"""
|
||||
content = ""
|
||||
if executed:
|
||||
try:
|
||||
response = await llm(
|
||||
[*convo, {"role": "user", "content": _FINALIZE_INSTRUCTION}], []
|
||||
)
|
||||
content = (response.content or "").strip()
|
||||
except Exception as e:
|
||||
logger.warning(f"Agent final synthesis failed: {e}")
|
||||
if not content:
|
||||
content = _fallback_summary(executed)
|
||||
return AgentResult(
|
||||
content=content,
|
||||
messages=convo,
|
||||
tool_calls=executed,
|
||||
steps=steps,
|
||||
iterations=iterations,
|
||||
stopped=stopped,
|
||||
)
|
||||
|
||||
|
||||
def _execute_confirmed(
|
||||
ctx: ToolContext,
|
||||
confirm_pending: dict[str, Any],
|
||||
@@ -275,13 +347,14 @@ async def run_agent(
|
||||
for index, call in enumerate(response.tool_calls):
|
||||
if quota is not None and len(executed) >= quota:
|
||||
logger.warning(f"Agent reached the tool-call quota ({quota})")
|
||||
return AgentResult(
|
||||
content=response.content or "",
|
||||
messages=convo,
|
||||
tool_calls=executed,
|
||||
steps=steps,
|
||||
iterations=iteration,
|
||||
stopped=STOP_QUOTA_EXCEEDED,
|
||||
# Keep the conversation valid for the synthesis call: the
|
||||
# assistant message announced every tool call of the batch.
|
||||
for skipped in response.tool_calls[index:]:
|
||||
convo.append(_deferred_tool_message(
|
||||
skipped, "Not executed: the tool-call quota was reached."
|
||||
))
|
||||
return await _finalize_answer(
|
||||
llm, convo, executed, steps, iteration, STOP_QUOTA_EXCEEDED
|
||||
)
|
||||
try:
|
||||
result = call_tool(call.name, ctx, call.arguments)
|
||||
@@ -327,11 +400,6 @@ async def run_agent(
|
||||
})
|
||||
|
||||
logger.warning(f"Agent reached max iterations ({max_iterations})")
|
||||
return AgentResult(
|
||||
content="",
|
||||
messages=convo,
|
||||
tool_calls=executed,
|
||||
steps=steps,
|
||||
iterations=max_iterations,
|
||||
stopped=STOP_MAX_ITERATIONS,
|
||||
return await _finalize_answer(
|
||||
llm, convo, executed, steps, max_iterations, STOP_MAX_ITERATIONS
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user