fix(ai): garantir une reponse finale quand la boucle d'agent epuise son budget (BUG-052)
CI / lint (push) Successful in 1m31s
CI / security (push) Successful in 1m2s
CI / test (push) Successful in 3m25s
CI / build (push) Successful in 56s
CI / e2e (push) Successful in 10m41s

This commit is contained in:
2026-09-16 23:28:56 -04:00
parent 62cff271d8
commit e3434d19ea
13 changed files with 155 additions and 30 deletions
+85 -17
View File
@@ -40,6 +40,15 @@ MAX_TOOL_RESULT_CHARS = 100_000
# Quota: maximum tool calls executed per agent run (``BOOKSLM_MAX_TOOL_CALLS``).
DEFAULT_MAX_TOOL_CALLS = int(os.environ.get("BOOKSLM_MAX_TOOL_CALLS", "25"))
# Sent as a last user turn when the loop stopped before the model produced an
# answer (iteration/quota budget exhausted while it was still calling tools).
_FINALIZE_INSTRUCTION = (
"N'appelle plus aucun outil. Réponds maintenant directement à l'utilisateur, "
"en français, à partir des informations déjà recueillies ci-dessus. "
"Structure la réponse en Markdown, cite les liens sources utiles, et si les "
"informations sont insuffisantes, dis-le explicitement."
)
# Stopping reasons
STOP_DONE = "done"
STOP_MAX_ITERATIONS = "max_iterations"
@@ -109,8 +118,8 @@ def _assistant_tool_message(content: str | None, tool_calls: list[Any]) -> dict[
}
def _deferred_tool_message(call: Any) -> dict[str, Any]:
"""Answer a tool call that was not reached because the run paused.
def _deferred_tool_message(call: Any, reason: str | None = None) -> dict[str, Any]:
"""Answer a tool call that was not reached because the run stopped early.
A single LLM response may carry several tool calls. When one of them is
mutating and pauses the run for confirmation, the assistant message already
@@ -125,7 +134,7 @@ def _deferred_tool_message(call: Any) -> dict[str, Any]:
"name": call.name,
"content": json.dumps({
"status": "deferred",
"reason": (
"reason": reason or (
"Not executed: the run paused to confirm an earlier tool call. "
"Re-issue this call if it is still needed."
),
@@ -133,6 +142,69 @@ def _deferred_tool_message(call: Any) -> dict[str, Any]:
}
def _fallback_summary(executed: list[ToolCallRecord]) -> str:
"""Deterministic non-empty answer built from the gathered tool results.
Used only if the final synthesis call fails or returns nothing, so a turn
never ends on an empty message (BUG-052).
"""
lines: list[str] = []
for record in executed:
data = record.result
if not isinstance(data, dict):
continue
for item in (data.get("results") or [])[:5]:
if not isinstance(item, dict):
continue
title = item.get("title") or item.get("url") or ""
url = item.get("url") or ""
lines.append(f"- [{title}]({url})" if url else f"- {title}")
if data.get("url") and data.get("text"):
title = data.get("title") or data["url"]
lines.append(f"- [{title}]({data['url']})")
if not lines:
return "Je n'ai pas pu produire de réponse à partir des résultats obtenus."
unique = list(dict.fromkeys(lines))
return "Voici les sources pertinentes trouvées :\n" + "\n".join(unique)
async def _finalize_answer(
llm: Callable[..., Any],
convo: list[dict[str, Any]],
executed: list[ToolCallRecord],
steps: list[dict[str, Any]],
iterations: int,
stopped: str,
) -> AgentResult:
"""Guarantee a textual answer when the loop stopped before producing one.
Web research often exhausts the iteration budget while the model is still
calling tools; returning ``content=""`` left the conversation with steps and
sources but no answer. One final tool-less call asks the model to synthesize
the gathered results, and a deterministic source list is used as a last
resort (BUG-052).
"""
content = ""
if executed:
try:
response = await llm(
[*convo, {"role": "user", "content": _FINALIZE_INSTRUCTION}], []
)
content = (response.content or "").strip()
except Exception as e:
logger.warning(f"Agent final synthesis failed: {e}")
if not content:
content = _fallback_summary(executed)
return AgentResult(
content=content,
messages=convo,
tool_calls=executed,
steps=steps,
iterations=iterations,
stopped=stopped,
)
def _execute_confirmed(
ctx: ToolContext,
confirm_pending: dict[str, Any],
@@ -275,13 +347,14 @@ async def run_agent(
for index, call in enumerate(response.tool_calls):
if quota is not None and len(executed) >= quota:
logger.warning(f"Agent reached the tool-call quota ({quota})")
return AgentResult(
content=response.content or "",
messages=convo,
tool_calls=executed,
steps=steps,
iterations=iteration,
stopped=STOP_QUOTA_EXCEEDED,
# Keep the conversation valid for the synthesis call: the
# assistant message announced every tool call of the batch.
for skipped in response.tool_calls[index:]:
convo.append(_deferred_tool_message(
skipped, "Not executed: the tool-call quota was reached."
))
return await _finalize_answer(
llm, convo, executed, steps, iteration, STOP_QUOTA_EXCEEDED
)
try:
result = call_tool(call.name, ctx, call.arguments)
@@ -327,11 +400,6 @@ async def run_agent(
})
logger.warning(f"Agent reached max iterations ({max_iterations})")
return AgentResult(
content="",
messages=convo,
tool_calls=executed,
steps=steps,
iterations=max_iterations,
stopped=STOP_MAX_ITERATIONS,
return await _finalize_answer(
llm, convo, executed, steps, max_iterations, STOP_MAX_ITERATIONS
)