fix agent inference timeout on slow local models

This commit is contained in:
Hamza Ayed
2026-10-04 15:04:35 +03:00
parent 0f045b27d4
commit 4f6353cf88
3 changed files with 23 additions and 3 deletions
+3 -1
View File
@@ -2207,7 +2207,9 @@ async def _execute_agent(
read_only_tool_results: dict[str, tuple[Any, list[str]]] = {}
remaining_tool_calls = MAX_AGENT_TOOL_CALLS - len(steps)
for call_index in range(remaining_tool_calls + 1):
completion = await get_completion(current_payload)
# Agent reasoning and tool-selection can take several minutes on a small
# local model. Keep this aligned with the SSE client/request limit below.
completion = await get_completion(current_payload, timeout_seconds=600.0)
final_message = completion["choices"][0].get("message", {})
tool_calls = final_message.get("tool_calls") or []
if not tool_calls: