fix(agent): stop reporting broken streams as output-length truncation (#36705)
A stream that drops mid-response after tokens are delivered (peer-closed connection, stale-stream reconnect) is converted into a synthetic finish_reason="length" stub. The conversation loop treated that network stall as a max-output-tokens truncation: when the dropped content was a tool call it retried exactly once, then hard-failed with "Response truncated due to output length limit" — even on large-output models that never hit any cap (e.g. Opus). - Tool-call truncation now retries up to 3 times (was 1) with a progressive max_tokens boost, and is stub-aware: a PARTIAL_STREAM_STUB_ID stall prints "Stream interrupted mid tool-call — retrying (n/3)" instead of the false "model hit max output tokens", and the give-up message distinguishes a network drop from a real truncation. - Length-continuation retries preserve the original request's output cap as a floor, so a high provider/model default isn't silently downshifted to 8K/12K on retry. - Added _requested_output_cap_from_api_kwargs() helper. Tests: stub-stall mid-tool-call recovery within 3 retries; continuation preserves a large provider-default output cap. Fixes #26425. Salvages the substance of #26427 (cap floor) and #9525 (retry bump), adapted to the post-refactor conversation_loop.py which handles all three api_modes uniformly. Co-authored-by: LeonSGP43 <cine.dreamer.one@gmail.com> Co-authored-by: ygd58 <ygd58@users.noreply.github.com>
This commit is contained in:
co-authored by
LeonSGP43
ygd58
parent
b571ec298d
commit
023149f665
+54
-10
@@ -1739,20 +1739,52 @@ def run_conversation(
|
||||
if agent.api_mode in {"chat_completions", "bedrock_converse", "anthropic_messages"}:
|
||||
assistant_message = _trunc_msg
|
||||
if assistant_message is not None and _trunc_has_tool_calls:
|
||||
if truncated_tool_call_retries < 1:
|
||||
_is_stub_stall = (
|
||||
getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID
|
||||
)
|
||||
if truncated_tool_call_retries < 3:
|
||||
truncated_tool_call_retries += 1
|
||||
agent._buffer_vprint(
|
||||
f"⚠️ Truncated tool call detected — retrying API call..."
|
||||
)
|
||||
if _is_stub_stall:
|
||||
# The stream broke mid tool-call (network /
|
||||
# peer-closed connection), not a real output
|
||||
# cap — say so instead of "max output tokens".
|
||||
agent._buffer_vprint(
|
||||
f"⚠️ Stream interrupted mid tool-call — "
|
||||
f"retrying ({truncated_tool_call_retries}/3)..."
|
||||
)
|
||||
else:
|
||||
agent._buffer_vprint(
|
||||
f"⚠️ Truncated tool call detected — "
|
||||
f"retrying API call "
|
||||
f"({truncated_tool_call_retries}/3)..."
|
||||
)
|
||||
# Boost max_tokens on each retry so the model has
|
||||
# more room to complete the tool-call JSON. A
|
||||
# network stall doesn't need a bigger budget, but
|
||||
# a genuine output-cap truncation does, and the
|
||||
# boost is harmless for the stall case.
|
||||
_tc_boost_base = agent.max_tokens if agent.max_tokens else 4096
|
||||
_tc_boost = _tc_boost_base * (truncated_tool_call_retries + 1)
|
||||
_tc_requested_cap = agent._requested_output_cap_from_api_kwargs(api_kwargs)
|
||||
if _tc_requested_cap is not None:
|
||||
_tc_boost = max(_tc_boost, _tc_requested_cap)
|
||||
_tc_boost_cap = max(32768, _tc_requested_cap or 0)
|
||||
agent._ephemeral_max_output_tokens = min(_tc_boost, _tc_boost_cap)
|
||||
# Don't append the broken response to messages;
|
||||
# just re-run the same API call from the current
|
||||
# message state, giving the model another chance.
|
||||
continue
|
||||
agent._flush_status_buffer()
|
||||
agent._vprint(
|
||||
f"{agent.log_prefix}⚠️ Truncated tool call response detected again — refusing to execute incomplete tool arguments.",
|
||||
force=True,
|
||||
)
|
||||
if _is_stub_stall:
|
||||
agent._vprint(
|
||||
f"{agent.log_prefix}⚠️ Stream kept dropping mid tool-call after 3 retries — the action was not executed.",
|
||||
force=True,
|
||||
)
|
||||
else:
|
||||
agent._vprint(
|
||||
f"{agent.log_prefix}⚠️ Truncated tool call response detected again — refusing to execute incomplete tool arguments.",
|
||||
force=True,
|
||||
)
|
||||
agent._cleanup_task_resources(effective_task_id)
|
||||
agent._persist_session(messages, conversation_history)
|
||||
return {
|
||||
@@ -1761,7 +1793,12 @@ def run_conversation(
|
||||
"api_calls": api_call_count,
|
||||
"completed": False,
|
||||
"partial": True,
|
||||
"error": "Response truncated due to output length limit",
|
||||
"error": (
|
||||
"Stream repeatedly dropped mid tool-call (network); "
|
||||
"the tool was not executed"
|
||||
if _is_stub_stall
|
||||
else "Response truncated due to output length limit"
|
||||
),
|
||||
}
|
||||
|
||||
# If we have prior messages, roll back to last complete state
|
||||
@@ -3412,9 +3449,16 @@ def run_conversation(
|
||||
# Progressively boost the output token budget on each retry.
|
||||
# Retry 1 → 2× base, retry 2 → 3× base, capped at 32 768.
|
||||
# Applies to all providers via _ephemeral_max_output_tokens.
|
||||
# If the original request already used a larger provider/model
|
||||
# default budget, keep that floor so continuation retries do
|
||||
# not accidentally downshift to a much smaller cap.
|
||||
_boost_base = agent.max_tokens if agent.max_tokens else 4096
|
||||
_boost = _boost_base * (length_continue_retries + 1)
|
||||
agent._ephemeral_max_output_tokens = min(_boost, 32768)
|
||||
_requested_cap = agent._requested_output_cap_from_api_kwargs(api_kwargs)
|
||||
if _requested_cap is not None:
|
||||
_boost = max(_boost, _requested_cap)
|
||||
_boost_cap = max(32768, _requested_cap or 0)
|
||||
agent._ephemeral_max_output_tokens = min(_boost, _boost_cap)
|
||||
continue
|
||||
|
||||
# Guard: if all retries exhausted without a successful response
|
||||
|
||||
Reference in New Issue
Block a user