fix(errors): stop empty-content stream stubs from poisoning the transcript

A stream that dies after delivering deltas but with 0 recovered chars (and
no captured tool call) produced a content-less assistant stub. That empty
non-final message is invalid for the Anthropic schema, so every later request
400s with 'all messages must have non-empty content' (INVALID_REQUEST_BODY).
On a large session the 400 was misclassified as context overflow, dropping the
loop into a compression spiral that ends in 'Cannot compress further' — a
misleading context-size error on a session nowhere near its limit.

Root cause: substitute a minimal '[response interrupted]' placeholder so the
stub is never empty.

Misclassification: add _INVALID_MESSAGE_BODY_PATTERNS (checked before the
context-overflow heuristic) to classify these as non-retryable format_error,
and teach the body extractors the litellm/Bedrock errorMessage/errorCode/
errorArgs shape so descriptive proxy errors are not mistaken for generic ones.

Adds RED-verified regression tests for the stub path and three classifier tests
(including a guard that real overflows still compress).
This commit is contained in:
Aaron Weiker 2026-07-25 08:50:39 -07:00 committed by Teknium
parent 202140db53
commit 207a6c969d
4 changed files with 237 additions and 1 deletions

View File

@ -3944,6 +3944,21 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
result["error"],
)
_stub_finish_reason = FINISH_REASON_LENGTH
# Never persist a content-less, tool-call-less assistant stub.
# When a stream dies before any text arrives (and no partial tool
# calls were captured), _partial_text is None → the stub becomes an
# empty assistant message. The Anthropic message schema (and the
# litellm/Bedrock proxies in front of it) reject any request whose
# transcript contains an empty non-final message:
# "all messages must have non-empty content except for the
# optional final assistant message" (400 INVALID_REQUEST_BODY)
# Once such a stub lands mid-transcript, EVERY subsequent turn 400s
# until it scrolls out of context — and the 400 gets misread as a
# context-overflow "Cannot compress further" loop. Substitute a
# minimal, honest placeholder so the message is always API-valid
# and the continuation prompt still reads as an interrupted turn.
if not _partial_text:
_partial_text = "[response interrupted]"
_stub_msg = SimpleNamespace(
role="assistant", content=_partial_text, tool_calls=None,
reasoning_content=None,

View File

@ -339,6 +339,30 @@ _MODEL_NOT_FOUND_PATTERNS = [
"no endpoints found that support tool use",
]
# Malformed-message-array 400s. Deterministic request-shape rejections that
# describe the *transcript* being invalid, not a parameter. The canonical
# case: a stream dies mid-response and Hermes persists a content-less
# assistant stub; on the next turn the Anthropic message schema (and the
# litellm/Bedrock proxies in front of it) reject the whole request with
# "all messages must have non-empty content except for the optional final
# assistant message" / errorCode INVALID_REQUEST_BODY
# These are NOT context overflow — the input may be tiny — but a large
# session used to mis-route them into the compression loop via the generic
# "400 + large session" heuristic below, ending in "Cannot compress further"
# every retry (the input is unchanged, so compression cannot help). Match
# the message-shape signals explicitly and fail fast as a format_error so the
# loop stops looping. The empty-stub creation is the root cause (fixed in
# chat_completion_helpers); this pattern stops the misclassification symptom
# for transcripts that already contain a poisoned stub.
_INVALID_MESSAGE_BODY_PATTERNS = [
"must have non-empty content",
"messages must have non-empty",
"invalid_request_body",
"text content blocks must be non-empty",
"content field is required",
"messages: at least one message is required",
]
# Request-validation patterns — the request is malformed and will fail
# identically on every retry. Some OpenAI-compatible gateways (notably
# codex.nekos.me) return these as 5xx instead of the standard 4xx, which
@ -1289,6 +1313,25 @@ def _classify_400(
should_fallback=True,
)
# Malformed message array (empty-content assistant stub, etc.). Must be
# checked BEFORE context_overflow: the input can be tiny, so the generic
# "400 + large session" heuristic would otherwise mis-route it into the
# compression loop and thrash until "Cannot compress further" on every
# retry (the request is unchanged, so compression cannot fix it). This is
# a deterministic request-shape rejection — fail fast as a non-retryable
# format_error and fall back. Checked against the message text AND the
# structured error code, since proxies (litellm/Bedrock) surface the
# signal in errorCode=INVALID_REQUEST_BODY.
if (
any(p in error_msg for p in _INVALID_MESSAGE_BODY_PATTERNS)
or error_code_lower == "invalid_request_body"
):
return result_fn(
FailoverReason.format_error,
retryable=False,
should_fallback=True,
)
# Empty-provider-response advisories must not enter compression. They
# often mention "max_tokens" as a possible cause and used to match the
# bare overflow pattern, then thrash compress until "Cannot compress
@ -1349,6 +1392,18 @@ def _classify_400(
# Responses API (and some providers) use flat body: {"message": "..."}
if not err_body_msg:
err_body_msg = str(body.get("message") or "").strip().lower()
# litellm / Bedrock proxies use a custom shape: {"errorMessage": "...",
# "errorCode": "...", "errorArgs": {"reason": "..."}}. Without these
# keys err_body_msg stays "" and a long, descriptive rejection is
# wrongly treated as a "generic" (bare) error below, which — on a
# large session — mis-routes into the compression loop. Recognize
# them so the is_generic heuristic sees the real message length.
if not err_body_msg:
err_body_msg = str(body.get("errorMessage") or "").strip().lower()
if not err_body_msg:
_args = body.get("errorArgs")
if isinstance(_args, dict):
err_body_msg = str(_args.get("reason") or "").strip().lower()
is_generic = len(err_body_msg) < 30 or err_body_msg in {"error", ""}
# Absolute token/message-count thresholds are only a proxy for smaller
# context windows. Large-context sessions can have many messages while
@ -1647,7 +1702,7 @@ def _extract_error_code(body: dict) -> str:
return nested_code
# Top-level code
code = body.get("code") or body.get("error_code") or ""
code = body.get("code") or body.get("error_code") or body.get("errorCode") or ""
if isinstance(code, (str, int)):
text = str(code).strip()
if text and text != "400":
@ -1667,6 +1722,16 @@ def _extract_message(error: Exception, body: dict) -> str:
msg = body.get("message", "")
if isinstance(msg, str) and msg.strip():
return msg.strip()[:500]
# litellm / Bedrock proxy shape: {"errorMessage": "...",
# "errorArgs": {"reason": "..."}}.
msg = body.get("errorMessage", "")
if isinstance(msg, str) and msg.strip():
return msg.strip()[:500]
args = body.get("errorArgs")
if isinstance(args, dict):
reason = args.get("reason", "")
if isinstance(reason, str) and reason.strip():
return reason.strip()[:500]
# Fallback to str(error)
return str(error)[:500]

View File

@ -1298,6 +1298,84 @@ class TestClassifyApiError:
result = classify_api_error(e, approx_tokens=100000, context_length=200000)
assert result.reason == FailoverReason.context_overflow
def test_400_empty_content_message_not_context_overflow(self):
"""Anthropic 'non-empty content' 400 → format_error, NOT compression.
Regression for the empty-assistant-stub bug: a stream dies with 0
recovered chars, an empty assistant message is persisted, and every
subsequent request 400s with 'all messages must have non-empty
content'. On a large session the generic '400 + large session'
heuristic used to mis-route this into the compression loop, ending in
'Cannot compress further' on every retry (compression can't fix a
malformed transcript). It must classify as a non-retryable
format_error so the loop stops looping.
"""
msg = ("all messages must have non-empty content except for the "
"optional final assistant message")
e = MockAPIError(
msg,
status_code=400,
body={"error": {"message": msg, "type": "invalid_request_error"}},
)
# Large session (many messages / tokens) to prove the overflow
# heuristic does NOT capture it.
result = classify_api_error(
e, approx_tokens=66000, context_length=200000, num_messages=219,
)
assert result.reason == FailoverReason.format_error
assert result.retryable is False
assert result.should_compress is not True
def test_400_litellm_invalid_request_body_shape(self):
"""litellm/Bedrock proxy shape (errorMessage/errorCode) → format_error.
The proxy in front of Anthropic surfaces the empty-content rejection
as {"errorMessage": "...non-empty content...", "errorCode":
"INVALID_REQUEST_BODY", "errorArgs": {"reason": "..."}}. Those keys
are not the standard error.message / message, so err_body_msg used to
come back empty is_generic=True mis-routed into compression on a
large session. Both the message pattern and the errorCode must be
recognized.
"""
proxy_msg = ("The provided request body is invalid: claude "
"messages.208: all messages must have non-empty content "
"except for the optional final assistant message")
e = MockAPIError(
proxy_msg,
status_code=400,
body={
"errorMessage": proxy_msg,
"errorCode": "INVALID_REQUEST_BODY",
"statusCode": 400,
"errorArgs": {"reason": "claude messages.208: ..."},
},
)
result = classify_api_error(
e, approx_tokens=66000, context_length=200000, num_messages=219,
)
assert result.reason == FailoverReason.format_error
assert result.retryable is False
assert result.should_compress is not True
def test_400_real_context_overflow_still_compresses(self):
"""Guard: the new empty-content guard must NOT swallow real overflows.
A genuine 'maximum context length' 400 must still route into
compression the fix is surgical, not a blanket 400format_error.
"""
msg = ("This model's maximum context length is 200000 tokens. "
"However, your messages resulted in 250000 tokens.")
e = MockAPIError(
msg,
status_code=400,
body={"error": {"message": msg, "type": "invalid_request_error"}},
)
result = classify_api_error(
e, approx_tokens=250000, context_length=200000, num_messages=219,
)
assert result.reason == FailoverReason.context_overflow
assert result.should_compress is True
# ── Peer closed + large session ──
def test_peer_closed_large_session(self):

View File

@ -1583,6 +1583,84 @@ class TestPartialToolCallWarning:
f"Unexpected warning on text-only partial stream: {content!r}"
)
@patch("run_agent.AIAgent._create_request_openai_client")
@patch("run_agent.AIAgent._close_request_openai_client")
def test_empty_partial_stream_never_yields_empty_content(
self, mock_close, mock_create,
):
"""Stream dies with 0 recovered chars and no tool call → stub content
must be a non-empty placeholder, never None/''.
Root-cause regression for the empty-assistant-stub bug: a stream
delivered some deltas, then died before any text landed in
``_current_streamed_assistant_text`` (and no tool call was captured).
The stub was built with ``content=None`` an empty assistant message
got persisted mid-transcript. The Anthropic message schema (and the
litellm/Bedrock proxies in front of it) then reject EVERY subsequent
request:
"all messages must have non-empty content except for the optional
final assistant message" (400 INVALID_REQUEST_BODY)
which the loop misreads as a context-overflow "Cannot compress
further" spiral. The stub must carry a minimal placeholder so the
message is always API-valid.
"""
from run_agent import AIAgent
class _StallError(RuntimeError):
pass
def _stalling_stream():
# A real content delta fires (deltas_were_sent=True → the
# post-delivery stub path runs), but the recovered-text
# accumulator is empty by the time the stub is built (cleared on
# reset in prod), and no tool call was captured — the exact
# "0 chars recovered, no tool call" production condition.
yield _make_stream_chunk(content="partial token")
raise _StallError("simulated upstream stall after a delta")
mock_client = MagicMock()
mock_client.chat.completions.create.side_effect = (
lambda *a, **kw: _stalling_stream()
)
mock_create.return_value = mock_client
agent = AIAgent(
api_key="test-key",
base_url="https://openrouter.ai/api/v1",
model="test/model",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
agent.api_mode = "chat_completions"
agent._interrupt_requested = False
agent._fire_stream_delta = lambda text: None
# Empty recovered text — this is what produced the empty stub in prod
# even though a delta was delivered to the platform above.
agent._current_streamed_assistant_text = ""
import os as _os
_prev = _os.environ.get("HERMES_STREAM_RETRIES")
_os.environ["HERMES_STREAM_RETRIES"] = "0"
try:
response = agent._interruptible_streaming_api_call({})
finally:
if _prev is None:
_os.environ.pop("HERMES_STREAM_RETRIES", None)
else:
_os.environ["HERMES_STREAM_RETRIES"] = _prev
content = response.choices[0].message.content
assert content, (
f"Empty-partial-stream stub must NOT have empty/None content "
f"(it poisons the transcript and 400s every later request). "
f"Got content={content!r}"
)
assert content.strip() != "", (
f"Stub content is whitespace-only, still API-invalid: {content!r}"
)
assert response.choices[0].message.tool_calls is None
class TestSilentRetryMidToolCall:
"""Regression: when the stream dies mid tool-call JSON after text was