From e0ad7eef0208e4ed1a841af5787b1717a4d38b16 Mon Sep 17 00:00:00 2001 From: tmatup <51425734+tmatup@users.noreply.github.com> Date: Mon, 28 Sep 2026 19:52:19 +0000 Subject: [PATCH] fix(codex): record provider error notifications on the turn MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex retries a stalled or failed provider stream itself and reports each attempt only through its app-server `error` notification (`willRetry`). `_CodexTurnState.dispatch` dropped that method, so a provider stall left no trace in task.json or task.log: adhoc-2026-09-28_16-14-30 v2 skill-flow-slack-channel-description lost 1,125 s of its 1,200 s budget to two stalls on a ~12.5K-token request before the first tool call, and the record shows only two long "thinking" messages. - dispatch routes `error` to a new `on_error`, which logs a WARNING and appends a `ProviderError` (at, message, kind, http_status, will_retry, details) to the turn state. - `AgentEndEvent` / `TurnRecord` carry `provider_errors`; EventCollector passes it through, so crashed partial turns keep it too. - REPORT_SCHEMA documents the field; golden snapshots gain the empty list. Closes part 1 of #151 (part 2, per-request TTFT timing, stays open). 🤖 Generated with Claude Code Co-Authored-By: [Claude](mailto:noreply@anthropic.com) Claude-Session: https://claude.ai/code/session_014Tw2Pqkugyik4uVJHo2Ref --- docs/REPORT_SCHEMA.md | 5 +- src/coder_eval/agents/codex_agent.py | 43 ++++++++++ src/coder_eval/models/__init__.py | 2 + src/coder_eval/models/results.py | 9 ++ src/coder_eval/models/telemetry.py | 18 ++++ src/coder_eval/streaming/collector.py | 1 + src/coder_eval/streaming/events.py | 2 + .../antigravity_a_single_text_turn.json | 1 + .../antigravity_b_tool_call_resolved.json | 1 + ...y_c_thinking_and_tool_same_generation.json | 1 + .../expected/antigravity_d_orphaned_tool.json | 1 + .../antigravity_e_multi_generation.json | 1 + .../expected/claude_a_single_text_turn.json | 1 + .../expected/claude_b_tool_use_result.json | 1 + .../claude_c_multi_emission_delta.json | 1 + .../expected/claude_d_subagent_terminal.json | 1 + .../claude_e_model_usage_and_backfill.json | 1 + .../expected/claude_f_orphaned_tool.json | 1 + .../claude_g_crash_format_placeholder.json | 1 + .../claude_h1_timeout_process_error.json | 1 + .../claude_h2_process_error_crash.json | 1 + .../claude_i_in_loop_deadline_break.json | 1 + .../expected/codex_a_agent_message_only.json | 1 + .../expected/codex_b_command_execution.json | 1 + .../codex_c_reasoning_placeholder.json | 1 + .../codex_d_cross_flush_is_error.json | 1 + .../expected/codex_e_orphan_tool.json | 1 + .../expected/codex_f_collab_fallback.json | 1 + .../expected/codex_g_items_rebuild.json | 1 + .../codex_h_no_turn_completed_crash.json | 1 + .../expected/opencode_a_single_text_turn.json | 1 + .../opencode_b_tool_call_resolved.json | 1 + .../opencode_c_multi_step_tiling.json | 1 + .../expected/opencode_d_orphaned_tool.json | 1 + .../opencode_e_error_after_generation.json | 1 + .../expected/pi_a_single_text_turn.json | 1 + .../expected/pi_b_tool_call_resolved.json | 1 + .../expected/pi_c_multi_turn_tiling.json | 1 + .../expected/pi_d_orphaned_tool.json | 1 + .../expected/pi_e_error_after_generation.json | 1 + .../expected/pi_f_duplicate_turn_end.json | 1 + tests/test_codex_agent.py | 85 +++++++++++++++++++ 42 files changed, 198 insertions(+), 1 deletion(-) diff --git a/docs/REPORT_SCHEMA.md b/docs/REPORT_SCHEMA.md index f7002e3ec..f0fb59e16 100644 --- a/docs/REPORT_SCHEMA.md +++ b/docs/REPORT_SCHEMA.md @@ -210,7 +210,10 @@ token/cost budget breach, which fires only after every criterion is already scor cache buckets, captured proxy-side on the LiteLLM open-weight backend and rendered by the evalboard as a per-call table; empty on every other backend), `num_turns`, `max_turns_exhausted`, -`result_summary` (`{is_error, subtype, stop_reason, result}`), `crashed`, +`result_summary` (`{is_error, subtype, stop_reason, result}`), `provider_errors` +(`list[ProviderError]` — `{at, message, kind, http_status, will_retry, details}`, one +row per model-provider error the agent's own client reported, including errors it +retried itself; filled by the codex adapter, empty on other backends), `crashed`, `crash_reason`. > **Token invariant.** Summing the four token buckets across `messages` diff --git a/src/coder_eval/agents/codex_agent.py b/src/coder_eval/agents/codex_agent.py index fbb150e98..bb3746a1b 100644 --- a/src/coder_eval/agents/codex_agent.py +++ b/src/coder_eval/agents/codex_agent.py @@ -11,6 +11,7 @@ import time from collections.abc import Callable from datetime import datetime +from enum import Enum from pathlib import Path from typing import Any, ClassVar, NamedTuple from urllib.parse import urlparse @@ -33,6 +34,7 @@ CommandTelemetry, ContentBlock, DirectRoute, + ProviderError, SystemPromptSemantics, TokenUsage, TranscriptMessage, @@ -274,6 +276,34 @@ def _message_uncached_input(m: AssistantMessage) -> int: _ZSH_PROFILE_NAMES = frozenset({".zshenv", ".zprofile", ".zshrc"}) +def _provider_error_from(payload: Any) -> ProviderError: + """Map a codex ``ErrorNotification`` payload onto a ``ProviderError``. + + ``codexErrorInfo`` is either a bare category (enum) or a one-key object whose + key is the category and whose value may carry ``httpStatusCode``. + """ + error = getattr(payload, "error", None) + kind: str | None = None + http_status: int | None = None + info = getattr(getattr(error, "codex_error_info", None), "root", None) + if isinstance(info, Enum): + kind = str(info.value) + elif hasattr(info, "model_dump"): + dumped = info.model_dump(by_alias=True, mode="json") + if len(dumped) == 1: + kind, body = next(iter(dumped.items())) + if isinstance(body, dict) and isinstance(body.get("httpStatusCode"), int): + http_status = body["httpStatusCode"] + return ProviderError( + at=datetime.now(), + message=str(getattr(error, "message", "") or ""), + kind=kind, + http_status=http_status, + will_retry=bool(getattr(payload, "will_retry", False)), + details=getattr(error, "additional_details", None), + ) + + def _get_item_root(notification: Any) -> Any: """Extract the typed item root from a Codex SDK notification. @@ -338,6 +368,7 @@ def __init__( # Live pump scratch (set during streaming). self.turn_result: Any = None self.latest_token_usage: Any = None + self.provider_errors: list[ProviderError] = [] self.agent_message_chunks: list[str] = [] # Sequence per executable item, assigned at item/started and reused at # item/completed via this id->seq map. @@ -546,8 +577,19 @@ def dispatch(self, notification: Any) -> bool: self.on_token_usage_updated(notification) elif method == "turn/completed": return self.on_turn_completed(notification) + elif method == "error": + self.on_error(notification) return False + def on_error(self, notification: Any) -> None: + """Record a provider error; codex retries a stalled stream silently otherwise.""" + err = _provider_error_from(notification.payload) + self.provider_errors.append(err) + self._agent._log.warning( + f"provider error (will_retry={err.will_retry}, kind={err.kind}, " + + f"http_status={err.http_status}): {err.message}" + ) + def on_item_started(self, notification: Any) -> None: """Emit ToolStartEvent + record the tool_use block for every tool-like item.""" root = _get_item_root(notification) @@ -765,6 +807,7 @@ def finalize(self, status: AgentEndStatus, *, crashed: bool = False, crash_reaso assistant_turn_count=1, messages=self.messages, num_turns=max(self.api_calls, 1), + provider_errors=list(self.provider_errors), crashed=crashed, crash_reason=crash_reason, max_turns_exhausted=status is AgentEndStatus.MAX_TURNS_EXHAUSTED, diff --git a/src/coder_eval/models/__init__.py b/src/coder_eval/models/__init__.py index ff633bbe4..0ee1538b8 100644 --- a/src/coder_eval/models/__init__.py +++ b/src/coder_eval/models/__init__.py @@ -228,6 +228,7 @@ CommandTelemetry, ContentBlock, ProviderCallCost, + ProviderError, ReconciliationMessage, SlowestCommandInfo, TokenUsage, @@ -349,6 +350,7 @@ "CommandStatistics", "ContentBlock", "ProviderCallCost", + "ProviderError", "ReconciliationMessage", "SlowestCommandInfo", "TokenUsage", diff --git a/src/coder_eval/models/results.py b/src/coder_eval/models/results.py index 867c1c249..f778130fa 100644 --- a/src/coder_eval/models/results.py +++ b/src/coder_eval/models/results.py @@ -26,6 +26,7 @@ CommandStatistics, CommandTelemetry, ProviderCallCost, + ProviderError, TokenUsage, TranscriptMessage, ) @@ -417,6 +418,14 @@ class TurnRecord(BaseModel): "these calls' cost_usd (the real bill), not the static rate-card estimate." ), ) + provider_errors: list[ProviderError] = Field( + default_factory=list, + description=( + "Model-provider errors the agent's own client reported during this turn, in arrival " + "order, including ones it retried itself (will_retry=True). Filled by the codex " + "adapter from its `error` notification; empty on backends that do not surface them." + ), + ) crashed: bool = Field( default=False, description=( diff --git a/src/coder_eval/models/telemetry.py b/src/coder_eval/models/telemetry.py index 6015c9563..210bbd308 100644 --- a/src/coder_eval/models/telemetry.py +++ b/src/coder_eval/models/telemetry.py @@ -134,6 +134,24 @@ class ProviderCallCost(BaseModel): output_tokens: int | None = Field(default=None, description="Completion tokens for this call.") +class ProviderError(BaseModel): + """One model-provider error the agent's own client reported mid-turn. + + Codex retries a stalled or failed provider stream internally and says so only + through its ``error`` notification (``willRetry``). Without this record a + 9-minute provider stall reads as pure model latency. + """ + + at: datetime = Field(description="When the harness received the error notification.") + message: str = Field(description="The client's error message.") + kind: str | None = Field( + default=None, description="Client error category (codex `codexErrorInfo`), e.g. responseStreamDisconnected." + ) + http_status: int | None = Field(default=None, description="Upstream HTTP status, when the client forwarded one.") + will_retry: bool = Field(description="True when the client retries the request itself after this error.") + details: str | None = Field(default=None, description="Extra detail the client attached, if any.") + + class ContentBlock(BaseModel): """One content block within a message, in emission order. diff --git a/src/coder_eval/streaming/collector.py b/src/coder_eval/streaming/collector.py index 33c9d1ba0..e507172d8 100644 --- a/src/coder_eval/streaming/collector.py +++ b/src/coder_eval/streaming/collector.py @@ -224,6 +224,7 @@ def build_turn_record(self) -> TurnRecord: num_turns=end.num_turns, max_turns_exhausted=end.max_turns_exhausted, result_summary=end.result_summary, + provider_errors=end.provider_errors, crashed=end.crashed, crash_reason=end.crash_reason, harness_startup_ms=startup_ms, diff --git a/src/coder_eval/streaming/events.py b/src/coder_eval/streaming/events.py index 505a4cddb..2ef0b8ef0 100644 --- a/src/coder_eval/streaming/events.py +++ b/src/coder_eval/streaming/events.py @@ -32,6 +32,7 @@ from coder_eval.models import ( CommandTelemetry, + ProviderError, ResultSummary, TokenUsage, TranscriptMessage, @@ -169,6 +170,7 @@ class AgentEndEvent(StreamEvent): num_turns: int | None = None max_turns_exhausted: bool = False result_summary: ResultSummary | None = None + provider_errors: list[ProviderError] = Field(default_factory=list) crashed: bool = False crash_reason: str | None = None duration_seconds: float = 0.0 diff --git a/tests/_fixtures/golden_streams/expected/antigravity_a_single_text_turn.json b/tests/_fixtures/golden_streams/expected/antigravity_a_single_text_turn.json index b29372c32..d727e3637 100644 --- a/tests/_fixtures/golden_streams/expected/antigravity_a_single_text_turn.json +++ b/tests/_fixtures/golden_streams/expected/antigravity_a_single_text_turn.json @@ -40,6 +40,7 @@ ], "model_used": "gemini-3.5-flash", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/antigravity_b_tool_call_resolved.json b/tests/_fixtures/golden_streams/expected/antigravity_b_tool_call_resolved.json index e346caafb..8f7e01a36 100644 --- a/tests/_fixtures/golden_streams/expected/antigravity_b_tool_call_resolved.json +++ b/tests/_fixtures/golden_streams/expected/antigravity_b_tool_call_resolved.json @@ -71,6 +71,7 @@ ], "model_used": "gemini-3.5-flash", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/antigravity_c_thinking_and_tool_same_generation.json b/tests/_fixtures/golden_streams/expected/antigravity_c_thinking_and_tool_same_generation.json index 89bcd10fb..6e4a4ce81 100644 --- a/tests/_fixtures/golden_streams/expected/antigravity_c_thinking_and_tool_same_generation.json +++ b/tests/_fixtures/golden_streams/expected/antigravity_c_thinking_and_tool_same_generation.json @@ -80,6 +80,7 @@ ], "model_used": "gemini-3.5-flash", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/antigravity_d_orphaned_tool.json b/tests/_fixtures/golden_streams/expected/antigravity_d_orphaned_tool.json index f5bcc178f..dd01381cb 100644 --- a/tests/_fixtures/golden_streams/expected/antigravity_d_orphaned_tool.json +++ b/tests/_fixtures/golden_streams/expected/antigravity_d_orphaned_tool.json @@ -60,6 +60,7 @@ ], "model_used": "gemini-3.5-flash", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/antigravity_e_multi_generation.json b/tests/_fixtures/golden_streams/expected/antigravity_e_multi_generation.json index e9ac990bf..2d68dddf7 100644 --- a/tests/_fixtures/golden_streams/expected/antigravity_e_multi_generation.json +++ b/tests/_fixtures/golden_streams/expected/antigravity_e_multi_generation.json @@ -94,6 +94,7 @@ ], "model_used": "gemini-3.5-flash", "num_turns": 3, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/claude_a_single_text_turn.json b/tests/_fixtures/golden_streams/expected/claude_a_single_text_turn.json index dd99b9f5b..f28d29f21 100644 --- a/tests/_fixtures/golden_streams/expected/claude_a_single_text_turn.json +++ b/tests/_fixtures/golden_streams/expected/claude_a_single_text_turn.json @@ -40,6 +40,7 @@ ], "model_used": "mock-model", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": "done", diff --git a/tests/_fixtures/golden_streams/expected/claude_b_tool_use_result.json b/tests/_fixtures/golden_streams/expected/claude_b_tool_use_result.json index 42b675546..426a1f8a6 100644 --- a/tests/_fixtures/golden_streams/expected/claude_b_tool_use_result.json +++ b/tests/_fixtures/golden_streams/expected/claude_b_tool_use_result.json @@ -62,6 +62,7 @@ ], "model_used": "mock-model", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": "done", diff --git a/tests/_fixtures/golden_streams/expected/claude_c_multi_emission_delta.json b/tests/_fixtures/golden_streams/expected/claude_c_multi_emission_delta.json index 80de2a507..c8113f367 100644 --- a/tests/_fixtures/golden_streams/expected/claude_c_multi_emission_delta.json +++ b/tests/_fixtures/golden_streams/expected/claude_c_multi_emission_delta.json @@ -116,6 +116,7 @@ ], "model_used": "mock-model", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": "done", diff --git a/tests/_fixtures/golden_streams/expected/claude_d_subagent_terminal.json b/tests/_fixtures/golden_streams/expected/claude_d_subagent_terminal.json index e9fa5477f..f4583c9f8 100644 --- a/tests/_fixtures/golden_streams/expected/claude_d_subagent_terminal.json +++ b/tests/_fixtures/golden_streams/expected/claude_d_subagent_terminal.json @@ -89,6 +89,7 @@ ], "model_used": "mock-model", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": "done", diff --git a/tests/_fixtures/golden_streams/expected/claude_e_model_usage_and_backfill.json b/tests/_fixtures/golden_streams/expected/claude_e_model_usage_and_backfill.json index 7b46dede4..4b481b44e 100644 --- a/tests/_fixtures/golden_streams/expected/claude_e_model_usage_and_backfill.json +++ b/tests/_fixtures/golden_streams/expected/claude_e_model_usage_and_backfill.json @@ -48,6 +48,7 @@ ], "model_used": "mock-model", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": "done", diff --git a/tests/_fixtures/golden_streams/expected/claude_f_orphaned_tool.json b/tests/_fixtures/golden_streams/expected/claude_f_orphaned_tool.json index f5f8cd9b2..0027d3b26 100644 --- a/tests/_fixtures/golden_streams/expected/claude_f_orphaned_tool.json +++ b/tests/_fixtures/golden_streams/expected/claude_f_orphaned_tool.json @@ -63,6 +63,7 @@ ], "model_used": "mock-model", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": "done", diff --git a/tests/_fixtures/golden_streams/expected/claude_g_crash_format_placeholder.json b/tests/_fixtures/golden_streams/expected/claude_g_crash_format_placeholder.json index 69b9cfeee..0d5ff6800 100644 --- a/tests/_fixtures/golden_streams/expected/claude_g_crash_format_placeholder.json +++ b/tests/_fixtures/golden_streams/expected/claude_g_crash_format_placeholder.json @@ -12,6 +12,7 @@ "messages": [], "model_used": null, "num_turns": null, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": null, diff --git a/tests/_fixtures/golden_streams/expected/claude_h1_timeout_process_error.json b/tests/_fixtures/golden_streams/expected/claude_h1_timeout_process_error.json index 8eee6edfb..1e5a2c4cc 100644 --- a/tests/_fixtures/golden_streams/expected/claude_h1_timeout_process_error.json +++ b/tests/_fixtures/golden_streams/expected/claude_h1_timeout_process_error.json @@ -12,6 +12,7 @@ "messages": [], "model_used": null, "num_turns": null, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": null, diff --git a/tests/_fixtures/golden_streams/expected/claude_h2_process_error_crash.json b/tests/_fixtures/golden_streams/expected/claude_h2_process_error_crash.json index 997a23868..cc7adfb8a 100644 --- a/tests/_fixtures/golden_streams/expected/claude_h2_process_error_crash.json +++ b/tests/_fixtures/golden_streams/expected/claude_h2_process_error_crash.json @@ -12,6 +12,7 @@ "messages": [], "model_used": null, "num_turns": null, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": null, diff --git a/tests/_fixtures/golden_streams/expected/claude_i_in_loop_deadline_break.json b/tests/_fixtures/golden_streams/expected/claude_i_in_loop_deadline_break.json index f44f93cec..20a518856 100644 --- a/tests/_fixtures/golden_streams/expected/claude_i_in_loop_deadline_break.json +++ b/tests/_fixtures/golden_streams/expected/claude_i_in_loop_deadline_break.json @@ -40,6 +40,7 @@ ], "model_used": "mock-model", "num_turns": null, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": null, diff --git a/tests/_fixtures/golden_streams/expected/codex_a_agent_message_only.json b/tests/_fixtures/golden_streams/expected/codex_a_agent_message_only.json index c755ee0c8..550cbf256 100644 --- a/tests/_fixtures/golden_streams/expected/codex_a_agent_message_only.json +++ b/tests/_fixtures/golden_streams/expected/codex_a_agent_message_only.json @@ -40,6 +40,7 @@ ], "model_used": "gpt-5-codex", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/codex_b_command_execution.json b/tests/_fixtures/golden_streams/expected/codex_b_command_execution.json index 685fe7cf6..47a95a64d 100644 --- a/tests/_fixtures/golden_streams/expected/codex_b_command_execution.json +++ b/tests/_fixtures/golden_streams/expected/codex_b_command_execution.json @@ -71,6 +71,7 @@ ], "model_used": "gpt-5-codex", "num_turns": 2, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/codex_c_reasoning_placeholder.json b/tests/_fixtures/golden_streams/expected/codex_c_reasoning_placeholder.json index 77a89092f..f4ae0de95 100644 --- a/tests/_fixtures/golden_streams/expected/codex_c_reasoning_placeholder.json +++ b/tests/_fixtures/golden_streams/expected/codex_c_reasoning_placeholder.json @@ -67,6 +67,7 @@ ], "model_used": "gpt-5-codex", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/codex_d_cross_flush_is_error.json b/tests/_fixtures/golden_streams/expected/codex_d_cross_flush_is_error.json index 8a316c139..68c7e39f3 100644 --- a/tests/_fixtures/golden_streams/expected/codex_d_cross_flush_is_error.json +++ b/tests/_fixtures/golden_streams/expected/codex_d_cross_flush_is_error.json @@ -62,6 +62,7 @@ ], "model_used": "gpt-5-codex", "num_turns": 2, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/codex_e_orphan_tool.json b/tests/_fixtures/golden_streams/expected/codex_e_orphan_tool.json index dcaaf846c..a4e79df1e 100644 --- a/tests/_fixtures/golden_streams/expected/codex_e_orphan_tool.json +++ b/tests/_fixtures/golden_streams/expected/codex_e_orphan_tool.json @@ -62,6 +62,7 @@ ], "model_used": "gpt-5-codex", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": null, diff --git a/tests/_fixtures/golden_streams/expected/codex_f_collab_fallback.json b/tests/_fixtures/golden_streams/expected/codex_f_collab_fallback.json index 3d8031b04..34cecd0b0 100644 --- a/tests/_fixtures/golden_streams/expected/codex_f_collab_fallback.json +++ b/tests/_fixtures/golden_streams/expected/codex_f_collab_fallback.json @@ -120,6 +120,7 @@ ], "model_used": "gpt-5-codex", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": null, diff --git a/tests/_fixtures/golden_streams/expected/codex_g_items_rebuild.json b/tests/_fixtures/golden_streams/expected/codex_g_items_rebuild.json index bd23e300b..b4c078f68 100644 --- a/tests/_fixtures/golden_streams/expected/codex_g_items_rebuild.json +++ b/tests/_fixtures/golden_streams/expected/codex_g_items_rebuild.json @@ -40,6 +40,7 @@ ], "model_used": "gpt-5-codex", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": null, diff --git a/tests/_fixtures/golden_streams/expected/codex_h_no_turn_completed_crash.json b/tests/_fixtures/golden_streams/expected/codex_h_no_turn_completed_crash.json index b0414c980..e9c7c2a38 100644 --- a/tests/_fixtures/golden_streams/expected/codex_h_no_turn_completed_crash.json +++ b/tests/_fixtures/golden_streams/expected/codex_h_no_turn_completed_crash.json @@ -40,6 +40,7 @@ ], "model_used": "gpt-5-codex", "num_turns": 1, + "provider_errors": [], "result_summary": null, "timestamp": "", "token_usage": { diff --git a/tests/_fixtures/golden_streams/expected/opencode_a_single_text_turn.json b/tests/_fixtures/golden_streams/expected/opencode_a_single_text_turn.json index 7a6758156..003ac376d 100644 --- a/tests/_fixtures/golden_streams/expected/opencode_a_single_text_turn.json +++ b/tests/_fixtures/golden_streams/expected/opencode_a_single_text_turn.json @@ -40,6 +40,7 @@ ], "model_used": "deepseek/deepseek-v4-pro", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": null, diff --git a/tests/_fixtures/golden_streams/expected/opencode_b_tool_call_resolved.json b/tests/_fixtures/golden_streams/expected/opencode_b_tool_call_resolved.json index 7953c6f08..dfbb4464d 100644 --- a/tests/_fixtures/golden_streams/expected/opencode_b_tool_call_resolved.json +++ b/tests/_fixtures/golden_streams/expected/opencode_b_tool_call_resolved.json @@ -89,6 +89,7 @@ ], "model_used": "deepseek/deepseek-v4-pro", "num_turns": 2, + "provider_errors": [], "result_summary": { "is_error": false, "result": null, diff --git a/tests/_fixtures/golden_streams/expected/opencode_c_multi_step_tiling.json b/tests/_fixtures/golden_streams/expected/opencode_c_multi_step_tiling.json index 01eeb86d0..1d8b56199 100644 --- a/tests/_fixtures/golden_streams/expected/opencode_c_multi_step_tiling.json +++ b/tests/_fixtures/golden_streams/expected/opencode_c_multi_step_tiling.json @@ -89,6 +89,7 @@ ], "model_used": "deepseek/deepseek-v4-pro", "num_turns": 2, + "provider_errors": [], "result_summary": { "is_error": false, "result": null, diff --git a/tests/_fixtures/golden_streams/expected/opencode_d_orphaned_tool.json b/tests/_fixtures/golden_streams/expected/opencode_d_orphaned_tool.json index b791f9ee7..947144a2a 100644 --- a/tests/_fixtures/golden_streams/expected/opencode_d_orphaned_tool.json +++ b/tests/_fixtures/golden_streams/expected/opencode_d_orphaned_tool.json @@ -71,6 +71,7 @@ ], "model_used": "deepseek/deepseek-v4-pro", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": null, diff --git a/tests/_fixtures/golden_streams/expected/opencode_e_error_after_generation.json b/tests/_fixtures/golden_streams/expected/opencode_e_error_after_generation.json index 66059b14a..5908b0e64 100644 --- a/tests/_fixtures/golden_streams/expected/opencode_e_error_after_generation.json +++ b/tests/_fixtures/golden_streams/expected/opencode_e_error_after_generation.json @@ -40,6 +40,7 @@ ], "model_used": "deepseek/deepseek-v4-pro", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": true, "result": "OpenCode error: 401 from the provider", diff --git a/tests/_fixtures/golden_streams/expected/pi_a_single_text_turn.json b/tests/_fixtures/golden_streams/expected/pi_a_single_text_turn.json index d794274ae..713ec67e8 100644 --- a/tests/_fixtures/golden_streams/expected/pi_a_single_text_turn.json +++ b/tests/_fixtures/golden_streams/expected/pi_a_single_text_turn.json @@ -40,6 +40,7 @@ ], "model_used": "openrouter/moonshotai/kimi-k3", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": null, diff --git a/tests/_fixtures/golden_streams/expected/pi_b_tool_call_resolved.json b/tests/_fixtures/golden_streams/expected/pi_b_tool_call_resolved.json index 06e3816e5..b0b156d90 100644 --- a/tests/_fixtures/golden_streams/expected/pi_b_tool_call_resolved.json +++ b/tests/_fixtures/golden_streams/expected/pi_b_tool_call_resolved.json @@ -138,6 +138,7 @@ ], "model_used": "openrouter/moonshotai/kimi-k3", "num_turns": 3, + "provider_errors": [], "result_summary": { "is_error": false, "result": null, diff --git a/tests/_fixtures/golden_streams/expected/pi_c_multi_turn_tiling.json b/tests/_fixtures/golden_streams/expected/pi_c_multi_turn_tiling.json index deebf4700..77db1ba7b 100644 --- a/tests/_fixtures/golden_streams/expected/pi_c_multi_turn_tiling.json +++ b/tests/_fixtures/golden_streams/expected/pi_c_multi_turn_tiling.json @@ -89,6 +89,7 @@ ], "model_used": "openrouter/moonshotai/kimi-k3", "num_turns": 2, + "provider_errors": [], "result_summary": { "is_error": false, "result": null, diff --git a/tests/_fixtures/golden_streams/expected/pi_d_orphaned_tool.json b/tests/_fixtures/golden_streams/expected/pi_d_orphaned_tool.json index c5e8d127e..126fd53ba 100644 --- a/tests/_fixtures/golden_streams/expected/pi_d_orphaned_tool.json +++ b/tests/_fixtures/golden_streams/expected/pi_d_orphaned_tool.json @@ -71,6 +71,7 @@ ], "model_used": "openrouter/moonshotai/kimi-k3", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": null, diff --git a/tests/_fixtures/golden_streams/expected/pi_e_error_after_generation.json b/tests/_fixtures/golden_streams/expected/pi_e_error_after_generation.json index f860de537..cfaef4072 100644 --- a/tests/_fixtures/golden_streams/expected/pi_e_error_after_generation.json +++ b/tests/_fixtures/golden_streams/expected/pi_e_error_after_generation.json @@ -57,6 +57,7 @@ ], "model_used": "openrouter/moonshotai/kimi-k3", "num_turns": 2, + "provider_errors": [], "result_summary": { "is_error": true, "result": "Pi error: provider returned 529 after 5 retries", diff --git a/tests/_fixtures/golden_streams/expected/pi_f_duplicate_turn_end.json b/tests/_fixtures/golden_streams/expected/pi_f_duplicate_turn_end.json index db88ceeca..467d8afa1 100644 --- a/tests/_fixtures/golden_streams/expected/pi_f_duplicate_turn_end.json +++ b/tests/_fixtures/golden_streams/expected/pi_f_duplicate_turn_end.json @@ -57,6 +57,7 @@ ], "model_used": "openrouter/moonshotai/kimi-k3", "num_turns": 1, + "provider_errors": [], "result_summary": { "is_error": false, "result": null, diff --git a/tests/test_codex_agent.py b/tests/test_codex_agent.py index a02a5e308..e2cf8a068 100644 --- a/tests/test_codex_agent.py +++ b/tests/test_codex_agent.py @@ -2730,3 +2730,88 @@ def test_both_sub_messages_still_share_one_window(self): def test_a_window_entirely_covered_by_its_tool_splits_zero_two_ways(self): published = self._published(window_ms=1000, tool_from_ms=0, tool_to_ms=1000, think_out=800, action_out=200) assert [m.generation_duration_ms for m in published] == [0.0, 0.0] + + +class TestProviderErrorNotifications: + """Codex's `error` notification (incl. `willRetry`) reaches the TurnRecord. + + Before this, `dispatch` dropped it: a provider stream that stalled for 9 + minutes and was retried by codex read as pure model latency (coder_eval#151). + """ + + @staticmethod + def _state(): + from coder_eval.agents.codex_agent import _CodexTurnState + from coder_eval.streaming.callbacks import CompositeStreamCallback + from coder_eval.streaming.collector import EventCollector + + collector = EventCollector() + agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX, model="gpt-5.5")) + st = _CodexTurnState( + agent, + emit=CompositeStreamCallback([collector]), + task_id="codex", + turn_id="codex-1", + collector=collector, + commands=[], + messages=[], + user_input="go", + iteration=1, + turn_start_time=time.monotonic(), + ) + return st, collector + + @staticmethod + def _notification(error: dict, *, will_retry: bool) -> SimpleNamespace: + from openai_codex.generated.v2_all import ErrorNotification + + payload = ErrorNotification.model_validate( + {"error": error, "threadId": "t1", "turnId": "u1", "willRetry": will_retry} + ) + return SimpleNamespace(method="error", payload=payload) + + def test_retried_stream_error_is_recorded_with_kind_and_status(self): + from coder_eval.streaming.events import AgentEndStatus + + st, collector = self._state() + note = self._notification( + { + "message": "stream disconnected before completion: idle timeout", + "codexErrorInfo": {"responseStreamDisconnected": {"httpStatusCode": 504}}, + "additionalDetails": "attempt 1/5", + }, + will_retry=True, + ) + + assert st.dispatch(note) is False # an error never ends the pump + st.finalize(AgentEndStatus.CRASHED, crashed=True, crash_reason="timeout") + + record = collector.build_turn_record() + assert len(record.provider_errors) == 1 + err = record.provider_errors[0] + assert err.will_retry is True + assert err.kind == "responseStreamDisconnected" + assert err.http_status == 504 + assert err.details == "attempt 1/5" + assert "idle timeout" in err.message + + def test_bare_category_and_missing_info(self): + from coder_eval.streaming.events import AgentEndStatus + + st, collector = self._state() + st.dispatch(self._notification({"message": "slow down", "codexErrorInfo": "serverOverloaded"}, will_retry=True)) + st.dispatch(self._notification({"message": "gave up"}, will_retry=False)) + st.finalize(AgentEndStatus.COMPLETED) + + errs = collector.build_turn_record().provider_errors + assert [(e.kind, e.http_status, e.will_retry) for e in errs] == [ + ("serverOverloaded", None, True), + (None, None, False), + ] + + def test_clean_turn_has_no_provider_errors(self): + from coder_eval.streaming.events import AgentEndStatus + + st, collector = self._state() + st.finalize(AgentEndStatus.COMPLETED) + assert collector.build_turn_record().provider_errors == []