Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion docs/REPORT_SCHEMA.md
Original file line number Diff line number Diff line change
Expand Up @@ -210,7 +210,10 @@ token/cost budget breach, which fires only after every criterion is already scor
cache buckets, captured proxy-side on the LiteLLM open-weight backend and rendered
by the evalboard as a per-call table; empty on every other
backend), `num_turns`, `max_turns_exhausted`,
`result_summary` (`{is_error, subtype, stop_reason, result}`), `crashed`,
`result_summary` (`{is_error, subtype, stop_reason, result}`), `provider_errors`
(`list[ProviderError]` — `{at, message, kind, http_status, will_retry, details}`, one
row per model-provider error the agent's own client reported, including errors it
retried itself; filled by the codex adapter, empty on other backends), `crashed`,
`crash_reason`.

> **Token invariant.** Summing the four token buckets across `messages`
Expand Down
43 changes: 43 additions & 0 deletions src/coder_eval/agents/codex_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
import time
from collections.abc import Callable
from datetime import datetime
from enum import Enum
from pathlib import Path
from typing import Any, ClassVar, NamedTuple
from urllib.parse import urlparse
Expand All @@ -33,6 +34,7 @@
CommandTelemetry,
ContentBlock,
DirectRoute,
ProviderError,
SystemPromptSemantics,
TokenUsage,
TranscriptMessage,
Expand Down Expand Up @@ -274,6 +276,34 @@ def _message_uncached_input(m: AssistantMessage) -> int:
_ZSH_PROFILE_NAMES = frozenset({".zshenv", ".zprofile", ".zshrc"})


def _provider_error_from(payload: Any) -> ProviderError:
"""Map a codex ``ErrorNotification`` payload onto a ``ProviderError``.

``codexErrorInfo`` is either a bare category (enum) or a one-key object whose
key is the category and whose value may carry ``httpStatusCode``.
"""
error = getattr(payload, "error", None)
kind: str | None = None
http_status: int | None = None
info = getattr(getattr(error, "codex_error_info", None), "root", None)
if isinstance(info, Enum):
kind = str(info.value)
elif hasattr(info, "model_dump"):
dumped = info.model_dump(by_alias=True, mode="json")
if len(dumped) == 1:
kind, body = next(iter(dumped.items()))
if isinstance(body, dict) and isinstance(body.get("httpStatusCode"), int):
http_status = body["httpStatusCode"]
return ProviderError(
at=datetime.now(),
message=str(getattr(error, "message", "") or ""),
kind=kind,
http_status=http_status,
will_retry=bool(getattr(payload, "will_retry", False)),
details=getattr(error, "additional_details", None),
)


def _get_item_root(notification: Any) -> Any:
"""Extract the typed item root from a Codex SDK notification.

Expand Down Expand Up @@ -338,6 +368,7 @@ def __init__(
# Live pump scratch (set during streaming).
self.turn_result: Any = None
self.latest_token_usage: Any = None
self.provider_errors: list[ProviderError] = []
self.agent_message_chunks: list[str] = []
# Sequence per executable item, assigned at item/started and reused at
# item/completed via this id->seq map.
Expand Down Expand Up @@ -546,8 +577,19 @@ def dispatch(self, notification: Any) -> bool:
self.on_token_usage_updated(notification)
elif method == "turn/completed":
return self.on_turn_completed(notification)
elif method == "error":
self.on_error(notification)
return False

def on_error(self, notification: Any) -> None:
"""Record a provider error; codex retries a stalled stream silently otherwise."""
err = _provider_error_from(notification.payload)
self.provider_errors.append(err)
self._agent._log.warning(
f"provider error (will_retry={err.will_retry}, kind={err.kind}, "
+ f"http_status={err.http_status}): {err.message}"
)

def on_item_started(self, notification: Any) -> None:
"""Emit ToolStartEvent + record the tool_use block for every tool-like item."""
root = _get_item_root(notification)
Expand Down Expand Up @@ -765,6 +807,7 @@ def finalize(self, status: AgentEndStatus, *, crashed: bool = False, crash_reaso
assistant_turn_count=1,
messages=self.messages,
num_turns=max(self.api_calls, 1),
provider_errors=list(self.provider_errors),
crashed=crashed,
crash_reason=crash_reason,
max_turns_exhausted=status is AgentEndStatus.MAX_TURNS_EXHAUSTED,
Expand Down
2 changes: 2 additions & 0 deletions src/coder_eval/models/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -228,6 +228,7 @@
CommandTelemetry,
ContentBlock,
ProviderCallCost,
ProviderError,
ReconciliationMessage,
SlowestCommandInfo,
TokenUsage,
Expand Down Expand Up @@ -349,6 +350,7 @@
"CommandStatistics",
"ContentBlock",
"ProviderCallCost",
"ProviderError",
"ReconciliationMessage",
"SlowestCommandInfo",
"TokenUsage",
Expand Down
9 changes: 9 additions & 0 deletions src/coder_eval/models/results.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@
CommandStatistics,
CommandTelemetry,
ProviderCallCost,
ProviderError,
TokenUsage,
TranscriptMessage,
)
Expand Down Expand Up @@ -417,6 +418,14 @@ class TurnRecord(BaseModel):
"these calls' cost_usd (the real bill), not the static rate-card estimate."
),
)
provider_errors: list[ProviderError] = Field(
default_factory=list,
description=(
"Model-provider errors the agent's own client reported during this turn, in arrival "
"order, including ones it retried itself (will_retry=True). Filled by the codex "
"adapter from its `error` notification; empty on backends that do not surface them."
),
)
crashed: bool = Field(
default=False,
description=(
Expand Down
18 changes: 18 additions & 0 deletions src/coder_eval/models/telemetry.py
Original file line number Diff line number Diff line change
Expand Up @@ -134,6 +134,24 @@ class ProviderCallCost(BaseModel):
output_tokens: int | None = Field(default=None, description="Completion tokens for this call.")


class ProviderError(BaseModel):
"""One model-provider error the agent's own client reported mid-turn.

Codex retries a stalled or failed provider stream internally and says so only
through its ``error`` notification (``willRetry``). Without this record a
9-minute provider stall reads as pure model latency.
"""

at: datetime = Field(description="When the harness received the error notification.")
message: str = Field(description="The client's error message.")
kind: str | None = Field(
default=None, description="Client error category (codex `codexErrorInfo`), e.g. responseStreamDisconnected."
)
http_status: int | None = Field(default=None, description="Upstream HTTP status, when the client forwarded one.")
will_retry: bool = Field(description="True when the client retries the request itself after this error.")
details: str | None = Field(default=None, description="Extra detail the client attached, if any.")


class ContentBlock(BaseModel):
"""One content block within a message, in emission order.

Expand Down
1 change: 1 addition & 0 deletions src/coder_eval/streaming/collector.py
Original file line number Diff line number Diff line change
Expand Up @@ -224,6 +224,7 @@ def build_turn_record(self) -> TurnRecord:
num_turns=end.num_turns,
max_turns_exhausted=end.max_turns_exhausted,
result_summary=end.result_summary,
provider_errors=end.provider_errors,
crashed=end.crashed,
crash_reason=end.crash_reason,
harness_startup_ms=startup_ms,
Expand Down
2 changes: 2 additions & 0 deletions src/coder_eval/streaming/events.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@

from coder_eval.models import (
CommandTelemetry,
ProviderError,
ResultSummary,
TokenUsage,
TranscriptMessage,
Expand Down Expand Up @@ -169,6 +170,7 @@ class AgentEndEvent(StreamEvent):
num_turns: int | None = None
max_turns_exhausted: bool = False
result_summary: ResultSummary | None = None
provider_errors: list[ProviderError] = Field(default_factory=list)
crashed: bool = False
crash_reason: str | None = None
duration_seconds: float = 0.0
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
],
"model_used": "gemini-3.5-flash",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,7 @@
],
"model_used": "gemini-3.5-flash",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -80,6 +80,7 @@
],
"model_used": "gemini-3.5-flash",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,7 @@
],
"model_used": "gemini-3.5-flash",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -94,6 +94,7 @@
],
"model_used": "gemini-3.5-flash",
"num_turns": 3,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
],
"model_used": "mock-model",
"num_turns": 1,
"provider_errors": [],
"result_summary": {
"is_error": false,
"result": "done",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,7 @@
],
"model_used": "mock-model",
"num_turns": 1,
"provider_errors": [],
"result_summary": {
"is_error": false,
"result": "done",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -116,6 +116,7 @@
],
"model_used": "mock-model",
"num_turns": 1,
"provider_errors": [],
"result_summary": {
"is_error": false,
"result": "done",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -89,6 +89,7 @@
],
"model_used": "mock-model",
"num_turns": 1,
"provider_errors": [],
"result_summary": {
"is_error": false,
"result": "done",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -48,6 +48,7 @@
],
"model_used": "mock-model",
"num_turns": 1,
"provider_errors": [],
"result_summary": {
"is_error": false,
"result": "done",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,7 @@
],
"model_used": "mock-model",
"num_turns": 1,
"provider_errors": [],
"result_summary": {
"is_error": false,
"result": "done",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
"messages": [],
"model_used": null,
"num_turns": null,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": null,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
"messages": [],
"model_used": null,
"num_turns": null,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": null,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
"messages": [],
"model_used": null,
"num_turns": null,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": null,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
],
"model_used": "mock-model",
"num_turns": null,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": null,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
],
"model_used": "gpt-5-codex",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,7 @@
],
"model_used": "gpt-5-codex",
"num_turns": 2,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -67,6 +67,7 @@
],
"model_used": "gpt-5-codex",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,7 @@
],
"model_used": "gpt-5-codex",
"num_turns": 2,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,7 @@
],
"model_used": "gpt-5-codex",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": null,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -120,6 +120,7 @@
],
"model_used": "gpt-5-codex",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": null,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
],
"model_used": "gpt-5-codex",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": null,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
],
"model_used": "gpt-5-codex",
"num_turns": 1,
"provider_errors": [],
"result_summary": null,
"timestamp": "<scrubbed>",
"token_usage": {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
],
"model_used": "deepseek/deepseek-v4-pro",
"num_turns": 1,
"provider_errors": [],
"result_summary": {
"is_error": false,
"result": null,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -89,6 +89,7 @@
],
"model_used": "deepseek/deepseek-v4-pro",
"num_turns": 2,
"provider_errors": [],
"result_summary": {
"is_error": false,
"result": null,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -89,6 +89,7 @@
],
"model_used": "deepseek/deepseek-v4-pro",
"num_turns": 2,
"provider_errors": [],
"result_summary": {
"is_error": false,
"result": null,
Expand Down
Loading
Loading