Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
70 changes: 1 addition & 69 deletions src/coder_eval/orchestrator.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@
import asyncio
import logging
import os
import re
import tempfile
import time
import uuid
Expand Down Expand Up @@ -80,6 +79,7 @@
from .result_metrics import turn_time_buckets, visible_turn_count
from .sandbox import Sandbox
from .simulation import DialogStopReason, SimulatorResult, UserSimulator, evaluate_stop
from .simulation.utterance import extract_utterance as _extract_utterance
from .streaming.callbacks import CompositeStreamCallback, StreamCallback, TaskScopedCallback, safe_emit
from .streaming.events import CriteriaCheckEvent, CriterionSummary
from .telemetry import Scalar, hash_identifier
Expand Down Expand Up @@ -153,13 +153,6 @@ async def _pump_stream(
log_fn("[%s] %s", label, line)


# Structural tags emitted by ClaudeCodeAgent._format_messages, which is the SSOT
# for the vocabulary. Other bracketed words (markdown footnotes, pylint codes,
# unknown SDK message types) are intentionally NOT matched — they pass through as
# content. Telemetry-only, not a correctness-critical parser.
_UTTERANCE_TAG_RE = re.compile(r"^\[(ASSISTANT|RESULT - SUCCESS|RESULT - ERROR|TOOL USE)\](?: (.*))?$")


class EvalRouteOverrides(NamedTuple):
"""``checker_context.api_route``'s override fields. Named fields (rather than
a bare tuple) so a future transposition at a call site is a typo'd attribute,
Expand Down Expand Up @@ -189,67 +182,6 @@ def _format_routing(route: ApiRoute, effective_model: str | None = None) -> str:
return name


def _extract_utterance(raw: str) -> str:
"""Collapse a ClaudeCodeAgent-formatted transcript to a clean utterance.

Input looks like::

[ASSISTANT] Sure, I'll do X.
[TOOL USE] Read
[RESULT - SUCCESS] Here is the answer...

Prefers a non-empty ``[RESULT - ...]`` payload — the SDK's canonical final
utterance, which duplicates the final assistant text and otherwise makes
conversation.log read as if every message is repeated. Falls back to
concatenated ``[ASSISTANT]`` blocks, including any content appearing before
the first tag. ``[TOOL USE]`` lines are dropped, and untagged input (a pinned
``initial_prompt``) is returned unchanged.

Asymmetric on purpose: ``[RESULT - SUCCESS]`` strips its label, while
``[RESULT - ERROR]`` KEEPS its prefix so the error state stays visible in the
log.
"""
if not raw:
return ""
lines = raw.splitlines()
if not any(_UTTERANCE_TAG_RE.match(ln) for ln in lines):
return raw

assistant_parts: list[str] = []
result_parts: list[str] = []
# Pre-tag content becomes an implicit ASSISTANT block (not dropped).
current_tag: str = "ASSISTANT"
current_buf: list[str] = []

def _flush() -> None:
text = "\n".join(current_buf).strip()
if not text:
return
if current_tag == "ASSISTANT":
assistant_parts.append(text)
elif current_tag == "RESULT - SUCCESS":
result_parts.append(text)
elif current_tag == "RESULT - ERROR":
result_parts.append(f"[RESULT - ERROR] {text}")
# TOOL USE is dropped.

for ln in lines:
match = _UTTERANCE_TAG_RE.match(ln)
if match:
_flush()
current_tag = match.group(1)
current_buf = [match.group(2) or ""]
else:
current_buf.append(ln)
_flush()

if result_parts:
return "\n\n".join(result_parts)
if assistant_parts:
return "\n\n".join(assistant_parts)
return raw


def _extract_failure_reason(result: CriterionResult) -> str | None:
"""Streaming-event wrapper around ``_short_failure_reason``.

Expand Down
6 changes: 5 additions & 1 deletion src/coder_eval/simulation/user_simulator.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@
from typing import TYPE_CHECKING, Any

from coder_eval.models import AgentKind, ApiRoute, SimulationConfig, parse_agent_config
from coder_eval.simulation.utterance import extract_utterance


if TYPE_CHECKING:
Expand Down Expand Up @@ -333,12 +334,15 @@ async def next_user_message(self, dialog_pairs: list[tuple[str, str]]) -> Simula
# Simulator emits one user utterance per call, so cap the inner loop at 1 turn.
turn = await self._agent.communicate(prompt, max_turns=1)
raw = turn.agent_output or ""
# agent_output is a tagged transcript (`[ASSISTANT] …` / `[RESULT - SUCCESS] …`
# repeating the same text), not the utterance. Send the agent the utterance.
utterance = extract_utterance(raw)
usage = turn.token_usage
input_tokens = usage.uncached_input_tokens if usage is not None else None
output_tokens = usage.output_tokens if usage is not None else None

stop_requested = self.config.stop_token in raw
cleaned = strip_stop_token(raw, self.config.stop_token) if stop_requested else raw.strip()
cleaned = strip_stop_token(utterance, self.config.stop_token) if stop_requested else utterance.strip()

# The agent still needs SOMETHING to react to after the stop token is
# stripped, and the dialog terminates on this turn anyway.
Expand Down
79 changes: 79 additions & 0 deletions src/coder_eval/simulation/utterance.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
"""Collapse a Claude Code agent's formatted transcript into one plain utterance.

``ClaudeCodeAgent`` reports ``agent_output`` as a tagged transcript, not as the
text the model said. Anything that treats that output as a message (the user
simulator's next turn, the conversation log) must collapse it first.
"""

from __future__ import annotations

import re


# Structural tags emitted by ClaudeCodeAgent._format_messages, which is the SSOT
# for the vocabulary. Other bracketed words (markdown footnotes, pylint codes,
# unknown SDK message types) are intentionally NOT matched — they pass through as
# content. The user simulator's next turn goes through this parser, so a tag it
# misses reaches the coding agent as literal text.
_UTTERANCE_TAG_RE = re.compile(r"^\[(ASSISTANT|RESULT - SUCCESS|RESULT - ERROR|TOOL USE)\](?: (.*))?$")


def extract_utterance(raw: str) -> str:
"""Collapse a ClaudeCodeAgent-formatted transcript to a clean utterance.

Input looks like::

[ASSISTANT] Sure, I'll do X.
[TOOL USE] Read
[RESULT - SUCCESS] Here is the answer...

Prefers a non-empty ``[RESULT - ...]`` payload — the SDK's canonical final
utterance, which duplicates the final assistant text; without this collapse
every simulated-user turn reaches the coding agent twice, wrapped in tags. Falls back to
concatenated ``[ASSISTANT]`` blocks, including any content appearing before
the first tag. ``[TOOL USE]`` lines are dropped, and untagged input (a pinned
``initial_prompt``) is returned unchanged.

Asymmetric on purpose: ``[RESULT - SUCCESS]`` strips its label, while
``[RESULT - ERROR]`` KEEPS its prefix so the error state stays visible in the
log.
"""
if not raw:
return ""
lines = raw.splitlines()
if not any(_UTTERANCE_TAG_RE.match(ln) for ln in lines):
return raw

assistant_parts: list[str] = []
result_parts: list[str] = []
# Pre-tag content becomes an implicit ASSISTANT block (not dropped).
current_tag: str = "ASSISTANT"
current_buf: list[str] = []

def _flush() -> None:
text = "\n".join(current_buf).strip()
if not text:
return
if current_tag == "ASSISTANT":
assistant_parts.append(text)
elif current_tag == "RESULT - SUCCESS":
result_parts.append(text)
elif current_tag == "RESULT - ERROR":
result_parts.append(f"[RESULT - ERROR] {text}")
# TOOL USE is dropped.

for ln in lines:
match = _UTTERANCE_TAG_RE.match(ln)
if match:
_flush()
current_tag = match.group(1)
current_buf = [match.group(2) or ""]
else:
current_buf.append(ln)
_flush()

if result_parts:
return "\n\n".join(result_parts)
if assistant_parts:
return "\n\n".join(assistant_parts)
return raw
27 changes: 27 additions & 0 deletions tests/test_user_simulator.py
Original file line number Diff line number Diff line change
Expand Up @@ -382,3 +382,30 @@ async def test_disabled_simulator_is_no_op(self):
await sim.start()
assert not stub.started # start() bails early when disabled
await sim.stop()


class TestTaggedTranscriptOutput:
"""The real simulator is a ClaudeCodeAgent, whose ``agent_output`` is a tagged
transcript: the same text twice, under ``[ASSISTANT]`` and
``[RESULT - SUCCESS]``. Plain-text stubs hid that every simulated-user turn
reached the coding agent in that raw form (seen in every simulated task of
adhoc-2026-09-28_16-14-30)."""

async def test_tagged_reply_reaches_the_agent_as_one_plain_utterance(self):
opener = "Hey! I need a flow that triages my inbox."
stub = TextStubAgent([f"[ASSISTANT] {opener}\n[RESULT - SUCCESS] {opener}"])
sim = await _make_started(
UserSimulator(config=_sim_cfg(), task_description="T", initial_prompt=None, agent_override=stub)
)
r = await sim.next_user_message([])
assert r.text == opener
assert "[ASSISTANT]" in r.raw_text # raw stays available for telemetry

async def test_tagged_stop_turn_is_detected_and_leaves_no_tags(self):
stub = TextStubAgent(["[ASSISTANT] <<<DONE>>>\n[RESULT - SUCCESS] <<<DONE>>>"])
sim = await _make_started(
UserSimulator(config=_sim_cfg(), task_description="T", initial_prompt="start", agent_override=stub)
)
r = await sim.next_user_message([_pair("start", "Built it.")])
assert r.stop_requested is True
assert "[" not in r.text
Loading