diff --git a/src/coder_eval/orchestrator.py b/src/coder_eval/orchestrator.py index c10f1deb2..8bc75cbfa 100644 --- a/src/coder_eval/orchestrator.py +++ b/src/coder_eval/orchestrator.py @@ -3,7 +3,6 @@ import asyncio import logging import os -import re import tempfile import time import uuid @@ -80,6 +79,7 @@ from .result_metrics import turn_time_buckets, visible_turn_count from .sandbox import Sandbox from .simulation import DialogStopReason, SimulatorResult, UserSimulator, evaluate_stop +from .simulation.utterance import extract_utterance as _extract_utterance from .streaming.callbacks import CompositeStreamCallback, StreamCallback, TaskScopedCallback, safe_emit from .streaming.events import CriteriaCheckEvent, CriterionSummary from .telemetry import Scalar, hash_identifier @@ -153,13 +153,6 @@ async def _pump_stream( log_fn("[%s] %s", label, line) -# Structural tags emitted by ClaudeCodeAgent._format_messages, which is the SSOT -# for the vocabulary. Other bracketed words (markdown footnotes, pylint codes, -# unknown SDK message types) are intentionally NOT matched — they pass through as -# content. Telemetry-only, not a correctness-critical parser. -_UTTERANCE_TAG_RE = re.compile(r"^\[(ASSISTANT|RESULT - SUCCESS|RESULT - ERROR|TOOL USE)\](?: (.*))?$") - - class EvalRouteOverrides(NamedTuple): """``checker_context.api_route``'s override fields. Named fields (rather than a bare tuple) so a future transposition at a call site is a typo'd attribute, @@ -189,67 +182,6 @@ def _format_routing(route: ApiRoute, effective_model: str | None = None) -> str: return name -def _extract_utterance(raw: str) -> str: - """Collapse a ClaudeCodeAgent-formatted transcript to a clean utterance. - - Input looks like:: - - [ASSISTANT] Sure, I'll do X. - [TOOL USE] Read - [RESULT - SUCCESS] Here is the answer... - - Prefers a non-empty ``[RESULT - ...]`` payload — the SDK's canonical final - utterance, which duplicates the final assistant text and otherwise makes - conversation.log read as if every message is repeated. Falls back to - concatenated ``[ASSISTANT]`` blocks, including any content appearing before - the first tag. ``[TOOL USE]`` lines are dropped, and untagged input (a pinned - ``initial_prompt``) is returned unchanged. - - Asymmetric on purpose: ``[RESULT - SUCCESS]`` strips its label, while - ``[RESULT - ERROR]`` KEEPS its prefix so the error state stays visible in the - log. - """ - if not raw: - return "" - lines = raw.splitlines() - if not any(_UTTERANCE_TAG_RE.match(ln) for ln in lines): - return raw - - assistant_parts: list[str] = [] - result_parts: list[str] = [] - # Pre-tag content becomes an implicit ASSISTANT block (not dropped). - current_tag: str = "ASSISTANT" - current_buf: list[str] = [] - - def _flush() -> None: - text = "\n".join(current_buf).strip() - if not text: - return - if current_tag == "ASSISTANT": - assistant_parts.append(text) - elif current_tag == "RESULT - SUCCESS": - result_parts.append(text) - elif current_tag == "RESULT - ERROR": - result_parts.append(f"[RESULT - ERROR] {text}") - # TOOL USE is dropped. - - for ln in lines: - match = _UTTERANCE_TAG_RE.match(ln) - if match: - _flush() - current_tag = match.group(1) - current_buf = [match.group(2) or ""] - else: - current_buf.append(ln) - _flush() - - if result_parts: - return "\n\n".join(result_parts) - if assistant_parts: - return "\n\n".join(assistant_parts) - return raw - - def _extract_failure_reason(result: CriterionResult) -> str | None: """Streaming-event wrapper around ``_short_failure_reason``. diff --git a/src/coder_eval/simulation/user_simulator.py b/src/coder_eval/simulation/user_simulator.py index 27f79d3b6..1927474ec 100644 --- a/src/coder_eval/simulation/user_simulator.py +++ b/src/coder_eval/simulation/user_simulator.py @@ -27,6 +27,7 @@ from typing import TYPE_CHECKING, Any from coder_eval.models import AgentKind, ApiRoute, SimulationConfig, parse_agent_config +from coder_eval.simulation.utterance import extract_utterance if TYPE_CHECKING: @@ -333,12 +334,15 @@ async def next_user_message(self, dialog_pairs: list[tuple[str, str]]) -> Simula # Simulator emits one user utterance per call, so cap the inner loop at 1 turn. turn = await self._agent.communicate(prompt, max_turns=1) raw = turn.agent_output or "" + # agent_output is a tagged transcript (`[ASSISTANT] …` / `[RESULT - SUCCESS] …` + # repeating the same text), not the utterance. Send the agent the utterance. + utterance = extract_utterance(raw) usage = turn.token_usage input_tokens = usage.uncached_input_tokens if usage is not None else None output_tokens = usage.output_tokens if usage is not None else None stop_requested = self.config.stop_token in raw - cleaned = strip_stop_token(raw, self.config.stop_token) if stop_requested else raw.strip() + cleaned = strip_stop_token(utterance, self.config.stop_token) if stop_requested else utterance.strip() # The agent still needs SOMETHING to react to after the stop token is # stripped, and the dialog terminates on this turn anyway. diff --git a/src/coder_eval/simulation/utterance.py b/src/coder_eval/simulation/utterance.py new file mode 100644 index 000000000..5c6db9203 --- /dev/null +++ b/src/coder_eval/simulation/utterance.py @@ -0,0 +1,79 @@ +"""Collapse a Claude Code agent's formatted transcript into one plain utterance. + +``ClaudeCodeAgent`` reports ``agent_output`` as a tagged transcript, not as the +text the model said. Anything that treats that output as a message (the user +simulator's next turn, the conversation log) must collapse it first. +""" + +from __future__ import annotations + +import re + + +# Structural tags emitted by ClaudeCodeAgent._format_messages, which is the SSOT +# for the vocabulary. Other bracketed words (markdown footnotes, pylint codes, +# unknown SDK message types) are intentionally NOT matched — they pass through as +# content. The user simulator's next turn goes through this parser, so a tag it +# misses reaches the coding agent as literal text. +_UTTERANCE_TAG_RE = re.compile(r"^\[(ASSISTANT|RESULT - SUCCESS|RESULT - ERROR|TOOL USE)\](?: (.*))?$") + + +def extract_utterance(raw: str) -> str: + """Collapse a ClaudeCodeAgent-formatted transcript to a clean utterance. + + Input looks like:: + + [ASSISTANT] Sure, I'll do X. + [TOOL USE] Read + [RESULT - SUCCESS] Here is the answer... + + Prefers a non-empty ``[RESULT - ...]`` payload — the SDK's canonical final + utterance, which duplicates the final assistant text; without this collapse + every simulated-user turn reaches the coding agent twice, wrapped in tags. Falls back to + concatenated ``[ASSISTANT]`` blocks, including any content appearing before + the first tag. ``[TOOL USE]`` lines are dropped, and untagged input (a pinned + ``initial_prompt``) is returned unchanged. + + Asymmetric on purpose: ``[RESULT - SUCCESS]`` strips its label, while + ``[RESULT - ERROR]`` KEEPS its prefix so the error state stays visible in the + log. + """ + if not raw: + return "" + lines = raw.splitlines() + if not any(_UTTERANCE_TAG_RE.match(ln) for ln in lines): + return raw + + assistant_parts: list[str] = [] + result_parts: list[str] = [] + # Pre-tag content becomes an implicit ASSISTANT block (not dropped). + current_tag: str = "ASSISTANT" + current_buf: list[str] = [] + + def _flush() -> None: + text = "\n".join(current_buf).strip() + if not text: + return + if current_tag == "ASSISTANT": + assistant_parts.append(text) + elif current_tag == "RESULT - SUCCESS": + result_parts.append(text) + elif current_tag == "RESULT - ERROR": + result_parts.append(f"[RESULT - ERROR] {text}") + # TOOL USE is dropped. + + for ln in lines: + match = _UTTERANCE_TAG_RE.match(ln) + if match: + _flush() + current_tag = match.group(1) + current_buf = [match.group(2) or ""] + else: + current_buf.append(ln) + _flush() + + if result_parts: + return "\n\n".join(result_parts) + if assistant_parts: + return "\n\n".join(assistant_parts) + return raw diff --git a/tests/test_user_simulator.py b/tests/test_user_simulator.py index ca1e5aa49..9e591aafb 100644 --- a/tests/test_user_simulator.py +++ b/tests/test_user_simulator.py @@ -382,3 +382,30 @@ async def test_disabled_simulator_is_no_op(self): await sim.start() assert not stub.started # start() bails early when disabled await sim.stop() + + +class TestTaggedTranscriptOutput: + """The real simulator is a ClaudeCodeAgent, whose ``agent_output`` is a tagged + transcript: the same text twice, under ``[ASSISTANT]`` and + ``[RESULT - SUCCESS]``. Plain-text stubs hid that every simulated-user turn + reached the coding agent in that raw form (seen in every simulated task of + adhoc-2026-09-28_16-14-30).""" + + async def test_tagged_reply_reaches_the_agent_as_one_plain_utterance(self): + opener = "Hey! I need a flow that triages my inbox." + stub = TextStubAgent([f"[ASSISTANT] {opener}\n[RESULT - SUCCESS] {opener}"]) + sim = await _make_started( + UserSimulator(config=_sim_cfg(), task_description="T", initial_prompt=None, agent_override=stub) + ) + r = await sim.next_user_message([]) + assert r.text == opener + assert "[ASSISTANT]" in r.raw_text # raw stays available for telemetry + + async def test_tagged_stop_turn_is_detected_and_leaves_no_tags(self): + stub = TextStubAgent(["[ASSISTANT] <<>>\n[RESULT - SUCCESS] <<>>"]) + sim = await _make_started( + UserSimulator(config=_sim_cfg(), task_description="T", initial_prompt="start", agent_override=stub) + ) + r = await sim.next_user_message([_pair("start", "Built it.")]) + assert r.stop_requested is True + assert "[" not in r.text