diff --git a/docs/agents/ANTIGRAVITY.md b/docs/agents/ANTIGRAVITY.md index 522dc8bf..ad96081b 100644 --- a/docs/agents/ANTIGRAVITY.md +++ b/docs/agents/ANTIGRAVITY.md @@ -109,6 +109,14 @@ Antigravity exposes a `thinking_level` field (`minimal` / `low` / `medium` / Antigravity-specific — Claude Code and Codex don't take this field. Thinking tokens are billed as **output** tokens (see [Telemetry](#telemetry)). +### `system_prompt` + +`agent.system_prompt` is passed to the SDK as `system_instructions`, whose string +shorthand maps to `TemplatedSystemInstructions` — a named section **appended** to +the harness's default system instructions, never a replacement. This matches the +append-only semantics of the shared config field across agents (Claude Code appends +via the `claude_code` preset; Codex via `developer_instructions`). + ### Skills (SKILL.md) Antigravity supports [Agent Skills](https://agentskills.io/specification) diff --git a/docs/agents/CLAUDE_CODE.md b/docs/agents/CLAUDE_CODE.md index 66670709..8b7aab0d 100644 --- a/docs/agents/CLAUDE_CODE.md +++ b/docs/agents/CLAUDE_CODE.md @@ -99,7 +99,8 @@ agent: | `allowed_tools` | `list[str] \| null` | Tool allowlist. Unset ⇒ all tools allowed. | | `disallowed_tools` | `list[str] \| null` | Tool denylist. (`ToolSearch` is always appended for Bedrock parity.) | | `plugins` | `list[{type: local, path}]` | Local plugin/skill directories; `$VAR` in `path` is expanded and resolved to an absolute path. | -| `system_prompt` | `str \| null` | **Replaces** the default system prompt (there is no *append* seam). Mutually exclusive with `system_prompt_file`. | +| `system_prompt` | `str \| null` | **Appended** to the default Claude Code system prompt (via the SDK's `claude_code` preset) — the default's behavioral guidance is always kept, whether or not this is set. Mutually exclusive with `system_prompt_file`. | +| `system_prompt_mode` | `"append"` (default) / `"replace"` | `replace` sends `system_prompt` as the **entire** system prompt (no preset). Used by judge sub-agents, which must not carry the coding-agent persona; rarely needed in tasks. | | `system_prompt_file` | `str \| null` | Path (relative to the task YAML) loaded into `system_prompt` at resolution. | | `setting_sources` | `list["user"\|"project"\|"local"] \| null` | Which host setting sources the SDK reads. Default resolves to `["project"]`. See [Sandbox isolation](#sandbox-isolation). | | `claude_settings` | `str \| dict \| null` | Passed to the SDK `--settings`. A dict is JSON-serialized; a str is a settings file path. Use `permissions.deny` to block tools/paths. | @@ -111,6 +112,17 @@ agent: > `setting_sources`, `include_partial_messages`, …) are rejected there — set those > through their typed fields or `-D run_limits.*`. MCP servers are not a YAML field. +> **System-prompt reproducibility.** In `append` mode the preset's *dynamic +> sections* (working directory, git status, auto-memory) are excluded so the system +> prompt stays identical across runs — the per-run sandbox tempdir path would +> otherwise be baked into it, breaking prompt caching and run comparability. The +> SDK re-injects the stripped content into the first user message, so the agent +> loses nothing. Note the default-prompt baseline tracks the installed Claude Code +> CLI version; `environment_info.claude_code_cli` in `run.json` records which +> version a run used, and `environment_info.system_prompt_semantics` +> (`append` / `replace`) records the prompt regime — runs predating that marker +> used replace-on-set / empty-on-unset semantics and are not score-comparable. + ### Setting fields from the CLI Any of these merge-resolve through `-D` / `--set` (see diff --git a/docs/agents/CODEX.md b/docs/agents/CODEX.md index b020e76b..7b6dc99f 100644 --- a/docs/agents/CODEX.md +++ b/docs/agents/CODEX.md @@ -213,6 +213,7 @@ The Codex SDK is synchronous. The agent uses `_run_async()` helper to detect and | **SDK Type** | Subprocess (CLI via JSON generator) | Sync client (app-server subprocess) | | **Command Tracking** | Full telemetry (tool name, params, duration) | Streamed telemetry: shell → `Bash`, apply_patch → `Write` | | **Model Selection** | Direct via `--model` or config | `agent.model` pinned into `thread_start` | +| **System prompt** | `system_prompt` appended to the default prompt (SDK `claude_code` preset) | `system_prompt` passed as `developer_instructions` on top of the Codex base prompt | | **Session Resume** | `--resume {session_id}` | Via thread ID | | **Permissions** | `permission_mode` + `allowed_tools` | `permission_mode` → sandbox/approval + `allowed_tools`/`disallowed_tools` → thread config | | **Tool Enforcement** | Not enforced by Coder Eval wrapper | `enabled_tools` honored; `disabled_tools` NOT enforced by the SDK | diff --git a/src/coder_eval/agents/antigravity_agent.py b/src/coder_eval/agents/antigravity_agent.py index 4e1475be..9c1d58ce 100644 --- a/src/coder_eval/agents/antigravity_agent.py +++ b/src/coder_eval/agents/antigravity_agent.py @@ -561,6 +561,9 @@ def get_environment_info(self) -> dict[str, Any]: return { "antigravity_model": self._effective_model(), "antigravity_thinking_level": self.config.thinking_level, + # Antigravity has always appended (TemplatedSystemInstructions); + # emitted for cross-agent uniformity of the marker. + "system_prompt_semantics": "append", } def _conversation_or_none(self) -> Any: diff --git a/src/coder_eval/agents/claude_code_agent.py b/src/coder_eval/agents/claude_code_agent.py index 71cb2267..3671fa7f 100644 --- a/src/coder_eval/agents/claude_code_agent.py +++ b/src/coder_eval/agents/claude_code_agent.py @@ -28,6 +28,10 @@ # its kill target and timeouts will no longer be enforced at the agent layer. from claude_agent_sdk._internal.transport.subprocess_cli import SubprocessCLITransport +# SystemPromptPreset is not re-exported from the SDK root, so claude_agent_sdk.types +# is the only import route (same treatment as evaluation/verdict_tool.py). +from claude_agent_sdk.types import SystemPromptPreset + from coder_eval.agent import Agent, AgentState from coder_eval.agents._logging import PrefixedAdapter, log_raw_sdk_event from coder_eval.agents.registry import AgentRegistry @@ -1173,6 +1177,25 @@ def _build_claude_query( if "ToolSearch" not in disallowed_tools: disallowed_tools.append("ToolSearch") + # The SDK maps system_prompt=None to `--system-prompt ""` (an explicit + # EMPTY custom prompt) and a plain string to a full replacement — either + # way Claude Code's default behavioral guidance (parallel tool-call + # batching, conciseness) is lost. So ALWAYS send the claude_code preset: + # without `append` the CLI runs its default prompt; with it the configured + # prompt is appended. exclude_dynamic_sections keeps the prompt static + # across runs (the per-run tempdir path would otherwise be baked into the + # system prompt, breaking prompt caching and run comparability); the SDK + # re-injects the stripped sections into the first user message. + # system_prompt_mode="replace" (judge sub-agents) opts out of the preset: + # the configured prompt IS the entire system prompt. + system_prompt: str | SystemPromptPreset + if self.config.system_prompt_mode == "replace" and self.config.system_prompt is not None: + system_prompt = self.config.system_prompt + else: + system_prompt = SystemPromptPreset(type="preset", preset="claude_code", exclude_dynamic_sections=True) + if self.config.system_prompt is not None: + system_prompt["append"] = self.config.system_prompt + # as_posix(), not str(): bash on Windows strips backslashes from unquoted # paths, so a redirect like `> D:\foo\bar` ends up writing to "Dfoobar". options = ClaudeAgentOptions( @@ -1192,7 +1215,7 @@ def _build_claude_query( # summing per-message values undercounts by 10x+. Without this flag # StreamEvents are suppressed by the SDK. include_partial_messages=True, - system_prompt=self.config.system_prompt, + system_prompt=system_prompt, setting_sources=self.config.setting_sources if self.config.setting_sources is not None else ["project"], resume=self._session_id, settings=json.dumps(self.config.claude_settings) @@ -1216,6 +1239,17 @@ def _build_claude_query( return options, transport, effective_model + def get_environment_info(self) -> dict[str, Any]: + """Record which system-prompt regime built this run's prompts. + + ``append`` = the claude_code preset (dynamic sections excluded) with the + configured system_prompt, if any, appended; ``replace`` = the configured + prompt is the ENTIRE system prompt (judge sub-agents). Runs from before + this marker existed used replace-on-set / empty-on-unset semantics — + trend dashboards must not pool scores across that boundary. + """ + return {"system_prompt_semantics": self.config.system_prompt_mode} + async def stop(self) -> None: """Stop the agent and clean up resources.""" self.client = None diff --git a/src/coder_eval/agents/codex_agent.py b/src/coder_eval/agents/codex_agent.py index c6e4b2d3..e3f40f7d 100644 --- a/src/coder_eval/agents/codex_agent.py +++ b/src/coder_eval/agents/codex_agent.py @@ -944,10 +944,16 @@ def get_environment_info(self) -> dict[str, Any]: recorded to avoid leaking any embedded credentials; the API key is never recorded. """ + # system_prompt_semantics: Codex appends system_prompt as + # developer_instructions on top of its base prompt. Runs from before this + # marker existed silently DROPPED the field — dashboards must not pool + # system_prompt-setting tasks across that boundary. + info: dict[str, Any] = {"system_prompt_semantics": "append"} base_url = self._resolve_base_url() if not base_url: - return {} + return info return { + **info, "codex_base_url_host": urlparse(base_url).hostname or "", "codex_wire_api": _CODEX_WIRE_API, "codex_api_version": self._resolve_api_version() or "", @@ -1276,6 +1282,14 @@ def _build_thread_options(self) -> dict[str, Any]: options["model"] = effective_model self._log.debug(f"Codex model pinned to {effective_model}") + # system_prompt maps to developer_instructions: injected ON TOP of Codex's + # base prompt, matching the append-only contract of the shared config field + # (Claude Code appends via the claude_code preset; Antigravity via + # TemplatedSystemInstructions). base_instructions (full replacement of the + # base prompt) is deliberately not exposed. + if self.config.system_prompt is not None: + options["developer_instructions"] = self.config.system_prompt + permission_mode = self.config.permission_mode.value approval_mode_str = _CODEX_APPROVAL_MODE diff --git a/src/coder_eval/criteria/agent_judge.py b/src/coder_eval/criteria/agent_judge.py index 20db8dab..bbbc6e48 100644 --- a/src/coder_eval/criteria/agent_judge.py +++ b/src/coder_eval/criteria/agent_judge.py @@ -63,6 +63,11 @@ logger = logging.getLogger(__name__) +# This is the judge's ENTIRE identity: _build_agent_config forces +# system_prompt_mode="replace" so the Claude Code coding-agent preset never +# reaches the scoring instrument — the judge must not carry an engineering +# persona (terse, proactively edits files) ahead of its grading role, and its +# verdicts must not shift when the preset does. _SYSTEM_PROMPT = """\ You are a strict code reviewer evaluating a project generated by a coding agent. @@ -263,6 +268,10 @@ def _build_agent_config( user_overrides["sdk_options"] = {**defaults.sdk_options, **user_overrides["sdk_options"]} config = defaults.model_copy(update=user_overrides, deep=True) config.system_prompt = system_prompt + # Force replace regardless of user YAML: the judge prompt is its entire + # identity — the coding-agent preset must never prefix the scoring + # instrument (see the note on _SYSTEM_PROMPT). + config.system_prompt_mode = "replace" # SECURITY: force setting_sources=[] regardless of user YAML so the SDK # does NOT load .claude/settings.json or .mcp.json from the judge's cwd. # Those files can install pre-LLM lifecycle hooks (SessionStart / diff --git a/src/coder_eval/models/agent_config.py b/src/coder_eval/models/agent_config.py index b4ad98fd..721997c1 100644 --- a/src/coder_eval/models/agent_config.py +++ b/src/coder_eval/models/agent_config.py @@ -151,7 +151,8 @@ class BaseAgentConfig(BaseModel): system_prompt: str | None = Field( default=None, description=( - "Custom system prompt. Replaces the default system prompt. " + "Custom system prompt, appended to the agent's default system prompt — never a replacement. " + "Each agent's doc page (docs/agents/) states the exact mechanism. " "Supports inline text or multi-line YAML strings. " "Mutually exclusive with system_prompt_file." ), @@ -197,6 +198,15 @@ class ClaudeCodeAgentConfig(BaseAgentConfig): type: Literal[AgentKind.CLAUDE_CODE] # type: ignore[assignment] + system_prompt_mode: Literal["append", "replace"] = Field( + default="append", + description=( + "How system_prompt combines with the Claude Code default prompt: 'append' layers it " + "after the SDK 'claude_code' preset, keeping the default's behavioral guidance; " + "'replace' sends it as the ENTIRE system prompt. Judge sub-agents force 'replace' so " + "the scoring instrument never carries the coding-agent persona." + ), + ) claude_settings: str | dict[str, Any] | None = MergeField( strategy="deep", default=None, diff --git a/tests/test_agent.py b/tests/test_agent.py index 2e4f7aaa..1005a8e9 100644 --- a/tests/test_agent.py +++ b/tests/test_agent.py @@ -377,6 +377,100 @@ async def test_claude_settings_none_default(): assert captured_options[0].settings is None +def _transport_command(options) -> list[str]: + """Render captured ClaudeAgentOptions into the actual CLI argv. + + The dict-shape assertions pin the values we set; this pins the SDK contract + (which flag the transport emits) — the surface the original replace-vs-append + bug lived on — and survives an SDK TypedDict reshape. + """ + from claude_agent_sdk._internal.transport.subprocess_cli import SubprocessCLITransport + + options.cli_path = "claude" + return SubprocessCLITransport(prompt="x", options=options)._build_command() + + +@pytest.mark.asyncio +async def test_system_prompt_appends_to_claude_code_preset(): + """system_prompt keeps the Claude Code default prompt and appends via the SDK preset.""" + config = parse_agent_config(type=AgentKind.CLAUDE_CODE, system_prompt="You are a coding agent.") + agent = ClaudeCodeAgent(config) + + captured_options = await _capture_sdk_options(agent) + + assert captured_options[0].system_prompt == { + "type": "preset", + "preset": "claude_code", + "exclude_dynamic_sections": True, + "append": "You are a coding agent.", + } + cmd = _transport_command(captured_options[0]) + assert "--append-system-prompt" in cmd + assert "--system-prompt" not in cmd + + +@pytest.mark.asyncio +async def test_system_prompt_unset_sends_bare_preset(): + """No system_prompt -> the bare claude_code preset, which the transport renders + as NO system-prompt flag (the CLI default). Passing None instead would emit + `--system-prompt \"\"` — an explicit EMPTY prompt that loses the default.""" + config = parse_agent_config(type=AgentKind.CLAUDE_CODE) + agent = ClaudeCodeAgent(config) + + captured_options = await _capture_sdk_options(agent) + + assert captured_options[0].system_prompt == { + "type": "preset", + "preset": "claude_code", + "exclude_dynamic_sections": True, + } + cmd = _transport_command(captured_options[0]) + assert "--append-system-prompt" not in cmd + assert "--system-prompt" not in cmd + + +@pytest.mark.asyncio +async def test_system_prompt_empty_string_appends_empty(): + """system_prompt: \"\" is configured, not unset — it appends (harmlessly), and a + future truthiness refactor must not route it into the preset-loss path.""" + config = parse_agent_config(type=AgentKind.CLAUDE_CODE, system_prompt="") + agent = ClaudeCodeAgent(config) + + captured_options = await _capture_sdk_options(agent) + + assert captured_options[0].system_prompt["append"] == "" + + +@pytest.mark.asyncio +async def test_system_prompt_mode_replace_sends_plain_string(): + """system_prompt_mode='replace' (the judge seam) sends the configured prompt as + the ENTIRE system prompt — no preset, no coding-agent persona.""" + config = parse_agent_config( + type=AgentKind.CLAUDE_CODE, system_prompt="You are a strict grader.", system_prompt_mode="replace" + ) + agent = ClaudeCodeAgent(config) + + captured_options = await _capture_sdk_options(agent) + + assert captured_options[0].system_prompt == "You are a strict grader." + cmd = _transport_command(captured_options[0]) + assert "--system-prompt" in cmd + assert "--append-system-prompt" not in cmd + + +def test_environment_info_reports_system_prompt_semantics(): + """The resolved system_prompt_mode lands in run.json (environment_info) so + trend dashboards can segment runs by prompt regime instead of pooling + pre-/post-append-semantics scores.""" + default_agent = ClaudeCodeAgent(parse_agent_config(type=AgentKind.CLAUDE_CODE)) + assert default_agent.get_environment_info() == {"system_prompt_semantics": "append"} + + judge_like = ClaudeCodeAgent( + parse_agent_config(type=AgentKind.CLAUDE_CODE, system_prompt="grader", system_prompt_mode="replace") + ) + assert judge_like.get_environment_info() == {"system_prompt_semantics": "replace"} + + @pytest.mark.asyncio async def test_sdk_options_forwarded_to_sdk(): """An sdk_options key (e.g. effort) is splatted into ClaudeAgentOptions.""" diff --git a/tests/test_agent_judge_criterion.py b/tests/test_agent_judge_criterion.py index cfd6b720..593764bc 100644 --- a/tests/test_agent_judge_criterion.py +++ b/tests/test_agent_judge_criterion.py @@ -715,6 +715,22 @@ def test_agent_judge_prompt_requires_findings(sandbox: Sandbox, direct_route: Di assert "findings" in user_msg.lower() +def test_agent_judge_system_prompt_replaces_not_appends(sandbox: Sandbox, direct_route: DirectRoute) -> None: + """The judge prompt is its ENTIRE identity: system_prompt_mode must be 'replace' + so the Claude Code coding-agent preset never prefixes the scoring instrument — + forced even when the user's YAML says 'append'.""" + criterion = AgentJudgeCriterion( + description="x", prompt="grade", agent={"type": "claude-code", "system_prompt_mode": "append"} + ) + mock_agent = _make_mock_agent('{"score": 0.5, "rationale": "ok"}') + with patch(_AGENT_PATCH_PATH, return_value=mock_agent) as mock_cls: + SuccessChecker(sandbox, init_registry=False, route=direct_route).check(criterion) + + (agent_config,) = mock_cls.call_args.args + assert agent_config.system_prompt_mode == "replace" + assert agent_config.system_prompt.startswith("You are a strict code reviewer") + + def test_agent_judge_transcript_captures_tool_calls(sandbox: Sandbox, direct_route: DirectRoute) -> None: """Tool calls made by the judge sub-agent must surface on the transcript so reviewers can audit the verdict.""" diff --git a/tests/test_antigravity_agent.py b/tests/test_antigravity_agent.py index 99747950..0f989b07 100644 --- a/tests/test_antigravity_agent.py +++ b/tests/test_antigravity_agent.py @@ -64,6 +64,14 @@ def test_effective_model_prefers_config_then_default(): assert unpinned._effective_model() == _DEFAULT_MODEL +def test_environment_info_reports_append_prompt_semantics(): + """Antigravity always appends system_prompt (TemplatedSystemInstructions); + the cross-agent marker in run.json records that regime.""" + agent = AntigravityAgent(parse_agent_config(type="antigravity")) + + assert agent.get_environment_info()["system_prompt_semantics"] == "append" + + def _make_skill(parent, name: str) -> None: d = parent / name d.mkdir(parents=True) diff --git a/tests/test_codex_agent.py b/tests/test_codex_agent.py index 95ee117f..33690b0a 100644 --- a/tests/test_codex_agent.py +++ b/tests/test_codex_agent.py @@ -125,6 +125,22 @@ def test_sandbox_is_full_access(self, monkeypatch, mode, in_container, os_name): assert agent._build_thread_options()["sandbox"] == Sandbox("full-access") +class TestSystemPrompt: + """system_prompt travels as developer_instructions — injected on top of Codex's + base prompt, mirroring the append-only semantics of the other agents.""" + + def test_system_prompt_forwarded_as_developer_instructions(self): + agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX, system_prompt="You are a coding agent.")) + + assert agent._build_thread_options()["developer_instructions"] == "You are a coding agent." + + def test_no_system_prompt_omits_developer_instructions(self): + """No system_prompt -> the key is absent, leaving the SDK default untouched.""" + agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX)) + + assert "developer_instructions" not in agent._build_thread_options() + + class TestCodexEnvironmentConfiguration: """Test _build_codex_env: only CODEX_API_KEY travels via env.""" @@ -315,10 +331,12 @@ def test_empty_api_version_falls_back(self, monkeypatch): class TestCodexEnvironmentInfo: """get_environment_info surfaces resolved custom-endpoint routing for run artifacts.""" - def test_no_base_url_emits_nothing(self, monkeypatch): + def test_no_base_url_emits_only_prompt_semantics(self, monkeypatch): + """Without a custom endpoint, only the cross-agent system-prompt marker is + emitted (Codex appends system_prompt as developer_instructions).""" monkeypatch.delenv("CODEX_BASE_URL", raising=False) agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX, model="gpt-5-codex")) - assert agent.get_environment_info() == {} + assert agent.get_environment_info() == {"system_prompt_semantics": "append"} def test_azure_routing_recorded(self, monkeypatch): """Host (not full URL), wire_api, api-version, and the deployment-name marker @@ -329,6 +347,7 @@ def test_azure_routing_recorded(self, monkeypatch): agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX, model="my-gpt5-deployment")) info = agent.get_environment_info() assert info == { + "system_prompt_semantics": "append", "codex_base_url_host": "my-res.openai.azure.com", "codex_wire_api": "responses", "codex_api_version": "2025-04-01-preview",