From 5fc99fd09eaa4078e0f7956977f3a4e66d12c581 Mon Sep 17 00:00:00 2001 From: Mihai Chirculescu Date: Fri, 7 Aug 2026 16:30:23 +0300 Subject: [PATCH 1/6] fix(agent): append system_prompt to the Claude Code preset instead of replacing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A plain-string ClaudeAgentOptions.system_prompt replaces Claude Code's entire default system prompt. Every experiment that sets even a one-line system_prompt silently strips the harness's behavioral guidance — observed in skills nightly runs as zero parallel tool calls (the batching instruction lives in the default prompt), heavy narration, and raw cat/sed over Read/Grep. Wrap the configured prompt in the SDK's claude_code preset with append so the default prompt survives. Co-Authored-By: Claude Fable 5 --- src/coder_eval/agents/claude_code_agent.py | 11 ++++++++- src/coder_eval/models/agent_config.py | 3 ++- tests/test_agent.py | 26 ++++++++++++++++++++++ 3 files changed, 38 insertions(+), 2 deletions(-) diff --git a/src/coder_eval/agents/claude_code_agent.py b/src/coder_eval/agents/claude_code_agent.py index 71cb2267..cfd5bc9a 100644 --- a/src/coder_eval/agents/claude_code_agent.py +++ b/src/coder_eval/agents/claude_code_agent.py @@ -20,6 +20,7 @@ TaskNotificationMessage, query, ) +from claude_agent_sdk.types import SystemPromptPreset # Private SDK import — the public `query()` API doesn't expose the subprocess # handle, but we need it to SIGKILL on timeout (the SDK's anyio task groups @@ -1173,6 +1174,14 @@ def _build_claude_query( if "ToolSearch" not in disallowed_tools: disallowed_tools.append("ToolSearch") + # A plain-string system_prompt would REPLACE Claude Code's default system + # prompt, dropping its behavioral guidance (parallel tool-call batching, + # conciseness). Always keep the default via the SDK preset and append the + # configured prompt after it. + system_prompt: SystemPromptPreset | None = None + if self.config.system_prompt is not None: + system_prompt = SystemPromptPreset(type="preset", preset="claude_code", append=self.config.system_prompt) + # as_posix(), not str(): bash on Windows strips backslashes from unquoted # paths, so a redirect like `> D:\foo\bar` ends up writing to "Dfoobar". options = ClaudeAgentOptions( @@ -1192,7 +1201,7 @@ def _build_claude_query( # summing per-message values undercounts by 10x+. Without this flag # StreamEvents are suppressed by the SDK. include_partial_messages=True, - system_prompt=self.config.system_prompt, + system_prompt=system_prompt, setting_sources=self.config.setting_sources if self.config.setting_sources is not None else ["project"], resume=self._session_id, settings=json.dumps(self.config.claude_settings) diff --git a/src/coder_eval/models/agent_config.py b/src/coder_eval/models/agent_config.py index b4ad98fd..b35d286d 100644 --- a/src/coder_eval/models/agent_config.py +++ b/src/coder_eval/models/agent_config.py @@ -151,7 +151,8 @@ class BaseAgentConfig(BaseModel): system_prompt: str | None = Field( default=None, description=( - "Custom system prompt. Replaces the default system prompt. " + "Custom system prompt, appended to the agent's default system prompt " + "(claude-code: the SDK 'claude_code' preset with append). " "Supports inline text or multi-line YAML strings. " "Mutually exclusive with system_prompt_file." ), diff --git a/tests/test_agent.py b/tests/test_agent.py index 2e4f7aaa..405d2996 100644 --- a/tests/test_agent.py +++ b/tests/test_agent.py @@ -377,6 +377,32 @@ async def test_claude_settings_none_default(): assert captured_options[0].settings is None +@pytest.mark.asyncio +async def test_system_prompt_appends_to_claude_code_preset(): + """system_prompt keeps the Claude Code default prompt and appends via the SDK preset.""" + config = parse_agent_config(type=AgentKind.CLAUDE_CODE, system_prompt="You are a coding agent.") + agent = ClaudeCodeAgent(config) + + captured_options = await _capture_sdk_options(agent) + + assert captured_options[0].system_prompt == { + "type": "preset", + "preset": "claude_code", + "append": "You are a coding agent.", + } + + +@pytest.mark.asyncio +async def test_system_prompt_none_leaves_sdk_default(): + """No system_prompt -> ClaudeAgentOptions.system_prompt stays None (SDK default prompt).""" + config = parse_agent_config(type=AgentKind.CLAUDE_CODE) + agent = ClaudeCodeAgent(config) + + captured_options = await _capture_sdk_options(agent) + + assert captured_options[0].system_prompt is None + + @pytest.mark.asyncio async def test_sdk_options_forwarded_to_sdk(): """An sdk_options key (e.g. effort) is splatted into ClaudeAgentOptions.""" From f0668343cb7d300c5c68a34ed59a31b3a0bd190e Mon Sep 17 00:00:00 2001 From: Mihai Chirculescu Date: Fri, 7 Aug 2026 16:35:15 +0300 Subject: [PATCH 2/6] style: sort claude_agent_sdk.types import Co-Authored-By: Claude Fable 5 --- src/coder_eval/agents/claude_code_agent.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/coder_eval/agents/claude_code_agent.py b/src/coder_eval/agents/claude_code_agent.py index cfd5bc9a..208cc574 100644 --- a/src/coder_eval/agents/claude_code_agent.py +++ b/src/coder_eval/agents/claude_code_agent.py @@ -20,7 +20,6 @@ TaskNotificationMessage, query, ) -from claude_agent_sdk.types import SystemPromptPreset # Private SDK import — the public `query()` API doesn't expose the subprocess # handle, but we need it to SIGKILL on timeout (the SDK's anyio task groups @@ -28,6 +27,7 @@ # CLI). If this import breaks on an SDK upgrade, the threaded watchdog loses # its kill target and timeouts will no longer be enforced at the agent layer. from claude_agent_sdk._internal.transport.subprocess_cli import SubprocessCLITransport +from claude_agent_sdk.types import SystemPromptPreset from coder_eval.agent import Agent, AgentState from coder_eval.agents._logging import PrefixedAdapter, log_raw_sdk_event From b0113d36bb7782431f7e0a655b33cce9a7e8c19f Mon Sep 17 00:00:00 2001 From: Mihai Chirculescu Date: Sat, 8 Aug 2026 09:36:53 +0300 Subject: [PATCH 3/6] fix(agent): align Codex system_prompt with the append-only contract CodexAgent silently dropped config.system_prompt; forward it as developer_instructions (injected on top of the Codex base prompt) to match the append semantics of Claude Code (claude_code preset) and Antigravity (TemplatedSystemInstructions, which already appended). Also document the ripple effects of append-only system_prompt: - agent_judge: the reviewer prompt is now layered after the full Claude Code preset instead of replacing it (accepted trade-off, noted in code) - BaseAgentConfig.system_prompt description states per-agent semantics - docs: fix the stale "Replaces the default" claim in CLAUDE_CODE.md, add a System prompt row to CODEX.md, document Antigravity's append shorthand Co-Authored-By: Claude Fable 5 --- docs/agents/ANTIGRAVITY.md | 8 ++++++++ docs/agents/CLAUDE_CODE.md | 2 +- docs/agents/CODEX.md | 1 + src/coder_eval/agents/codex_agent.py | 8 ++++++++ src/coder_eval/criteria/agent_judge.py | 6 ++++++ src/coder_eval/models/agent_config.py | 5 +++-- tests/test_codex_agent.py | 16 ++++++++++++++++ 7 files changed, 43 insertions(+), 3 deletions(-) diff --git a/docs/agents/ANTIGRAVITY.md b/docs/agents/ANTIGRAVITY.md index 522dc8bf..ad96081b 100644 --- a/docs/agents/ANTIGRAVITY.md +++ b/docs/agents/ANTIGRAVITY.md @@ -109,6 +109,14 @@ Antigravity exposes a `thinking_level` field (`minimal` / `low` / `medium` / Antigravity-specific — Claude Code and Codex don't take this field. Thinking tokens are billed as **output** tokens (see [Telemetry](#telemetry)). +### `system_prompt` + +`agent.system_prompt` is passed to the SDK as `system_instructions`, whose string +shorthand maps to `TemplatedSystemInstructions` — a named section **appended** to +the harness's default system instructions, never a replacement. This matches the +append-only semantics of the shared config field across agents (Claude Code appends +via the `claude_code` preset; Codex via `developer_instructions`). + ### Skills (SKILL.md) Antigravity supports [Agent Skills](https://agentskills.io/specification) diff --git a/docs/agents/CLAUDE_CODE.md b/docs/agents/CLAUDE_CODE.md index 66670709..5614b499 100644 --- a/docs/agents/CLAUDE_CODE.md +++ b/docs/agents/CLAUDE_CODE.md @@ -99,7 +99,7 @@ agent: | `allowed_tools` | `list[str] \| null` | Tool allowlist. Unset ⇒ all tools allowed. | | `disallowed_tools` | `list[str] \| null` | Tool denylist. (`ToolSearch` is always appended for Bedrock parity.) | | `plugins` | `list[{type: local, path}]` | Local plugin/skill directories; `$VAR` in `path` is expanded and resolved to an absolute path. | -| `system_prompt` | `str \| null` | **Replaces** the default system prompt (there is no *append* seam). Mutually exclusive with `system_prompt_file`. | +| `system_prompt` | `str \| null` | **Appended** to the default Claude Code system prompt (via the SDK's `claude_code` preset with `append`) — the default's behavioral guidance is always kept. Mutually exclusive with `system_prompt_file`. | | `system_prompt_file` | `str \| null` | Path (relative to the task YAML) loaded into `system_prompt` at resolution. | | `setting_sources` | `list["user"\|"project"\|"local"] \| null` | Which host setting sources the SDK reads. Default resolves to `["project"]`. See [Sandbox isolation](#sandbox-isolation). | | `claude_settings` | `str \| dict \| null` | Passed to the SDK `--settings`. A dict is JSON-serialized; a str is a settings file path. Use `permissions.deny` to block tools/paths. | diff --git a/docs/agents/CODEX.md b/docs/agents/CODEX.md index b020e76b..7b6dc99f 100644 --- a/docs/agents/CODEX.md +++ b/docs/agents/CODEX.md @@ -213,6 +213,7 @@ The Codex SDK is synchronous. The agent uses `_run_async()` helper to detect and | **SDK Type** | Subprocess (CLI via JSON generator) | Sync client (app-server subprocess) | | **Command Tracking** | Full telemetry (tool name, params, duration) | Streamed telemetry: shell → `Bash`, apply_patch → `Write` | | **Model Selection** | Direct via `--model` or config | `agent.model` pinned into `thread_start` | +| **System prompt** | `system_prompt` appended to the default prompt (SDK `claude_code` preset) | `system_prompt` passed as `developer_instructions` on top of the Codex base prompt | | **Session Resume** | `--resume {session_id}` | Via thread ID | | **Permissions** | `permission_mode` + `allowed_tools` | `permission_mode` → sandbox/approval + `allowed_tools`/`disallowed_tools` → thread config | | **Tool Enforcement** | Not enforced by Coder Eval wrapper | `enabled_tools` honored; `disabled_tools` NOT enforced by the SDK | diff --git a/src/coder_eval/agents/codex_agent.py b/src/coder_eval/agents/codex_agent.py index c6e4b2d3..4b3af0d0 100644 --- a/src/coder_eval/agents/codex_agent.py +++ b/src/coder_eval/agents/codex_agent.py @@ -1276,6 +1276,14 @@ def _build_thread_options(self) -> dict[str, Any]: options["model"] = effective_model self._log.debug(f"Codex model pinned to {effective_model}") + # system_prompt maps to developer_instructions: injected ON TOP of Codex's + # base prompt, matching the append-only contract of the shared config field + # (Claude Code appends via the claude_code preset; Antigravity via + # TemplatedSystemInstructions). base_instructions (full replacement of the + # base prompt) is deliberately not exposed. + if self.config.system_prompt is not None: + options["developer_instructions"] = self.config.system_prompt + permission_mode = self.config.permission_mode.value approval_mode_str = _CODEX_APPROVAL_MODE diff --git a/src/coder_eval/criteria/agent_judge.py b/src/coder_eval/criteria/agent_judge.py index 20db8dab..3af1fea4 100644 --- a/src/coder_eval/criteria/agent_judge.py +++ b/src/coder_eval/criteria/agent_judge.py @@ -63,6 +63,12 @@ logger = logging.getLogger(__name__) +# system_prompt is append-only (see BaseAgentConfig.system_prompt), so this is +# layered AFTER the full Claude Code preset rather than replacing it: the judge +# carries the coding-agent identity plus this reviewer role, and pays the +# preset's prompt tokens on every call. Accepted trade-off — the preset's +# tool-usage guidance helps the investigation, and the submit_verdict contract +# below still governs the output. _SYSTEM_PROMPT = """\ You are a strict code reviewer evaluating a project generated by a coding agent. diff --git a/src/coder_eval/models/agent_config.py b/src/coder_eval/models/agent_config.py index b35d286d..8f389798 100644 --- a/src/coder_eval/models/agent_config.py +++ b/src/coder_eval/models/agent_config.py @@ -151,8 +151,9 @@ class BaseAgentConfig(BaseModel): system_prompt: str | None = Field( default=None, description=( - "Custom system prompt, appended to the agent's default system prompt " - "(claude-code: the SDK 'claude_code' preset with append). " + "Custom system prompt, appended to the agent's default system prompt — never a replacement " + "(claude-code: the SDK 'claude_code' preset with append; codex: developer_instructions " + "on top of the base prompt; antigravity: TemplatedSystemInstructions sections). " "Supports inline text or multi-line YAML strings. " "Mutually exclusive with system_prompt_file." ), diff --git a/tests/test_codex_agent.py b/tests/test_codex_agent.py index 95ee117f..f9f009dc 100644 --- a/tests/test_codex_agent.py +++ b/tests/test_codex_agent.py @@ -125,6 +125,22 @@ def test_sandbox_is_full_access(self, monkeypatch, mode, in_container, os_name): assert agent._build_thread_options()["sandbox"] == Sandbox("full-access") +class TestSystemPrompt: + """system_prompt travels as developer_instructions — injected on top of Codex's + base prompt, mirroring the append-only semantics of the other agents.""" + + def test_system_prompt_forwarded_as_developer_instructions(self): + agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX, system_prompt="You are a coding agent.")) + + assert agent._build_thread_options()["developer_instructions"] == "You are a coding agent." + + def test_no_system_prompt_omits_developer_instructions(self): + """No system_prompt -> the key is absent, leaving the SDK default untouched.""" + agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX)) + + assert "developer_instructions" not in agent._build_thread_options() + + class TestCodexEnvironmentConfiguration: """Test _build_codex_env: only CODEX_API_KEY travels via env.""" From 92ebce9b96ba612053bb734b23614bb9b2e6d871 Mon Sep 17 00:00:00 2001 From: Mihai Chirculescu Date: Sat, 8 Aug 2026 10:00:00 +0300 Subject: [PATCH 4/6] =?UTF-8?q?fix(agent):=20address=20PR=20review=20?= =?UTF-8?q?=E2=80=94=20unconditional=20preset,=20judge=20replace=20seam?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Blockers from the PR #92 review: - system_prompt unset no longer loses the preset: the SDK maps None to --system-prompt "" (an explicit EMPTY prompt), so _build_options now always sends the claude_code preset — bare (CLI default prompt) when unset, with `append` when configured. This fixes the common no- system_prompt case, which previously ran without Claude Code's default behavioral guidance. - agent_judge no longer inherits the coding-agent preset: new ClaudeCodeAgentConfig.system_prompt_mode ("append" default / "replace"), forced to "replace" in _build_agent_config next to the existing security floors, so the judge prompt stays its entire identity and verdicts can't shift with the preset. Pinned by test. - exclude_dynamic_sections=True on the preset keeps the system prompt static across runs (no per-run tempdir path baked in); the SDK re-injects the stripped sections into the first user message. - Transport-level tests: captured options are rendered through SubprocessCLITransport._build_command() asserting the exact flag emitted (--append-system-prompt vs --system-prompt vs none) — the surface the original bug lived on. Also pins system_prompt: "" and the renamed unset-case test (the old name asserted a false SDK contract). - BaseAgentConfig.system_prompt description is agent-neutral again; the claude-specific mechanism lives on ClaudeCodeAgentConfig + docs/agents/. MIGRATION NOTE: system_prompt semantics on claude-code changed from replace to append, and runs WITHOUT system_prompt now get the real Claude Code default prompt instead of an empty one. Scores are comparable only within one semantics regime — re-baseline judged tasks (e.g. tasks/python_cli_simulated_judged/echo_simulated_judged.yaml, whose prompt was written against replace semantics) and pin runs to the CLI version recorded in environment_info.claude_code_cli. Co-Authored-By: Claude Fable 5 --- docs/agents/CLAUDE_CODE.md | 12 ++++- src/coder_eval/agents/claude_code_agent.py | 28 +++++++--- src/coder_eval/criteria/agent_judge.py | 15 +++--- src/coder_eval/models/agent_config.py | 14 +++-- tests/test_agent.py | 61 ++++++++++++++++++++-- tests/test_agent_judge_criterion.py | 16 ++++++ 6 files changed, 126 insertions(+), 20 deletions(-) diff --git a/docs/agents/CLAUDE_CODE.md b/docs/agents/CLAUDE_CODE.md index 5614b499..c752f15b 100644 --- a/docs/agents/CLAUDE_CODE.md +++ b/docs/agents/CLAUDE_CODE.md @@ -99,7 +99,8 @@ agent: | `allowed_tools` | `list[str] \| null` | Tool allowlist. Unset ⇒ all tools allowed. | | `disallowed_tools` | `list[str] \| null` | Tool denylist. (`ToolSearch` is always appended for Bedrock parity.) | | `plugins` | `list[{type: local, path}]` | Local plugin/skill directories; `$VAR` in `path` is expanded and resolved to an absolute path. | -| `system_prompt` | `str \| null` | **Appended** to the default Claude Code system prompt (via the SDK's `claude_code` preset with `append`) — the default's behavioral guidance is always kept. Mutually exclusive with `system_prompt_file`. | +| `system_prompt` | `str \| null` | **Appended** to the default Claude Code system prompt (via the SDK's `claude_code` preset) — the default's behavioral guidance is always kept, whether or not this is set. Mutually exclusive with `system_prompt_file`. | +| `system_prompt_mode` | `"append"` (default) / `"replace"` | `replace` sends `system_prompt` as the **entire** system prompt (no preset). Used by judge sub-agents, which must not carry the coding-agent persona; rarely needed in tasks. | | `system_prompt_file` | `str \| null` | Path (relative to the task YAML) loaded into `system_prompt` at resolution. | | `setting_sources` | `list["user"\|"project"\|"local"] \| null` | Which host setting sources the SDK reads. Default resolves to `["project"]`. See [Sandbox isolation](#sandbox-isolation). | | `claude_settings` | `str \| dict \| null` | Passed to the SDK `--settings`. A dict is JSON-serialized; a str is a settings file path. Use `permissions.deny` to block tools/paths. | @@ -111,6 +112,15 @@ agent: > `setting_sources`, `include_partial_messages`, …) are rejected there — set those > through their typed fields or `-D run_limits.*`. MCP servers are not a YAML field. +> **System-prompt reproducibility.** In `append` mode the preset's *dynamic +> sections* (working directory, git status, auto-memory) are excluded so the system +> prompt stays identical across runs — the per-run sandbox tempdir path would +> otherwise be baked into it, breaking prompt caching and run comparability. The +> SDK re-injects the stripped content into the first user message, so the agent +> loses nothing. Note the default-prompt baseline tracks the installed Claude Code +> CLI version; `environment_info.claude_code_cli` in `run.json` records which +> version a run used. + ### Setting fields from the CLI Any of these merge-resolve through `-D` / `--set` (see diff --git a/src/coder_eval/agents/claude_code_agent.py b/src/coder_eval/agents/claude_code_agent.py index 208cc574..6dccda46 100644 --- a/src/coder_eval/agents/claude_code_agent.py +++ b/src/coder_eval/agents/claude_code_agent.py @@ -27,6 +27,9 @@ # CLI). If this import breaks on an SDK upgrade, the threaded watchdog loses # its kill target and timeouts will no longer be enforced at the agent layer. from claude_agent_sdk._internal.transport.subprocess_cli import SubprocessCLITransport + +# SystemPromptPreset is not re-exported from the SDK root, so claude_agent_sdk.types +# is the only import route (same treatment as evaluation/verdict_tool.py). from claude_agent_sdk.types import SystemPromptPreset from coder_eval.agent import Agent, AgentState @@ -1174,13 +1177,24 @@ def _build_claude_query( if "ToolSearch" not in disallowed_tools: disallowed_tools.append("ToolSearch") - # A plain-string system_prompt would REPLACE Claude Code's default system - # prompt, dropping its behavioral guidance (parallel tool-call batching, - # conciseness). Always keep the default via the SDK preset and append the - # configured prompt after it. - system_prompt: SystemPromptPreset | None = None - if self.config.system_prompt is not None: - system_prompt = SystemPromptPreset(type="preset", preset="claude_code", append=self.config.system_prompt) + # The SDK maps system_prompt=None to `--system-prompt ""` (an explicit + # EMPTY custom prompt) and a plain string to a full replacement — either + # way Claude Code's default behavioral guidance (parallel tool-call + # batching, conciseness) is lost. So ALWAYS send the claude_code preset: + # without `append` the CLI runs its default prompt; with it the configured + # prompt is appended. exclude_dynamic_sections keeps the prompt static + # across runs (the per-run tempdir path would otherwise be baked into the + # system prompt, breaking prompt caching and run comparability); the SDK + # re-injects the stripped sections into the first user message. + # system_prompt_mode="replace" (judge sub-agents) opts out of the preset: + # the configured prompt IS the entire system prompt. + system_prompt: str | SystemPromptPreset + if self.config.system_prompt_mode == "replace" and self.config.system_prompt is not None: + system_prompt = self.config.system_prompt + else: + system_prompt = SystemPromptPreset(type="preset", preset="claude_code", exclude_dynamic_sections=True) + if self.config.system_prompt is not None: + system_prompt["append"] = self.config.system_prompt # as_posix(), not str(): bash on Windows strips backslashes from unquoted # paths, so a redirect like `> D:\foo\bar` ends up writing to "Dfoobar". diff --git a/src/coder_eval/criteria/agent_judge.py b/src/coder_eval/criteria/agent_judge.py index 3af1fea4..bbbc6e48 100644 --- a/src/coder_eval/criteria/agent_judge.py +++ b/src/coder_eval/criteria/agent_judge.py @@ -63,12 +63,11 @@ logger = logging.getLogger(__name__) -# system_prompt is append-only (see BaseAgentConfig.system_prompt), so this is -# layered AFTER the full Claude Code preset rather than replacing it: the judge -# carries the coding-agent identity plus this reviewer role, and pays the -# preset's prompt tokens on every call. Accepted trade-off — the preset's -# tool-usage guidance helps the investigation, and the submit_verdict contract -# below still governs the output. +# This is the judge's ENTIRE identity: _build_agent_config forces +# system_prompt_mode="replace" so the Claude Code coding-agent preset never +# reaches the scoring instrument — the judge must not carry an engineering +# persona (terse, proactively edits files) ahead of its grading role, and its +# verdicts must not shift when the preset does. _SYSTEM_PROMPT = """\ You are a strict code reviewer evaluating a project generated by a coding agent. @@ -269,6 +268,10 @@ def _build_agent_config( user_overrides["sdk_options"] = {**defaults.sdk_options, **user_overrides["sdk_options"]} config = defaults.model_copy(update=user_overrides, deep=True) config.system_prompt = system_prompt + # Force replace regardless of user YAML: the judge prompt is its entire + # identity — the coding-agent preset must never prefix the scoring + # instrument (see the note on _SYSTEM_PROMPT). + config.system_prompt_mode = "replace" # SECURITY: force setting_sources=[] regardless of user YAML so the SDK # does NOT load .claude/settings.json or .mcp.json from the judge's cwd. # Those files can install pre-LLM lifecycle hooks (SessionStart / diff --git a/src/coder_eval/models/agent_config.py b/src/coder_eval/models/agent_config.py index 8f389798..721997c1 100644 --- a/src/coder_eval/models/agent_config.py +++ b/src/coder_eval/models/agent_config.py @@ -151,9 +151,8 @@ class BaseAgentConfig(BaseModel): system_prompt: str | None = Field( default=None, description=( - "Custom system prompt, appended to the agent's default system prompt — never a replacement " - "(claude-code: the SDK 'claude_code' preset with append; codex: developer_instructions " - "on top of the base prompt; antigravity: TemplatedSystemInstructions sections). " + "Custom system prompt, appended to the agent's default system prompt — never a replacement. " + "Each agent's doc page (docs/agents/) states the exact mechanism. " "Supports inline text or multi-line YAML strings. " "Mutually exclusive with system_prompt_file." ), @@ -199,6 +198,15 @@ class ClaudeCodeAgentConfig(BaseAgentConfig): type: Literal[AgentKind.CLAUDE_CODE] # type: ignore[assignment] + system_prompt_mode: Literal["append", "replace"] = Field( + default="append", + description=( + "How system_prompt combines with the Claude Code default prompt: 'append' layers it " + "after the SDK 'claude_code' preset, keeping the default's behavioral guidance; " + "'replace' sends it as the ENTIRE system prompt. Judge sub-agents force 'replace' so " + "the scoring instrument never carries the coding-agent persona." + ), + ) claude_settings: str | dict[str, Any] | None = MergeField( strategy="deep", default=None, diff --git a/tests/test_agent.py b/tests/test_agent.py index 405d2996..59aa019e 100644 --- a/tests/test_agent.py +++ b/tests/test_agent.py @@ -377,6 +377,19 @@ async def test_claude_settings_none_default(): assert captured_options[0].settings is None +def _transport_command(options) -> list[str]: + """Render captured ClaudeAgentOptions into the actual CLI argv. + + The dict-shape assertions pin the values we set; this pins the SDK contract + (which flag the transport emits) — the surface the original replace-vs-append + bug lived on — and survives an SDK TypedDict reshape. + """ + from claude_agent_sdk._internal.transport.subprocess_cli import SubprocessCLITransport + + options.cli_path = "claude" + return SubprocessCLITransport(prompt="x", options=options)._build_command() + + @pytest.mark.asyncio async def test_system_prompt_appends_to_claude_code_preset(): """system_prompt keeps the Claude Code default prompt and appends via the SDK preset.""" @@ -388,19 +401,61 @@ async def test_system_prompt_appends_to_claude_code_preset(): assert captured_options[0].system_prompt == { "type": "preset", "preset": "claude_code", + "exclude_dynamic_sections": True, "append": "You are a coding agent.", } + cmd = _transport_command(captured_options[0]) + assert "--append-system-prompt" in cmd + assert "--system-prompt" not in cmd @pytest.mark.asyncio -async def test_system_prompt_none_leaves_sdk_default(): - """No system_prompt -> ClaudeAgentOptions.system_prompt stays None (SDK default prompt).""" +async def test_system_prompt_unset_sends_bare_preset(): + """No system_prompt -> the bare claude_code preset, which the transport renders + as NO system-prompt flag (the CLI default). Passing None instead would emit + `--system-prompt \"\"` — an explicit EMPTY prompt that loses the default.""" config = parse_agent_config(type=AgentKind.CLAUDE_CODE) agent = ClaudeCodeAgent(config) captured_options = await _capture_sdk_options(agent) - assert captured_options[0].system_prompt is None + assert captured_options[0].system_prompt == { + "type": "preset", + "preset": "claude_code", + "exclude_dynamic_sections": True, + } + cmd = _transport_command(captured_options[0]) + assert "--append-system-prompt" not in cmd + assert "--system-prompt" not in cmd + + +@pytest.mark.asyncio +async def test_system_prompt_empty_string_appends_empty(): + """system_prompt: \"\" is configured, not unset — it appends (harmlessly), and a + future truthiness refactor must not route it into the preset-loss path.""" + config = parse_agent_config(type=AgentKind.CLAUDE_CODE, system_prompt="") + agent = ClaudeCodeAgent(config) + + captured_options = await _capture_sdk_options(agent) + + assert captured_options[0].system_prompt["append"] == "" + + +@pytest.mark.asyncio +async def test_system_prompt_mode_replace_sends_plain_string(): + """system_prompt_mode='replace' (the judge seam) sends the configured prompt as + the ENTIRE system prompt — no preset, no coding-agent persona.""" + config = parse_agent_config( + type=AgentKind.CLAUDE_CODE, system_prompt="You are a strict grader.", system_prompt_mode="replace" + ) + agent = ClaudeCodeAgent(config) + + captured_options = await _capture_sdk_options(agent) + + assert captured_options[0].system_prompt == "You are a strict grader." + cmd = _transport_command(captured_options[0]) + assert "--system-prompt" in cmd + assert "--append-system-prompt" not in cmd @pytest.mark.asyncio diff --git a/tests/test_agent_judge_criterion.py b/tests/test_agent_judge_criterion.py index cfd6b720..593764bc 100644 --- a/tests/test_agent_judge_criterion.py +++ b/tests/test_agent_judge_criterion.py @@ -715,6 +715,22 @@ def test_agent_judge_prompt_requires_findings(sandbox: Sandbox, direct_route: Di assert "findings" in user_msg.lower() +def test_agent_judge_system_prompt_replaces_not_appends(sandbox: Sandbox, direct_route: DirectRoute) -> None: + """The judge prompt is its ENTIRE identity: system_prompt_mode must be 'replace' + so the Claude Code coding-agent preset never prefixes the scoring instrument — + forced even when the user's YAML says 'append'.""" + criterion = AgentJudgeCriterion( + description="x", prompt="grade", agent={"type": "claude-code", "system_prompt_mode": "append"} + ) + mock_agent = _make_mock_agent('{"score": 0.5, "rationale": "ok"}') + with patch(_AGENT_PATCH_PATH, return_value=mock_agent) as mock_cls: + SuccessChecker(sandbox, init_registry=False, route=direct_route).check(criterion) + + (agent_config,) = mock_cls.call_args.args + assert agent_config.system_prompt_mode == "replace" + assert agent_config.system_prompt.startswith("You are a strict code reviewer") + + def test_agent_judge_transcript_captures_tool_calls(sandbox: Sandbox, direct_route: DirectRoute) -> None: """Tool calls made by the judge sub-agent must surface on the transcript so reviewers can audit the verdict.""" From 3083fc87b1773921ac1a0339a0a41ec10b433b53 Mon Sep 17 00:00:00 2001 From: Mihai Chirculescu Date: Sat, 8 Aug 2026 10:19:50 +0300 Subject: [PATCH 5/6] feat(agent): record system_prompt_semantics marker in environment_info Trend dashboards need to segment runs by system-prompt regime instead of silently pooling pre-/post-append-semantics scores (PR #92 review, cross-run comparability blocker). Each built-in agent now emits system_prompt_semantics via get_environment_info(), merged into run.json: - claude-code: the resolved system_prompt_mode ("append" / "replace") - codex: "append" (developer_instructions; previously the field was silently dropped, so codex runs also cross a semantics boundary here) - antigravity: "append" (unchanged behavior, emitted for uniformity) Runs without the marker predate the change and used replace-on-set / empty-on-unset (claude-code) or dropped (codex) semantics. Co-Authored-By: Claude Fable 5 --- docs/agents/CLAUDE_CODE.md | 4 +++- src/coder_eval/agents/antigravity_agent.py | 3 +++ src/coder_eval/agents/claude_code_agent.py | 11 +++++++++++ src/coder_eval/agents/codex_agent.py | 8 +++++++- tests/test_agent.py | 13 +++++++++++++ tests/test_antigravity_agent.py | 8 ++++++++ tests/test_codex_agent.py | 7 +++++-- 7 files changed, 50 insertions(+), 4 deletions(-) diff --git a/docs/agents/CLAUDE_CODE.md b/docs/agents/CLAUDE_CODE.md index c752f15b..8b7aab0d 100644 --- a/docs/agents/CLAUDE_CODE.md +++ b/docs/agents/CLAUDE_CODE.md @@ -119,7 +119,9 @@ agent: > SDK re-injects the stripped content into the first user message, so the agent > loses nothing. Note the default-prompt baseline tracks the installed Claude Code > CLI version; `environment_info.claude_code_cli` in `run.json` records which -> version a run used. +> version a run used, and `environment_info.system_prompt_semantics` +> (`append` / `replace`) records the prompt regime — runs predating that marker +> used replace-on-set / empty-on-unset semantics and are not score-comparable. ### Setting fields from the CLI diff --git a/src/coder_eval/agents/antigravity_agent.py b/src/coder_eval/agents/antigravity_agent.py index 4e1475be..9c1d58ce 100644 --- a/src/coder_eval/agents/antigravity_agent.py +++ b/src/coder_eval/agents/antigravity_agent.py @@ -561,6 +561,9 @@ def get_environment_info(self) -> dict[str, Any]: return { "antigravity_model": self._effective_model(), "antigravity_thinking_level": self.config.thinking_level, + # Antigravity has always appended (TemplatedSystemInstructions); + # emitted for cross-agent uniformity of the marker. + "system_prompt_semantics": "append", } def _conversation_or_none(self) -> Any: diff --git a/src/coder_eval/agents/claude_code_agent.py b/src/coder_eval/agents/claude_code_agent.py index 6dccda46..3671fa7f 100644 --- a/src/coder_eval/agents/claude_code_agent.py +++ b/src/coder_eval/agents/claude_code_agent.py @@ -1239,6 +1239,17 @@ def _build_claude_query( return options, transport, effective_model + def get_environment_info(self) -> dict[str, Any]: + """Record which system-prompt regime built this run's prompts. + + ``append`` = the claude_code preset (dynamic sections excluded) with the + configured system_prompt, if any, appended; ``replace`` = the configured + prompt is the ENTIRE system prompt (judge sub-agents). Runs from before + this marker existed used replace-on-set / empty-on-unset semantics — + trend dashboards must not pool scores across that boundary. + """ + return {"system_prompt_semantics": self.config.system_prompt_mode} + async def stop(self) -> None: """Stop the agent and clean up resources.""" self.client = None diff --git a/src/coder_eval/agents/codex_agent.py b/src/coder_eval/agents/codex_agent.py index 4b3af0d0..e3f40f7d 100644 --- a/src/coder_eval/agents/codex_agent.py +++ b/src/coder_eval/agents/codex_agent.py @@ -944,10 +944,16 @@ def get_environment_info(self) -> dict[str, Any]: recorded to avoid leaking any embedded credentials; the API key is never recorded. """ + # system_prompt_semantics: Codex appends system_prompt as + # developer_instructions on top of its base prompt. Runs from before this + # marker existed silently DROPPED the field — dashboards must not pool + # system_prompt-setting tasks across that boundary. + info: dict[str, Any] = {"system_prompt_semantics": "append"} base_url = self._resolve_base_url() if not base_url: - return {} + return info return { + **info, "codex_base_url_host": urlparse(base_url).hostname or "", "codex_wire_api": _CODEX_WIRE_API, "codex_api_version": self._resolve_api_version() or "", diff --git a/tests/test_agent.py b/tests/test_agent.py index 59aa019e..1005a8e9 100644 --- a/tests/test_agent.py +++ b/tests/test_agent.py @@ -458,6 +458,19 @@ async def test_system_prompt_mode_replace_sends_plain_string(): assert "--append-system-prompt" not in cmd +def test_environment_info_reports_system_prompt_semantics(): + """The resolved system_prompt_mode lands in run.json (environment_info) so + trend dashboards can segment runs by prompt regime instead of pooling + pre-/post-append-semantics scores.""" + default_agent = ClaudeCodeAgent(parse_agent_config(type=AgentKind.CLAUDE_CODE)) + assert default_agent.get_environment_info() == {"system_prompt_semantics": "append"} + + judge_like = ClaudeCodeAgent( + parse_agent_config(type=AgentKind.CLAUDE_CODE, system_prompt="grader", system_prompt_mode="replace") + ) + assert judge_like.get_environment_info() == {"system_prompt_semantics": "replace"} + + @pytest.mark.asyncio async def test_sdk_options_forwarded_to_sdk(): """An sdk_options key (e.g. effort) is splatted into ClaudeAgentOptions.""" diff --git a/tests/test_antigravity_agent.py b/tests/test_antigravity_agent.py index 99747950..0f989b07 100644 --- a/tests/test_antigravity_agent.py +++ b/tests/test_antigravity_agent.py @@ -64,6 +64,14 @@ def test_effective_model_prefers_config_then_default(): assert unpinned._effective_model() == _DEFAULT_MODEL +def test_environment_info_reports_append_prompt_semantics(): + """Antigravity always appends system_prompt (TemplatedSystemInstructions); + the cross-agent marker in run.json records that regime.""" + agent = AntigravityAgent(parse_agent_config(type="antigravity")) + + assert agent.get_environment_info()["system_prompt_semantics"] == "append" + + def _make_skill(parent, name: str) -> None: d = parent / name d.mkdir(parents=True) diff --git a/tests/test_codex_agent.py b/tests/test_codex_agent.py index f9f009dc..33690b0a 100644 --- a/tests/test_codex_agent.py +++ b/tests/test_codex_agent.py @@ -331,10 +331,12 @@ def test_empty_api_version_falls_back(self, monkeypatch): class TestCodexEnvironmentInfo: """get_environment_info surfaces resolved custom-endpoint routing for run artifacts.""" - def test_no_base_url_emits_nothing(self, monkeypatch): + def test_no_base_url_emits_only_prompt_semantics(self, monkeypatch): + """Without a custom endpoint, only the cross-agent system-prompt marker is + emitted (Codex appends system_prompt as developer_instructions).""" monkeypatch.delenv("CODEX_BASE_URL", raising=False) agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX, model="gpt-5-codex")) - assert agent.get_environment_info() == {} + assert agent.get_environment_info() == {"system_prompt_semantics": "append"} def test_azure_routing_recorded(self, monkeypatch): """Host (not full URL), wire_api, api-version, and the deployment-name marker @@ -345,6 +347,7 @@ def test_azure_routing_recorded(self, monkeypatch): agent = CodexAgent(parse_agent_config(type=AgentKind.CODEX, model="my-gpt5-deployment")) info = agent.get_environment_info() assert info == { + "system_prompt_semantics": "append", "codex_base_url_host": "my-res.openai.azure.com", "codex_wire_api": "responses", "codex_api_version": "2025-04-01-preview", From 0e444f003da9f59ac1cdd8231f4bf6d082b98b82 Mon Sep 17 00:00:00 2001 From: Mihai Chirculescu Date: Mon, 10 Aug 2026 11:21:18 +0300 Subject: [PATCH 6/6] =?UTF-8?q?fix(agent):=20address=20PR=20review=20?= =?UTF-8?q?=E2=80=94=20simulator=20replace=20mode,=20preset-aware=20report?= =?UTF-8?q?s,=20replace=20validator?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Blockers: - UserSimulator now sets system_prompt_mode="replace" so the roleplay persona is the simulator's ENTIRE system prompt (the claude_code coding-agent preset no longer prefixes it on dialog-mode runs), and SubAgentRunner fail-louds on any identity-prompt config left in append mode (mirrors the setting_sources guard). - reports.collect_agent_settings_rows understands the persisted SystemPromptPreset dict: renders the appended prompt text (never the dict repr), omits the row for a bare preset, and surfaces a "System Prompt Mode: replace" row for plain strings. REPORT_SCHEMA.md documents the sdk_options.system_prompt type change and the environment_info.system_prompt_semantics segmentation rule. Non-blocking: - ClaudeCodeAgentConfig rejects system_prompt_mode="replace" with no system_prompt/system_prompt_file; _effective_prompt_mode() is the single source of truth for both the options builder and the system_prompt_semantics marker, so run.json can never disagree with the wire. - Softened the false "never a replacement" prose on BaseAgentConfig.system_prompt. Nits: - Antigravity forwards system_prompt verbatim ("" no longer dropped by `or None`), matching Claude Code / Codex `is not None` semantics. - Corrected the stale Codex get_environment_info docstring and the "always kept" row in docs/agents/CLAUDE_CODE.md; AB_EXPERIMENTS.md lists system_prompt_mode as a variant lever. - Typed the test helpers (_transport_command / _capture_sdk_options), asserted the whole preset dict in the empty-string test, narrowed _transport_command's docstring to argv-pinning only. Tests: simulator replace-mode assertion, SubAgentRunner guard, replace-with-unset-prompt validator case, environment_info merge survival at the orchestrator seam, and report tests fed from a real dump_dataclass(ClaudeAgentOptions). Co-Authored-By: Claude Fable 5 --- docs/AB_EXPERIMENTS.md | 4 +- docs/REPORT_SCHEMA.md | 8 +++ docs/agents/CLAUDE_CODE.md | 4 +- src/coder_eval/agents/antigravity_agent.py | 5 +- src/coder_eval/agents/claude_code_agent.py | 20 +++++- src/coder_eval/agents/codex_agent.py | 14 ++-- src/coder_eval/evaluation/sub_agent.py | 11 +++ src/coder_eval/models/agent_config.py | 20 +++++- src/coder_eval/reports.py | 25 ++++++- src/coder_eval/simulation/user_simulator.py | 6 ++ tests/test_agent.py | 41 ++++++++--- tests/test_orchestrator.py | 9 ++- tests/test_reports.py | 78 +++++++++++++++++++++ tests/test_sub_agent_runner.py | 22 ++++++ tests/test_user_simulator.py | 13 ++++ 15 files changed, 254 insertions(+), 26 deletions(-) diff --git a/docs/AB_EXPERIMENTS.md b/docs/AB_EXPERIMENTS.md index 040d9b97..3c7612e5 100644 --- a/docs/AB_EXPERIMENTS.md +++ b/docs/AB_EXPERIMENTS.md @@ -133,8 +133,8 @@ From `ExperimentVariant` (`coder_eval/models/experiment.py`): The `agent` dict is the lever for most A/B tests. Anything on `AgentConfig` is fair game: `model`, `permission_mode`, `allowed_tools`, `disallowed_tools`, -`plugins`, `system_prompt` / `system_prompt_file`, `setting_sources`, -`claude_settings`, `sdk_options`. +`plugins`, `system_prompt` / `system_prompt_file`, `system_prompt_mode` +(append-vs-replace arms), `setting_sources`, `claude_settings`, `sdk_options`. > **Path-resolution gotcha.** Relative file paths in variant config resolve > against _different_ base directories depending on the field: diff --git a/docs/REPORT_SCHEMA.md b/docs/REPORT_SCHEMA.md index df4bf740..53f1469d 100644 --- a/docs/REPORT_SCHEMA.md +++ b/docs/REPORT_SCHEMA.md @@ -137,6 +137,14 @@ The authoritative per-replicate record. `ClaudeAgentOptions` dump), `sandbox_path`, `task_config` (`{resolved, source_yaml, source_file, lineage}` — `lineage` maps each field to `{value, source, source_detail}` so you can trace which config layer set it). +`environment_info.system_prompt_semantics` (`"append"` / `"replace"`) records the +system-prompt regime the agent ran with; runs predating the marker used +replace-on-set / empty-on-unset semantics and are not score-comparable, so +consumers should segment on it (absent key ⇒ pre-append regime). +`sdk_options.system_prompt` is a `SystemPromptPreset` dict +(`{type: "preset", preset: "claude_code", exclude_dynamic_sections: true, append?: str}`) +on append-mode Claude Code runs and a plain string only in replace mode — it is +no longer `str | null`, so consumers must not string-handle it unconditionally. **Telemetry/totals:** `total_token_usage` ([TokenUsage](#tokenusage)), `command_stats` (`CommandStatistics`), `total_assistant_turns`, `expected_commands` / diff --git a/docs/agents/CLAUDE_CODE.md b/docs/agents/CLAUDE_CODE.md index 8b7aab0d..a84a7ccd 100644 --- a/docs/agents/CLAUDE_CODE.md +++ b/docs/agents/CLAUDE_CODE.md @@ -99,8 +99,8 @@ agent: | `allowed_tools` | `list[str] \| null` | Tool allowlist. Unset ⇒ all tools allowed. | | `disallowed_tools` | `list[str] \| null` | Tool denylist. (`ToolSearch` is always appended for Bedrock parity.) | | `plugins` | `list[{type: local, path}]` | Local plugin/skill directories; `$VAR` in `path` is expanded and resolved to an absolute path. | -| `system_prompt` | `str \| null` | **Appended** to the default Claude Code system prompt (via the SDK's `claude_code` preset) — the default's behavioral guidance is always kept, whether or not this is set. Mutually exclusive with `system_prompt_file`. | -| `system_prompt_mode` | `"append"` (default) / `"replace"` | `replace` sends `system_prompt` as the **entire** system prompt (no preset). Used by judge sub-agents, which must not carry the coding-agent persona; rarely needed in tasks. | +| `system_prompt` | `str \| null` | **Appended** to the default Claude Code system prompt (via the SDK's `claude_code` preset) — the default's behavioral guidance is kept unless `system_prompt_mode: replace` opts out. Mutually exclusive with `system_prompt_file`. | +| `system_prompt_mode` | `"append"` (default) / `"replace"` | `replace` sends `system_prompt` as the **entire** system prompt (no preset) and requires `system_prompt` / `system_prompt_file` to be set (validated at load). Used by judge sub-agents and the user simulator, which must not carry the coding-agent persona; rarely needed in tasks. | | `system_prompt_file` | `str \| null` | Path (relative to the task YAML) loaded into `system_prompt` at resolution. | | `setting_sources` | `list["user"\|"project"\|"local"] \| null` | Which host setting sources the SDK reads. Default resolves to `["project"]`. See [Sandbox isolation](#sandbox-isolation). | | `claude_settings` | `str \| dict \| null` | Passed to the SDK `--settings`. A dict is JSON-serialized; a str is a settings file path. Use `permissions.deny` to block tools/paths. | diff --git a/src/coder_eval/agents/antigravity_agent.py b/src/coder_eval/agents/antigravity_agent.py index 9c1d58ce..84bfcca7 100644 --- a/src/coder_eval/agents/antigravity_agent.py +++ b/src/coder_eval/agents/antigravity_agent.py @@ -348,7 +348,10 @@ async def start( # Autonomous execution: approve every tool call (incl. run_command), # which the default LocalAgentConfig policy would otherwise deny. policies=[policy.allow_all()], - system_instructions=self.config.system_prompt or None, + # Forward verbatim (None stays None): an explicit "" is a + # configured-but-empty prompt and must be forwarded, matching + # the `is not None` semantics in the Claude Code / Codex agents. + system_instructions=self.config.system_prompt, # Skill discovery: hand the harness the search-path roots that parent # the UiPath skill dirs. Unlike Codex (which symlinks into # .agents/skills/), Antigravity takes skill search paths natively. diff --git a/src/coder_eval/agents/claude_code_agent.py b/src/coder_eval/agents/claude_code_agent.py index 3671fa7f..f0d94425 100644 --- a/src/coder_eval/agents/claude_code_agent.py +++ b/src/coder_eval/agents/claude_code_agent.py @@ -10,7 +10,7 @@ from contextlib import suppress from datetime import datetime, timedelta from pathlib import Path -from typing import Any, ClassVar +from typing import Any, ClassVar, Literal from claude_agent_sdk import ( ClaudeAgentOptions, @@ -1189,7 +1189,8 @@ def _build_claude_query( # system_prompt_mode="replace" (judge sub-agents) opts out of the preset: # the configured prompt IS the entire system prompt. system_prompt: str | SystemPromptPreset - if self.config.system_prompt_mode == "replace" and self.config.system_prompt is not None: + if self._effective_prompt_mode() == "replace": + assert self.config.system_prompt is not None # guaranteed by _effective_prompt_mode system_prompt = self.config.system_prompt else: system_prompt = SystemPromptPreset(type="preset", preset="claude_code", exclude_dynamic_sections=True) @@ -1239,6 +1240,19 @@ def _build_claude_query( return options, transport, effective_model + def _effective_prompt_mode(self) -> Literal["append", "replace"]: + """The system-prompt regime that actually goes on the wire. + + Single source of truth for both the options builder and the + ``system_prompt_semantics`` run-record marker, so the persisted regime + can never disagree with what was sent. ``replace`` requires a configured + prompt (the config validator rejects the pair at load, but a mutated or + hand-built config falls back to the preset here — fail open to append). + """ + if self.config.system_prompt_mode == "replace" and self.config.system_prompt is not None: + return "replace" + return "append" + def get_environment_info(self) -> dict[str, Any]: """Record which system-prompt regime built this run's prompts. @@ -1248,7 +1262,7 @@ def get_environment_info(self) -> dict[str, Any]: this marker existed used replace-on-set / empty-on-unset semantics — trend dashboards must not pool scores across that boundary. """ - return {"system_prompt_semantics": self.config.system_prompt_mode} + return {"system_prompt_semantics": self._effective_prompt_mode()} async def stop(self) -> None: """Stop the agent and clean up resources.""" diff --git a/src/coder_eval/agents/codex_agent.py b/src/coder_eval/agents/codex_agent.py index e3f40f7d..2a33441b 100644 --- a/src/coder_eval/agents/codex_agent.py +++ b/src/coder_eval/agents/codex_agent.py @@ -937,12 +937,14 @@ def _close_client(self) -> None: def get_environment_info(self) -> dict[str, Any]: """Record the resolved Codex routing so runs are auditable/comparable. - Only emits when a custom endpoint is configured (CODEX_BASE_URL). On a - custom endpoint the model is an operator-chosen alias (a deployment name - on Azure), so two operators' ``gpt-5-codex`` deployments are otherwise - indistinguishable in run artifacts. The host (not the full URL) is - recorded to avoid leaking any embedded credentials; the API key is never - recorded. + Always emits ``system_prompt_semantics``. The routing keys + (``codex_base_url_host`` / ``codex_wire_api`` / ``codex_api_version`` / + ``codex_model_is_deployment``) are added only when a custom endpoint is + configured (CODEX_BASE_URL): on a custom endpoint the model is an + operator-chosen alias (a deployment name on Azure), so two operators' + ``gpt-5-codex`` deployments are otherwise indistinguishable in run + artifacts. The host (not the full URL) is recorded to avoid leaking any + embedded credentials; the API key is never recorded. """ # system_prompt_semantics: Codex appends system_prompt as # developer_instructions on top of its base prompt. Runs from before this diff --git a/src/coder_eval/evaluation/sub_agent.py b/src/coder_eval/evaluation/sub_agent.py index 88d1bc2e..3e82facc 100644 --- a/src/coder_eval/evaluation/sub_agent.py +++ b/src/coder_eval/evaluation/sub_agent.py @@ -87,6 +87,17 @@ def __init__( "SubAgentRunner requires agent_config.setting_sources=[] so the SDK does not " + "load .claude/settings.json or .mcp.json from the sub-agent's working directory." ) + # A sub-agent's system_prompt is its entire identity (judge instructions, + # simulator persona) — the claude_code coding-agent preset must never + # prefix it. Same fail-loud contract as setting_sources above: callers + # own their config, so a misconfigured one raises instead of being + # silently mutated. + if agent_config.system_prompt is not None and agent_config.system_prompt_mode != "replace": + raise ValueError( + "SubAgentRunner requires agent_config.system_prompt_mode='replace' when a " + + "system_prompt is set: the sub-agent prompt is its entire identity and must " + + "not be appended to the claude_code coding-agent preset." + ) assert sandbox.sandbox_dir is not None, "sandbox not initialized" self._sandbox = sandbox self._agent_config = agent_config diff --git a/src/coder_eval/models/agent_config.py b/src/coder_eval/models/agent_config.py index 721997c1..c7eda5a2 100644 --- a/src/coder_eval/models/agent_config.py +++ b/src/coder_eval/models/agent_config.py @@ -151,7 +151,8 @@ class BaseAgentConfig(BaseModel): system_prompt: str | None = Field( default=None, description=( - "Custom system prompt, appended to the agent's default system prompt — never a replacement. " + "Custom system prompt. Built-in agents layer it on top of their default system prompt " + "rather than replacing it; Claude Code can opt out via system_prompt_mode: replace. " "Each agent's doc page (docs/agents/) states the exact mechanism. " "Supports inline text or multi-line YAML strings. " "Mutually exclusive with system_prompt_file." @@ -254,6 +255,23 @@ def _validate_sdk_options_keys(cls, v: dict[str, Any]) -> dict[str, Any]: ) return v + @model_validator(mode="after") + def check_replace_mode_has_prompt(self) -> Self: + """Reject ``system_prompt_mode: replace`` with no prompt to replace with. + + Without a configured prompt the options builder would fall back to the + claude_code preset (the append regime) while run.json's + ``system_prompt_semantics`` marker could label the run 'replace' — + silently mis-bucketing trend dashboards. ``system_prompt_file`` counts: + the task loader inlines it into ``system_prompt`` at resolution time. + """ + if self.system_prompt_mode == "replace" and self.system_prompt is None and self.system_prompt_file is None: + raise ValueError( + "system_prompt_mode='replace' requires system_prompt (or system_prompt_file) to be set — " + + "there is no prompt to replace the Claude Code default with" + ) + return self + class CodexAgentConfig(BaseAgentConfig): """Codex agent configuration.""" diff --git a/src/coder_eval/reports.py b/src/coder_eval/reports.py index 5ff79eab..180c7fa9 100644 --- a/src/coder_eval/reports.py +++ b/src/coder_eval/reports.py @@ -57,6 +57,24 @@ def resolve_agent_settings(task_dicts: list[dict[str, Any]]) -> tuple[dict[str, return None, False +def _unwrap_system_prompt(value: Any) -> tuple[str | None, str | None]: + """Reduce a persisted ``sdk_options.system_prompt`` value to (text, mode). + + Claude Code append-mode runs persist a ``SystemPromptPreset`` dict + (``{'type': 'preset', 'preset': 'claude_code', ..., 'append': }``); + replace-mode runs persist a plain string. Returns the configured prompt + text (None when nothing was configured — a bare preset dict carries no + custom prompt and gets no row) and the regime worth surfacing ('replace' + for a plain string; None for the default append regime). + """ + if isinstance(value, dict): + append = value.get("append") + return (str(append) if append is not None else None), None + if isinstance(value, str): + return value, "replace" + return None, None + + def collect_agent_settings_rows(settings_source: dict[str, Any], is_sdk: bool) -> list[tuple[str, str]]: """Extract ordered label/value pairs from an agent settings dict. @@ -87,11 +105,14 @@ def collect_agent_settings_rows(settings_source: dict[str, Any], is_sdk: bool) - betas = settings_source.get("betas") if betas: rows.append(("Betas", ", ".join(betas))) - if settings_source.get("system_prompt") is not None: - prompt_str = str(settings_source["system_prompt"]).replace("\n", " ") + prompt_text, prompt_mode = _unwrap_system_prompt(settings_source.get("system_prompt")) + if prompt_text is not None: + prompt_str = prompt_text.replace("\n", " ") if len(prompt_str) > SYSTEM_PROMPT_PREVIEW_CHARS: prompt_str = prompt_str[:SYSTEM_PROMPT_PREVIEW_CHARS] + "..." rows.append(("System Prompt", prompt_str)) + if prompt_mode is not None: + rows.append(("System Prompt Mode", prompt_mode)) plugins = settings_source.get("plugins") if isinstance(plugins, list): diff --git a/src/coder_eval/simulation/user_simulator.py b/src/coder_eval/simulation/user_simulator.py index dca83d57..2001a073 100644 --- a/src/coder_eval/simulation/user_simulator.py +++ b/src/coder_eval/simulation/user_simulator.py @@ -211,6 +211,12 @@ def __init__( setting_sources=[], permission_mode="default", system_prompt=self._system_prompt, + # The roleplay persona IS the simulator's entire identity: 'replace' + # keeps the claude_code coding-agent preset from prefixing it (which + # would contradict the persona's own "stay in character" instruction + # and change every dialog-mode evaluation). Mirrors the judge seam + # in criteria/agent_judge.py. + system_prompt_mode="replace", ) # parse_agent_config returns a union, but type=CLAUDE_CODE guarantees ClaudeCodeAgentConfig assert isinstance(agent_config, ClaudeCodeAgentConfig) diff --git a/tests/test_agent.py b/tests/test_agent.py index 1005a8e9..6cec3156 100644 --- a/tests/test_agent.py +++ b/tests/test_agent.py @@ -6,7 +6,7 @@ from unittest.mock import MagicMock, patch import pytest -from claude_agent_sdk import ProcessError +from claude_agent_sdk import ClaudeAgentOptions, ProcessError from coder_eval.agent import AgentState from coder_eval.agents.claude_code_agent import ClaudeCodeAgent @@ -181,11 +181,11 @@ async def _capture_sdk_options( *, env_path_prepend: list[str] | None = None, max_turns: int | None = None, -) -> "list": +) -> list[ClaudeAgentOptions]: """Run one communicate() turn with a mocked query() and return captured options list.""" import tempfile - captured_options: list = [] + captured_options: list[ClaudeAgentOptions] = [] class ResultMessage: def __init__(self, session_id: str = "s-1") -> None: @@ -377,12 +377,15 @@ async def test_claude_settings_none_default(): assert captured_options[0].settings is None -def _transport_command(options) -> list[str]: +def _transport_command(options: ClaudeAgentOptions) -> list[str]: """Render captured ClaudeAgentOptions into the actual CLI argv. - The dict-shape assertions pin the values we set; this pins the SDK contract - (which flag the transport emits) — the surface the original replace-vs-append - bug lived on — and survives an SDK TypedDict reshape. + The dict-shape assertions pin the values we set; this pins the argv half of + the SDK contract (which flag the transport emits) — the surface the original + replace-vs-append bug lived on — and survives an SDK TypedDict reshape. + Note ``exclude_dynamic_sections`` never reaches argv: the SDK sends it as a + control-protocol ``excludeDynamicSections`` initialize field, which this + helper cannot see. """ from claude_agent_sdk._internal.transport.subprocess_cli import SubprocessCLITransport @@ -438,7 +441,12 @@ async def test_system_prompt_empty_string_appends_empty(): captured_options = await _capture_sdk_options(agent) - assert captured_options[0].system_prompt["append"] == "" + assert captured_options[0].system_prompt == { + "type": "preset", + "preset": "claude_code", + "exclude_dynamic_sections": True, + "append": "", + } @pytest.mark.asyncio @@ -471,6 +479,23 @@ def test_environment_info_reports_system_prompt_semantics(): assert judge_like.get_environment_info() == {"system_prompt_semantics": "replace"} +def test_system_prompt_mode_replace_requires_prompt(): + """The fourth cell of the mode x prompt matrix: 'replace' with no prompt is + rejected at config validation — otherwise the options builder would fall + back to the preset (append regime) while run.json recorded 'replace'.""" + import pydantic + + with pytest.raises(pydantic.ValidationError, match="system_prompt_mode='replace' requires"): + parse_agent_config(type=AgentKind.CLAUDE_CODE, system_prompt_mode="replace") + + # system_prompt_file satisfies the requirement: task_loader inlines it into + # system_prompt at resolution time. + config = parse_agent_config( + type=AgentKind.CLAUDE_CODE, system_prompt_file="prompt.md", system_prompt_mode="replace" + ) + assert config.system_prompt_mode == "replace" + + @pytest.mark.asyncio async def test_sdk_options_forwarded_to_sdk(): """An sdk_options key (e.g. effort) is splatted into ClaudeAgentOptions.""" diff --git a/tests/test_orchestrator.py b/tests/test_orchestrator.py index 7eee4c19..0b0ed816 100644 --- a/tests/test_orchestrator.py +++ b/tests/test_orchestrator.py @@ -603,7 +603,10 @@ def get_sdk_options(self): return {"env": {"PATH": os.environ.get("PATH", "")}} def get_environment_info(self): - return {} + # Non-empty on purpose: pins the orchestrator merge seam + # (environment_info.update(agent.get_environment_info())) that + # carries agent markers like system_prompt_semantics into run.json. + return {"system_prompt_semantics": "append"} async def create_dummy_agent(_self): return DummyAgent() @@ -635,6 +638,10 @@ async def create_dummy_agent(_self): await orchestrator._setup() + # An agent-supplied environment_info key survives the merge into the + # run record (the cross-repo contract seam external consumers read). + assert orchestrator.result.environment_info["system_prompt_semantics"] == "append" + assert isinstance(orchestrator.sandbox, Sandbox) assert orchestrator.sandbox.sandbox_dir is not None assert not orchestrator.sandbox.is_persistent diff --git a/tests/test_reports.py b/tests/test_reports.py index 756a345f..fb3dc559 100644 --- a/tests/test_reports.py +++ b/tests/test_reports.py @@ -904,6 +904,84 @@ def test_generate_markdown_sdk_options_defaults_hidden(): assert "**System Prompt**" not in report_md +def test_agent_settings_rows_system_prompt_preset_shapes(): + """Feed a REAL dump_dataclass(ClaudeAgentOptions) — the shape a Claude Code run + actually persists into sdk_options — not a hand-built dict: append-mode runs + carry a SystemPromptPreset dict, and the report must render the appended + prompt text (never the dict repr), omit the row for a bare preset, and + surface the regime for a replace-mode plain string.""" + from claude_agent_sdk import ClaudeAgentOptions + from claude_agent_sdk.types import SystemPromptPreset + + from coder_eval.reports import collect_agent_settings_rows + from coder_eval.utils import dump_dataclass + + appended = SystemPromptPreset( + type="preset", preset="claude_code", exclude_dynamic_sections=True, append="Be terse." + ) + rows = dict(collect_agent_settings_rows(dump_dataclass(ClaudeAgentOptions(system_prompt=appended)), is_sdk=True)) + assert rows["System Prompt"] == "Be terse." + assert "System Prompt Mode" not in rows + + bare = SystemPromptPreset(type="preset", preset="claude_code", exclude_dynamic_sections=True) + rows = dict(collect_agent_settings_rows(dump_dataclass(ClaudeAgentOptions(system_prompt=bare)), is_sdk=True)) + assert "System Prompt" not in rows + + rows = dict(collect_agent_settings_rows(dump_dataclass(ClaudeAgentOptions(system_prompt="Grader.")), is_sdk=True)) + assert rows["System Prompt"] == "Grader." + assert rows["System Prompt Mode"] == "replace" + + +def test_generate_markdown_system_prompt_preset_not_dict_repr(): + """End-to-end: a preset-shaped sdk_options.system_prompt renders as prompt text + in the Markdown report, and a bare preset (no configured prompt) keeps the + System Prompt row absent — the pre-preset behavior for unset prompts.""" + summary = RunSummary( + run_id="test-run", + start_time=datetime(2025, 10, 11, 12, 0, 0), + end_time=datetime(2025, 10, 11, 12, 1, 0), + total_duration_seconds=60.0, + tasks_run=1, + tasks_succeeded=1, + tasks_failed=0, + tasks_error=0, + task_results=[ + _make_task_result( + "task1", + "SUCCESS", + 1.0, + 30.0, + iteration_count=1, + sdk_options={ + "permission_mode": "bypassPermissions", + "allowed_tools": [], + "system_prompt": { + "type": "preset", + "preset": "claude_code", + "exclude_dynamic_sections": True, + "append": "You are a careful engineer.", + }, + }, + ), + ], + framework_version="0.1.0", + environment_info={}, + ) + + report_md = ReportGenerator.generate_markdown(summary) + assert "**System Prompt**: You are a careful engineer." in report_md + assert "{'type': 'preset'" not in report_md + + # Bare preset == no configured prompt: the row stays absent. + summary.task_results[0]["sdk_options"]["system_prompt"] = { + "type": "preset", + "preset": "claude_code", + "exclude_dynamic_sections": True, + } + report_md = ReportGenerator.generate_markdown(summary) + assert "**System Prompt**" not in report_md + + def test_generate_markdown_no_agent_settings(): """Test that Agent Settings section is omitted when no task has agent_config.""" summary = RunSummary( diff --git a/tests/test_sub_agent_runner.py b/tests/test_sub_agent_runner.py index a40a31aa..68a55ba2 100644 --- a/tests/test_sub_agent_runner.py +++ b/tests/test_sub_agent_runner.py @@ -46,6 +46,7 @@ def _make_agent_config() -> ClaudeCodeAgentConfig: permission_mode="bypassPermissions", allowed_tools=["Read"], system_prompt="x", + system_prompt_mode="replace", # identity-prompt contract setting_sources=[], # security contract ), ) @@ -438,6 +439,27 @@ def test_runner_asserts_setting_sources_empty(sandbox: Sandbox, bad_sources: lis ) +def test_runner_asserts_replace_mode_for_identity_prompt(sandbox: Sandbox) -> None: + """Identity-prompt contract: a sub-agent's system_prompt is its entire identity, + so append mode (which would prefix the claude_code coding-agent preset) is + rejected at construction rather than silently changing the sub-agent's persona.""" + bad_config = parse_agent_config( + type=AgentKind.CLAUDE_CODE, + model="claude-opus-4-6", + permission_mode="bypassPermissions", + allowed_tools=["Read"], + system_prompt="x", # mode defaults to 'append' + setting_sources=[], + ) + with pytest.raises(ValueError, match="system_prompt_mode"): + SubAgentRunner( + sandbox=sandbox, + agent_config=bad_config, + ignore_patterns=[], + route=DirectRoute(), + ) + + # --- symlink + pattern filtering (unit + end-to-end) --- diff --git a/tests/test_user_simulator.py b/tests/test_user_simulator.py index 3adb9647..ca1e5aa4 100644 --- a/tests/test_user_simulator.py +++ b/tests/test_user_simulator.py @@ -70,6 +70,19 @@ def test_system_prompt_override_used_verbatim(self): ) assert sim.system_prompt == "CUSTOM TEMPLATE BODY" + def test_persona_replaces_coding_agent_preset(self): + """The persona is the simulator's ENTIRE system prompt: the agent config + must use system_prompt_mode='replace' so the claude_code coding-agent + preset never prefixes the roleplay identity (which would contradict its + own "stay in character" instruction on every dialog-mode run).""" + sim = UserSimulator( + config=_sim_cfg(), + task_description="A task", + initial_prompt="Start", + ) + assert sim._agent_config.system_prompt_mode == "replace" + assert sim._agent_config.system_prompt == sim.system_prompt + def test_opener_wording_when_no_initial_prompt(self): sim = UserSimulator( config=_sim_cfg(persona="BA", goal="build dice roller"),