feat(prompt): universal task-completion guidance + local Python toolchain probe (#34340)
* fix(codex): surface error code in Responses 'failed' status errors
When a Codex Responses turn ends with status=failed, the response carries
the failure details under `response.error` as
`{code, message, param, ...}`. The previous extractor pulled only
`message`, so users seeing a rate-limit failure got a bare "Slow down"
string indistinguishable from a generic stream truncation; an
internal_error with empty message degraded to a dict dump
("{'code': 'internal_error', 'message': ''}").
Extract a `_format_responses_error()` helper that:
- prefixes `code` when both code and message are present
(e.g. 'rate_limit_exceeded: Slow down')
- falls back to the bare `code` when message is empty
- accepts both dict and attribute-style payloads (SDK and JSON-RPC paths)
- preserves the prior status-only fallback when no error payload exists
Apply the same helper at the sibling site in
`codex_app_server_session.run_turn()` so codex-CLI subprocess turn
failures get the same treatment.
Tests:
- 8 new unit tests for `_format_responses_error` covering both shapes,
empty/missing fields, non-string fields, and the status-only fallback.
- 2 regression tests on `_normalize_codex_response` for failed status
with and without a code, asserting the exact RuntimeError message.
- All 3603 tests in tests/agent/ pass.
Adapted from anomalyco/opencode#28757.
* feat(prompt): universal task-completion guidance + local Python toolchain probe
Two cross-model failure modes get a single-line answer in the cached
system prompt. Both gated by config (default on), both add zero overhead
when not needed, both verified via real AIAgent prompt builds.
## What changed
`TASK_COMPLETION_GUIDANCE` — short prompt block applied to ALL models.
Targets two failure modes observed on a real Sarasota real-estate build
task: (1) Opus stopped after writing an 85-byte stub and gave a prose
response with finish_reason=stop on call #3 of 90; (2) DeepSeek pushed
through a PEP-668 wall, then returned fabricated listings instead of
admitting the blocker. Both behaviors are model-family-agnostic, so the
guidance lives outside the existing tool_use_enforcement gate (~192
tokens, paid once per session via prefix cache).
`tools/env_probe.py` — local Python toolchain probe. Detects
python3/pip/uv/PEP-668 state and emits ONE short line in the system
prompt when something is non-default. Emits NOTHING when the env is
clean (zero token cost for normal users). Skipped entirely for remote
terminal backends (docker/modal/ssh) — they have their own probe.
Example output on a broken environment (the actual case):
Python toolchain: python3=3.11.15 (no pip module),
python=missing (use python3), pip→python3.12 (mismatch),
PEP 668=yes (use venv or uv).
## Config
Both flags live under `agent.` in config.yaml, default True:
agent:
task_completion_guidance: true # universal "finish the job" block
environment_probe: true # local Python toolchain hints
Neither addition required a `_config_version` bump — deep-merge fills
defaults in for existing user configs.
## Validation
| Test surface | Result |
|---|---|
| tests/tools/test_env_probe.py | 10/10 pass (probe unit) |
| tests/run_agent/test_run_agent.py — new classes | 8/8 pass (integration) |
| TestToolUseEnforcementConfig | 17/17 pass (no regression) |
| TestBuildSystemPrompt | 9/9 pass (no regression) |
| TestInvalidateSystemPrompt | 2/2 pass (no regression) |
| tests/agent/test_prompt_builder.py | 124/124 pass (no regression) |
| tests/hermes_cli/ | 5662/5662 pass (config defaults) |
| E2E AIAgent build (broken env) | Both blocks present, 2,178 chars |
| E2E AIAgent build (clean env) | 771-char net overhead, env probe silent |
This commit is contained in:
@@ -1323,6 +1323,178 @@ class TestToolUseEnforcementConfig:
|
||||
assert TOOL_USE_ENFORCEMENT_GUIDANCE not in prompt
|
||||
|
||||
|
||||
class TestTaskCompletionGuidance:
|
||||
"""Tests for the universal task-completion / no-fabrication guidance
|
||||
(config.yaml ``agent.task_completion_guidance``).
|
||||
|
||||
Unlike tool_use_enforcement, this block is model-family-agnostic — it
|
||||
targets cross-model failure modes (stopping after a stub; fabricating
|
||||
output when blocked) and should appear for every model by default."""
|
||||
|
||||
def _make_agent(self, model="anthropic/claude-opus-4.8",
|
||||
task_completion_guidance=True, **extra_cfg):
|
||||
agent_cfg = {"task_completion_guidance": task_completion_guidance}
|
||||
agent_cfg.update(extra_cfg)
|
||||
with (
|
||||
patch(
|
||||
"run_agent.get_tool_definitions",
|
||||
return_value=_make_tool_defs("terminal", "web_search"),
|
||||
),
|
||||
patch("run_agent.check_toolset_requirements", return_value={}),
|
||||
patch("run_agent.OpenAI"),
|
||||
patch(
|
||||
"hermes_cli.config.load_config",
|
||||
return_value={"agent": agent_cfg},
|
||||
),
|
||||
):
|
||||
a = AIAgent(
|
||||
model=model,
|
||||
api_key="test-key-1234567890",
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
)
|
||||
a.client = MagicMock()
|
||||
return a
|
||||
|
||||
def test_default_injects_for_claude(self):
|
||||
"""The block must reach Claude by default — that's the
|
||||
primary motivating model family."""
|
||||
from agent.prompt_builder import TASK_COMPLETION_GUIDANCE
|
||||
agent = self._make_agent(model="anthropic/claude-opus-4.8")
|
||||
prompt = agent._build_system_prompt()
|
||||
assert TASK_COMPLETION_GUIDANCE in prompt
|
||||
|
||||
def test_default_injects_for_deepseek(self):
|
||||
"""And for DeepSeek — the other model that failed the Sarasota
|
||||
real-estate task by fabricating output."""
|
||||
from agent.prompt_builder import TASK_COMPLETION_GUIDANCE
|
||||
agent = self._make_agent(model="deepseek/deepseek-v4-flash")
|
||||
prompt = agent._build_system_prompt()
|
||||
assert TASK_COMPLETION_GUIDANCE in prompt
|
||||
|
||||
def test_default_injects_for_gpt(self):
|
||||
"""Also reaches model families that already get enforcement —
|
||||
it's additive, not exclusive."""
|
||||
from agent.prompt_builder import TASK_COMPLETION_GUIDANCE
|
||||
agent = self._make_agent(model="openai/gpt-5.4")
|
||||
prompt = agent._build_system_prompt()
|
||||
assert TASK_COMPLETION_GUIDANCE in prompt
|
||||
|
||||
def test_false_disables(self):
|
||||
from agent.prompt_builder import TASK_COMPLETION_GUIDANCE
|
||||
agent = self._make_agent(
|
||||
model="anthropic/claude-opus-4.8", task_completion_guidance=False
|
||||
)
|
||||
prompt = agent._build_system_prompt()
|
||||
assert TASK_COMPLETION_GUIDANCE not in prompt
|
||||
|
||||
def test_no_tools_no_injection(self):
|
||||
"""Same gate as tool_use_enforcement — no tools means no guidance.
|
||||
The guidance refers to ``tool calls`` and ``tool output``; without
|
||||
tools it would be advice for a capability the agent doesn't have."""
|
||||
from agent.prompt_builder import TASK_COMPLETION_GUIDANCE
|
||||
with (
|
||||
patch("run_agent.get_tool_definitions", return_value=[]),
|
||||
patch("run_agent.check_toolset_requirements", return_value={}),
|
||||
patch("run_agent.OpenAI"),
|
||||
patch(
|
||||
"hermes_cli.config.load_config",
|
||||
return_value={"agent": {"task_completion_guidance": True}},
|
||||
),
|
||||
):
|
||||
a = AIAgent(
|
||||
api_key="test-key-1234567890",
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
enabled_toolsets=[],
|
||||
)
|
||||
a.client = MagicMock()
|
||||
assert TASK_COMPLETION_GUIDANCE not in a._build_system_prompt()
|
||||
|
||||
|
||||
class TestEnvironmentProbeIntegration:
|
||||
"""Tests for the local Python toolchain probe wiring (config.yaml
|
||||
``agent.environment_probe``). The probe itself is unit-tested in
|
||||
tests/tools/test_env_probe.py; this class confirms it lands in the
|
||||
system prompt when enabled and stays out when disabled."""
|
||||
|
||||
def _make_agent(self, model="anthropic/claude-opus-4.8",
|
||||
environment_probe=True):
|
||||
with (
|
||||
patch(
|
||||
"run_agent.get_tool_definitions",
|
||||
return_value=_make_tool_defs("terminal"),
|
||||
),
|
||||
patch("run_agent.check_toolset_requirements", return_value={}),
|
||||
patch("run_agent.OpenAI"),
|
||||
patch(
|
||||
"hermes_cli.config.load_config",
|
||||
return_value={"agent": {"environment_probe": environment_probe}},
|
||||
),
|
||||
):
|
||||
a = AIAgent(
|
||||
model=model,
|
||||
api_key="test-key-1234567890",
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
)
|
||||
a.client = MagicMock()
|
||||
return a
|
||||
|
||||
def test_probe_appears_when_problem_detected(self, monkeypatch):
|
||||
"""When the probe finds something off, the line lands in the prompt."""
|
||||
from tools import env_probe
|
||||
env_probe._reset_cache_for_tests()
|
||||
monkeypatch.setattr(env_probe, "_python_version_of",
|
||||
lambda b: {"python3": "3.11.15"}.get(b))
|
||||
monkeypatch.setattr(env_probe, "_has_pip_module", lambda b: False)
|
||||
monkeypatch.setattr(env_probe, "_detect_pep668", lambda b: True)
|
||||
monkeypatch.setattr(env_probe, "_pip_python_version", lambda: "3.12")
|
||||
monkeypatch.setattr(env_probe.shutil, "which",
|
||||
lambda name: None if name == "uv" else "/usr/bin/" + name)
|
||||
|
||||
agent = self._make_agent(environment_probe=True)
|
||||
prompt = agent._build_system_prompt()
|
||||
assert "Python toolchain:" in prompt
|
||||
assert "3.11.15" in prompt
|
||||
|
||||
def test_probe_silent_on_clean_env(self, monkeypatch):
|
||||
"""Clean environment → probe emits nothing → no line in prompt."""
|
||||
from tools import env_probe
|
||||
env_probe._reset_cache_for_tests()
|
||||
monkeypatch.setattr(env_probe, "_python_version_of",
|
||||
lambda b: "3.13.3" if b == "python3" else None)
|
||||
monkeypatch.setattr(env_probe, "_has_pip_module", lambda b: True)
|
||||
monkeypatch.setattr(env_probe, "_detect_pep668", lambda b: False)
|
||||
monkeypatch.setattr(env_probe, "_pip_python_version", lambda: "3.13")
|
||||
monkeypatch.setattr(env_probe.shutil, "which", lambda name: None)
|
||||
|
||||
agent = self._make_agent(environment_probe=True)
|
||||
prompt = agent._build_system_prompt()
|
||||
assert "Python toolchain:" not in prompt
|
||||
|
||||
def test_probe_disabled_by_config(self, monkeypatch):
|
||||
"""Even with detectable problems, the probe stays out when disabled."""
|
||||
from tools import env_probe
|
||||
env_probe._reset_cache_for_tests()
|
||||
monkeypatch.setattr(env_probe, "_python_version_of",
|
||||
lambda b: {"python3": "3.11.15"}.get(b))
|
||||
monkeypatch.setattr(env_probe, "_has_pip_module", lambda b: False)
|
||||
monkeypatch.setattr(env_probe, "_detect_pep668", lambda b: True)
|
||||
monkeypatch.setattr(env_probe, "_pip_python_version", lambda: "3.12")
|
||||
monkeypatch.setattr(env_probe.shutil, "which", lambda name: None)
|
||||
|
||||
agent = self._make_agent(environment_probe=False)
|
||||
prompt = agent._build_system_prompt()
|
||||
assert "Python toolchain:" not in prompt
|
||||
|
||||
|
||||
class TestInvalidateSystemPrompt:
|
||||
def test_clears_cache(self, agent):
|
||||
agent._cached_system_prompt = "cached value"
|
||||
|
||||
Reference in New Issue
Block a user