treat guardrail rejections like any other LLM error

This commit is contained in:
Ahmed Allam
2026-07-27 23:49:35 +00:00
parent 1814a63de0
commit 8209e92081
5 changed files with 3 additions and 194 deletions
-119
View File
@@ -6,21 +6,15 @@ import asyncio
import contextlib
import json
from typing import Any
from unittest.mock import MagicMock
import pytest
from agents import RunConfig
from agents.memory import SQLiteSession
from agents.tool_context import ToolContext
from strix.config import codex
from strix.core import execution
from strix.core.agents import AgentCoordinator
from strix.core.execution import (
_handle_content_guardrail,
_notify_parent_on_terminal,
_notify_root_on_budget_reserve,
respawn_subagents,
)
from strix.tools.finish.tool import finish_scan
@@ -501,116 +495,3 @@ async def test_terminal_notice_does_not_cancel_parent_stream(tmp_path: Any) -> N
assert stream.cancelled is False
assert coordinator.pending_counts.get("root", 0) > 0
session.close()
@pytest.mark.asyncio
async def test_guardrail_interactive_parks_agent_wakeable(tmp_path: Any) -> None:
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
await coordinator.register("child", "recon", parent_id="root")
exc = codex.CodexContentGuardrailError("chatgpt/gpt-5.6-sol")
result = await _handle_content_guardrail(coordinator, "child", exc, interactive=True)
assert result is None
assert coordinator.statuses["child"] == "waiting"
assert "STRIX_LLM" in coordinator.errors["child"]
waiter = asyncio.create_task(coordinator.wait_for_message("child"))
await asyncio.sleep(0)
assert not waiter.done()
session = SQLiteSession("child", tmp_path / "agents.db")
await coordinator.attach_runtime("child", session=session)
await coordinator.send("child", {"from": "user", "content": "switched model, resume"})
await asyncio.wait_for(waiter, timeout=1.0)
session.close()
@pytest.mark.asyncio
async def test_guardrail_noninteractive_fails_only_blocked_agent(tmp_path: Any) -> None:
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
await coordinator.register("child", "recon", parent_id="root")
session = SQLiteSession("root", tmp_path / "agents.db")
await coordinator.attach_runtime("root", session=session)
exc = codex.CodexContentGuardrailError("chatgpt/gpt-5.6-sol")
result = await _handle_content_guardrail(coordinator, "child", exc, interactive=False)
assert result is None
assert coordinator.statuses["child"] == "failed"
assert "STRIX_LLM" in coordinator.errors["child"]
assert coordinator.statuses["root"] == "running"
assert coordinator.pending_counts.get("root", 0) > 0
session.close()
@pytest.mark.asyncio
async def test_resume_revives_guardrail_parked_child_but_not_plain_waiting(
tmp_path: Any, monkeypatch: pytest.MonkeyPatch
) -> None:
coordinator = AgentCoordinator()
await coordinator.register("root", "strix", parent_id=None)
await coordinator.register("blocked", "recon", parent_id="root")
await coordinator.register("peer_waiter", "recon", parent_id="root")
await coordinator.set_status("blocked", "waiting", error="STRIX_LLM guardrail")
await coordinator.set_status("peer_waiter", "waiting")
parked: dict[str, bool] = {}
async def _fake_start_child_runner(**kwargs: Any) -> None:
parked[kwargs["child_id"]] = bool(kwargs["start_parked"])
monkeypatch.setattr(execution, "_start_child_runner", _fake_start_child_runner)
await respawn_subagents(
coordinator=coordinator,
factory=lambda **_kwargs: object(),
agents_db_path=tmp_path / "agents.db",
sessions_to_close=[],
run_config=MagicMock(),
max_turns=10,
interactive=True,
parent_ctx={"agent_id": "root", "parent_id": None},
root_id="root",
)
assert parked["blocked"] is False
assert parked["peer_waiter"] is True
class _StubLlmSettings:
def __init__(self, model: str) -> None:
self.model = model
self.reasoning_effort = None
self.force_required_tool_choice = False
self.timeout = None
self.prompt_cache = False
class _StubSettings:
def __init__(self, model: str) -> None:
self.llm = _StubLlmSettings(model)
def test_refresh_run_config_model_adopts_changed_model(
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(execution, "reload_settings", lambda: _StubSettings("openai/gpt-5"))
monkeypatch.setattr(execution, "configure_sdk_model_defaults", lambda _settings: None)
run_config = RunConfig(model="chatgpt/gpt-5.6-sol")
refreshed = execution.refresh_run_config_model(run_config)
assert refreshed is not run_config
assert refreshed.model == "openai/gpt-5"
assert run_config.model == "chatgpt/gpt-5.6-sol"
def test_refresh_run_config_model_keeps_unchanged_model(
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(execution, "reload_settings", lambda: _StubSettings("chatgpt/gpt-5.6-sol"))
run_config = RunConfig(model="chatgpt/gpt-5.6-sol")
assert execution.refresh_run_config_model(run_config) is run_config