mirror of
https://github.com/usestrix/strix.git
synced 2026-08-16 17:27:26 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d97dc1b2ef | ||
|
|
8b464dae5d | ||
|
|
b7bf52c468 | ||
|
|
ceff3b5408 | ||
|
|
44bb3abdf8 | ||
|
|
5726c2d4ef | ||
|
|
69a60f3b7a | ||
|
|
5602bc23ca | ||
|
|
a9deb84260 |
+11
-3
@@ -23,7 +23,7 @@ from strix.tools.agents_graph.tools import (
|
||||
send_message_to_agent,
|
||||
stop_agent,
|
||||
view_agent_graph,
|
||||
wait_for_message,
|
||||
wait_for_agents,
|
||||
)
|
||||
from strix.tools.finish.tool import finish_scan
|
||||
from strix.tools.load_skill.tool import load_skill
|
||||
@@ -49,6 +49,7 @@ from strix.tools.reporting.tool import (
|
||||
get_report,
|
||||
list_reports,
|
||||
)
|
||||
from strix.tools.respond.tool import respond_to_user
|
||||
from strix.tools.thinking.tool import think
|
||||
from strix.tools.todo.tools import (
|
||||
create_todo,
|
||||
@@ -345,6 +346,10 @@ def _make_shell_configurator(*, chat_completions: bool) -> Any:
|
||||
return configure
|
||||
|
||||
|
||||
# Tools that hand control away by parking the agent rather than ending the scan.
|
||||
_PARKING_TOOLS: frozenset[str] = frozenset({"respond_to_user", "wait_for_agents"})
|
||||
|
||||
|
||||
def _lifecycle_tool_completed(tool_name: str, output: Any) -> bool:
|
||||
if tool_name == "agent_finish":
|
||||
completion_key = "agent_completed"
|
||||
@@ -363,7 +368,7 @@ def _lifecycle_tool_completed(tool_name: str, output: Any) -> bool:
|
||||
|
||||
|
||||
def _wait_tool_parked(tool_name: str, output: Any) -> bool:
|
||||
if tool_name != "wait_for_message" or not isinstance(output, str):
|
||||
if tool_name not in _PARKING_TOOLS or not isinstance(output, str):
|
||||
return False
|
||||
try:
|
||||
parsed = json.loads(output)
|
||||
@@ -425,7 +430,7 @@ _BASE_TOOLS: tuple[Tool, ...] = (
|
||||
scope_rules,
|
||||
view_agent_graph,
|
||||
send_message_to_agent,
|
||||
wait_for_message,
|
||||
wait_for_agents,
|
||||
create_agent,
|
||||
stop_agent,
|
||||
)
|
||||
@@ -509,6 +514,9 @@ def build_strix_agent(
|
||||
)
|
||||
|
||||
agent_tools = [*_EXTRA_TOOLS, *(extra_tools or [])]
|
||||
if interactive:
|
||||
# Yielding to the user is only meaningful when one is attached.
|
||||
agent_tools.append(respond_to_user)
|
||||
if is_root:
|
||||
tools: list[Tool] = [*_BASE_TOOLS, *agent_tools, finish_scan]
|
||||
else:
|
||||
|
||||
@@ -31,28 +31,26 @@ INTER-AGENT MESSAGES:
|
||||
|
||||
{% if interactive %}
|
||||
INTERACTIVE BEHAVIOR:
|
||||
- You are in an interactive conversation with a user
|
||||
- CRITICAL: A message WITHOUT a tool call IMMEDIATELY STOPS your entire execution and waits for user input. This is a HARD SYSTEM CONSTRAINT, not a suggestion.
|
||||
- Statements like "Planning the assessment..." or "I'll now scan..." or "Starting with..." WITHOUT a tool call will HALT YOUR WORK COMPLETELY. The system interprets no-tool-call as "I'm done, waiting for the user."
|
||||
- If you want to plan, call the think tool. If you want to act, call the appropriate tool. There is NO valid reason to output text without a tool call while working on a task.
|
||||
- The ONLY time you may send a message without a tool call is when you are genuinely DONE and presenting final results, or when you NEED the user to answer a question before continuing.
|
||||
- EVERY message while working MUST contain exactly one tool call — this is what keeps execution moving. No tool call = execution stops.
|
||||
- You may include brief explanatory text BEFORE the tool call
|
||||
- Respond naturally when the user asks questions or gives instructions
|
||||
- For simple conversation, acknowledgements, or direct questions that you can answer from current context, reply in plain text and stop. Do NOT call think just to prepare wording.
|
||||
- If you use a tool to answer a user question (for example list_todos, view_agent_graph, or a file read), then after the tool result arrives, provide the answer in plain text and stop unless the user explicitly asked you to continue working.
|
||||
- Never loop through think or other tools just to prepare, polish, confirm, or announce a final answer. Once you know the answer, say it.
|
||||
- NEVER send empty messages — if you have nothing to do or say, call the wait_for_message tool
|
||||
- If you catch yourself about to describe multiple steps without a tool call, STOP and call the think tool instead
|
||||
- You are in an interactive conversation with a user.
|
||||
- HOW EXECUTION ENDS: your turn ends ONLY when you make an explicit lifecycle tool call. Plain text NEVER ends your turn and NEVER hands control to the user — text is shown to the user, and then execution continues.
|
||||
- To answer the user and hand control back, call respond_to_user. It delivers your message AND parks you for their reply in one call, so there is no way to answer and then forget to stop. This is the ONLY way to yield to the user.
|
||||
- To wait on another AGENT (a child's report, a peer's reply), call wait_for_agents. That is not a way to reach the user.
|
||||
- To end the whole engagement, call the lifecycle tool: finish_scan (root) or agent_finish (subagent).
|
||||
- A turn that ends with plain text and no tool call does NOT stop you: the system nudges you to continue and will re-run you. Do not rely on going silent to pause — it will not pause you.
|
||||
- Answering a user question: put the answer in respond_to_user's message. Do not write the answer as plain text and then fall silent — that does not reach a stopping point, it just triggers a continuation nudge.
|
||||
- You may include brief explanatory text before a tool call, and you can narrate while you work — plain text is shown to the user as you go. Narrating is free; respond_to_user is specifically the act of WAITING for the user, so do not call it just to give a status update.
|
||||
- Respond naturally when the user asks questions or gives instructions.
|
||||
- While actively working on a task, every turn should carry exactly one tool call — use think to plan, the appropriate tool to act, and respond_to_user only when you genuinely need the user.
|
||||
- Never loop through think or other tools just to prepare, polish, confirm, or announce an answer. Once you know the answer, send it with respond_to_user.
|
||||
{% else %}
|
||||
AUTONOMOUS BEHAVIOR:
|
||||
- Work autonomously by default
|
||||
- You should NOT ask for user input or confirmation - you should always proceed with your task autonomously.
|
||||
- Minimize user messaging: avoid redundancy and repetition; consolidate updates into a single concise message
|
||||
- NEVER send an empty or blank message. If you have no content to output or need to wait (for user input, subagent results, or any other reason), you MUST call the wait_for_message tool (or another appropriate tool) instead of emitting an empty response.
|
||||
- If there is nothing to execute and no user query to answer any more: do NOT send filler/repetitive text — either call wait_for_message or finish your work (subagents: agent_finish; root: finish_scan)
|
||||
- While the agent loop is running, almost every output MUST be a tool call. Do NOT send plain text messages; act via tools. If idle, use wait_for_message; when done, use agent_finish (subagents) or finish_scan (root)
|
||||
- A text-only turn — even one — IMMEDIATELY ends the scan/run with no report written. The lifecycle tools (``finish_scan`` for root, ``agent_finish`` for subagents) are the ONLY valid way to terminate. If you find yourself wanting to say "Done!" or "Scan complete" without a tool call, call the lifecycle tool instead — the report and termination signal both flow through it.
|
||||
- NEVER send an empty or blank message. If you have no content to output or need to wait for subagent results, you MUST call the wait_for_agents tool (or another appropriate tool) instead of emitting an empty response.
|
||||
- There is no user attached to this run, so there is nobody to ask and nothing to yield to. If there is nothing left to execute: do NOT send filler/repetitive text — either call wait_for_agents (only if you are genuinely expecting another agent to message you) or finish your work (subagents: agent_finish; root: finish_scan)
|
||||
- While the agent loop is running, almost every output MUST be a tool call. Do NOT send plain text messages; act via tools. If waiting on another agent, use wait_for_agents; when done, use agent_finish (subagents) or finish_scan (root)
|
||||
- A text-only turn does nothing: it neither ends the run nor yields — it just wastes a turn and forces a retry. The lifecycle tools (``finish_scan`` for root, ``agent_finish`` for subagents) are the ONLY way to terminate, and the report flows through them. If you find yourself wanting to say "Done!" or "Scan complete" without a tool call, call the lifecycle tool instead.
|
||||
{% endif %}
|
||||
</communication_rules>
|
||||
|
||||
|
||||
+126
-32
@@ -24,6 +24,11 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
Status = Literal["running", "waiting", "completed", "stopped", "crashed", "failed", "budget_paused"]
|
||||
|
||||
# Why an agent parked. The user can message any agent, so this - not the agent's
|
||||
# position in the tree - decides whether waiting is bounded: only an agent waiting
|
||||
# on other agents is re-checked on a timer.
|
||||
WaitKind = Literal["user", "agents", "stalled"]
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class AgentRuntime:
|
||||
@@ -32,6 +37,8 @@ class AgentRuntime:
|
||||
stream: Any | None = None
|
||||
interrupt_on_message: bool = False
|
||||
wake: asyncio.Event = field(default_factory=asyncio.Event)
|
||||
mailbox: list[dict[str, Any]] = field(default_factory=list)
|
||||
user_wake_required: bool = False
|
||||
|
||||
|
||||
class AgentCoordinator:
|
||||
@@ -44,6 +51,9 @@ class AgentCoordinator:
|
||||
self.metadata: dict[str, dict[str, Any]] = {}
|
||||
self.pending_counts: dict[str, int] = {}
|
||||
self.errors: dict[str, str] = {}
|
||||
self.recovery_counts: dict[str, int] = {}
|
||||
self.idle_resume_counts: dict[str, int] = {}
|
||||
self.wait_kinds: dict[str, WaitKind] = {}
|
||||
self.runtimes: dict[str, AgentRuntime] = {}
|
||||
self._lock = asyncio.Lock()
|
||||
self._snapshot_path: Path | None = None
|
||||
@@ -179,11 +189,58 @@ class AgentCoordinator:
|
||||
if agent_id in self.statuses:
|
||||
self.statuses[agent_id] = "running"
|
||||
self.errors.pop(agent_id, None)
|
||||
self.wait_kinds.pop(agent_id, None)
|
||||
self.runtimes.setdefault(agent_id, AgentRuntime()).user_wake_required = False
|
||||
await self._maybe_snapshot()
|
||||
|
||||
async def park_waiting(self, agent_id: str) -> None:
|
||||
async def park_waiting(self, agent_id: str, *, wait_kind: WaitKind) -> None:
|
||||
"""Park an agent, recording what it is waiting on so the driver can time it."""
|
||||
async with self._lock:
|
||||
if agent_id in self.statuses:
|
||||
self.wait_kinds[agent_id] = wait_kind
|
||||
await self.set_status(agent_id, "waiting")
|
||||
|
||||
async def wait_kind_of(self, agent_id: str) -> WaitKind | None:
|
||||
async with self._lock:
|
||||
return self.wait_kinds.get(agent_id)
|
||||
|
||||
async def record_recovery(self, agent_id: str) -> int:
|
||||
"""Count a turn that ended without a lifecycle tool call; return the new total.
|
||||
|
||||
Persisted so a resumed agent cannot earn a fresh nudge budget on every
|
||||
auto-resume and loop forever.
|
||||
"""
|
||||
async with self._lock:
|
||||
count = self.recovery_counts.get(agent_id, 0) + 1
|
||||
self.recovery_counts[agent_id] = count
|
||||
await self._maybe_snapshot()
|
||||
return count
|
||||
|
||||
async def reset_recovery(self, agent_id: str) -> None:
|
||||
"""Clear the nudge budget after real progress (new message or a lifecycle tool)."""
|
||||
async with self._lock:
|
||||
if self.recovery_counts.pop(agent_id, None) is None:
|
||||
return
|
||||
await self._maybe_snapshot()
|
||||
|
||||
async def record_idle_resume(self, agent_id: str) -> int:
|
||||
"""Count an auto-resume that no message triggered; return the new total.
|
||||
|
||||
An agent that parks again after every auto-resume would otherwise burn a
|
||||
model turn per timeout for the rest of the scan.
|
||||
"""
|
||||
async with self._lock:
|
||||
count = self.idle_resume_counts.get(agent_id, 0) + 1
|
||||
self.idle_resume_counts[agent_id] = count
|
||||
await self._maybe_snapshot()
|
||||
return count
|
||||
|
||||
async def reset_idle_resumes(self, agent_id: str) -> None:
|
||||
async with self._lock:
|
||||
if self.idle_resume_counts.pop(agent_id, None) is None:
|
||||
return
|
||||
await self._maybe_snapshot()
|
||||
|
||||
async def set_status(
|
||||
self, agent_id: str, status: Status | str, *, error: str | None = None
|
||||
) -> None:
|
||||
@@ -196,6 +253,7 @@ class AgentCoordinator:
|
||||
elif status == "running":
|
||||
self.errors.pop(agent_id, None)
|
||||
runtime = self.runtimes.setdefault(agent_id, AgentRuntime())
|
||||
runtime.user_wake_required = status in {"failed", "crashed"}
|
||||
runtime.wake.set()
|
||||
logger.info("agent.status %s=%s", agent_id, status)
|
||||
await self._maybe_snapshot()
|
||||
@@ -203,49 +261,47 @@ class AgentCoordinator:
|
||||
async def send(
|
||||
self, target_agent_id: str, message: dict[str, Any], *, interrupt: bool = True
|
||||
) -> bool:
|
||||
"""Deliver a user/peer message by appending it to the target SDK session."""
|
||||
if message.get("from") == "user" and self._budget_paused:
|
||||
"""Queue a user/peer message in the target's mailbox and wake it."""
|
||||
from_user = message.get("from") == "user"
|
||||
if from_user and self._budget_paused:
|
||||
await self.resume_from_budget_pause(exclude=target_agent_id)
|
||||
async with self._lock:
|
||||
if target_agent_id not in self.statuses:
|
||||
logger.debug("agent.send dropped unknown target=%s", target_agent_id)
|
||||
return False
|
||||
runtime = self.runtimes.setdefault(target_agent_id, AgentRuntime())
|
||||
session = runtime.session
|
||||
runtime.mailbox.append(dict(message))
|
||||
self.pending_counts[target_agent_id] = self.pending_counts.get(target_agent_id, 0) + 1
|
||||
if from_user:
|
||||
runtime.user_wake_required = False
|
||||
runtime.wake.set()
|
||||
stream = runtime.stream
|
||||
interrupt_on_message = runtime.interrupt_on_message
|
||||
if session is None:
|
||||
logger.warning(
|
||||
"agent.send dropped target=%s because its SDK session is not attached",
|
||||
target_agent_id,
|
||||
)
|
||||
return False
|
||||
try:
|
||||
async with session_write_lock(session):
|
||||
await session.add_items([self._message_to_session_item(message)])
|
||||
except Exception:
|
||||
logger.exception(
|
||||
"agent.send failed to append to SDK session target=%s",
|
||||
target_agent_id,
|
||||
)
|
||||
return False
|
||||
async with self._lock:
|
||||
self.pending_counts[target_agent_id] = self.pending_counts.get(target_agent_id, 0) + 1
|
||||
self.runtimes.setdefault(target_agent_id, AgentRuntime()).wake.set()
|
||||
if stream is not None and interrupt and interrupt_on_message:
|
||||
stream.cancel(mode="immediate")
|
||||
await self._maybe_snapshot()
|
||||
return True
|
||||
|
||||
async def wait_for_message(self, agent_id: str) -> None:
|
||||
async def wait_for_message(self, agent_id: str, *, timeout: float | None = None) -> bool:
|
||||
"""Wait until a message is ready for ``agent_id``; False on ``timeout``."""
|
||||
while True:
|
||||
async with self._lock:
|
||||
runtime = self.runtimes.setdefault(agent_id, AgentRuntime())
|
||||
reserve_exit = self._reserve_stopped and self.parent_of.get(agent_id) is not None
|
||||
if self._budget_stopped or reserve_exit or self.pending_counts.get(agent_id, 0) > 0:
|
||||
return
|
||||
wake = self.runtimes.setdefault(agent_id, AgentRuntime()).wake
|
||||
pending_ready = (
|
||||
self.pending_counts.get(agent_id, 0) > 0 and not runtime.user_wake_required
|
||||
)
|
||||
if self._budget_stopped or reserve_exit or pending_ready:
|
||||
return True
|
||||
wake = runtime.wake
|
||||
wake.clear()
|
||||
await wake.wait()
|
||||
if timeout is None:
|
||||
await wake.wait()
|
||||
else:
|
||||
try:
|
||||
await asyncio.wait_for(wake.wait(), timeout)
|
||||
except TimeoutError:
|
||||
return False
|
||||
|
||||
async def consume_pending(
|
||||
self,
|
||||
@@ -253,17 +309,38 @@ class AgentCoordinator:
|
||||
*,
|
||||
include_items: bool = False,
|
||||
) -> tuple[int, list[Any]]:
|
||||
"""Drain the agent's mailbox into its own SDK session."""
|
||||
async with self._lock:
|
||||
count = self.pending_counts.get(agent_id, 0)
|
||||
runtime = self.runtimes.setdefault(agent_id, AgentRuntime())
|
||||
queued = list(runtime.mailbox)
|
||||
runtime.mailbox.clear()
|
||||
count = max(self.pending_counts.get(agent_id, 0), len(queued))
|
||||
self.pending_counts[agent_id] = 0
|
||||
session = self.runtimes.get(agent_id, AgentRuntime()).session
|
||||
session = runtime.session
|
||||
if count <= 0:
|
||||
return 0, []
|
||||
items = [self._message_to_session_item(m) for m in queued]
|
||||
if items:
|
||||
if session is None:
|
||||
logger.warning(
|
||||
"agent %s has no SDK session attached; %d queued messages were not persisted",
|
||||
agent_id,
|
||||
len(items),
|
||||
)
|
||||
else:
|
||||
try:
|
||||
async with session_write_lock(session):
|
||||
await session.add_items(items)
|
||||
except Exception:
|
||||
logger.exception(
|
||||
"failed to append %d queued messages to the session of %s",
|
||||
len(items),
|
||||
agent_id,
|
||||
)
|
||||
await self._maybe_snapshot()
|
||||
if not include_items or session is None:
|
||||
if not include_items:
|
||||
return count, []
|
||||
items = await session.get_items()
|
||||
return count, list(items[-count:])
|
||||
return count, items
|
||||
|
||||
async def request_stop(self, agent_id: str) -> None:
|
||||
async with self._lock:
|
||||
@@ -374,6 +451,14 @@ class AgentCoordinator:
|
||||
"names": dict(self.names),
|
||||
"metadata": {aid: dict(md) for aid, md in self.metadata.items()},
|
||||
"pending_counts": dict(self.pending_counts),
|
||||
"recovery_counts": dict(self.recovery_counts),
|
||||
"idle_resume_counts": dict(self.idle_resume_counts),
|
||||
"wait_kinds": dict(self.wait_kinds),
|
||||
"mailboxes": {
|
||||
aid: [dict(m) for m in runtime.mailbox]
|
||||
for aid, runtime in self.runtimes.items()
|
||||
if runtime.mailbox
|
||||
},
|
||||
"errors": dict(self.errors),
|
||||
"budget_stopped": self._budget_stopped,
|
||||
"reserve_stopped": self._reserve_stopped,
|
||||
@@ -388,6 +473,15 @@ class AgentCoordinator:
|
||||
self.metadata = {aid: dict(md) for aid, md in snap.get("metadata", {}).items()}
|
||||
self.pending_counts = dict(snap.get("pending_counts", {}))
|
||||
self.errors = dict(snap.get("errors", {}))
|
||||
self.recovery_counts = dict(snap.get("recovery_counts", {}))
|
||||
self.idle_resume_counts = dict(snap.get("idle_resume_counts", {}))
|
||||
self.wait_kinds = dict(snap.get("wait_kinds", {}))
|
||||
mailboxes = snap.get("mailboxes", {})
|
||||
if isinstance(mailboxes, dict):
|
||||
for aid, msgs in mailboxes.items():
|
||||
if isinstance(msgs, list):
|
||||
runtime = self.runtimes.setdefault(aid, AgentRuntime())
|
||||
runtime.mailbox = [dict(m) for m in msgs if isinstance(m, dict)]
|
||||
self._budget_stopped = bool(snap.get("budget_stopped", False))
|
||||
self._reserve_stopped = bool(snap.get("reserve_stopped", False))
|
||||
self._budget_paused = bool(snap.get("budget_paused", False))
|
||||
|
||||
+332
-126
@@ -9,6 +9,7 @@ import uuid
|
||||
from collections.abc import Callable
|
||||
from typing import TYPE_CHECKING, Any, cast
|
||||
|
||||
import litellm
|
||||
from agents import RunConfig, Runner
|
||||
from agents.exceptions import AgentsException, MaxTurnsExceeded, UserError
|
||||
from agents.sandbox.errors import ExecTransportError
|
||||
@@ -16,9 +17,7 @@ from docker import errors as docker_errors # type: ignore[import-untyped, unuse
|
||||
from openai import (
|
||||
APIConnectionError,
|
||||
APIError,
|
||||
APIStatusError,
|
||||
APITimeoutError,
|
||||
RateLimitError,
|
||||
)
|
||||
|
||||
from strix.config import codex
|
||||
@@ -31,6 +30,8 @@ from strix.core.inputs import child_initial_input
|
||||
from strix.core.sessions import (
|
||||
enforce_image_budget,
|
||||
open_agent_session,
|
||||
replace_session_items,
|
||||
seed_initial_input,
|
||||
strip_all_images_from_session,
|
||||
)
|
||||
from strix.llm.compaction import is_context_overflow, maybe_compact
|
||||
@@ -55,6 +56,23 @@ _INPUT_REJECTION_CODES = frozenset({400, 404, 422})
|
||||
_MAX_COMPACTIONS_PER_CYCLE = 2
|
||||
|
||||
|
||||
class ProviderRefusalError(AgentsException):
|
||||
"""Raised when a provider returns a structured refusal instead of an exception."""
|
||||
|
||||
|
||||
def _structured_provider_refusal(result: Any) -> str | None:
|
||||
for item in getattr(result, "new_items", ()) or ():
|
||||
raw_item = getattr(item, "raw_item", None)
|
||||
for content in getattr(raw_item, "content", ()) or ():
|
||||
if getattr(content, "type", None) != "refusal":
|
||||
continue
|
||||
refusal = getattr(content, "refusal", None)
|
||||
if isinstance(refusal, str) and refusal.strip():
|
||||
return refusal.strip()
|
||||
return "The model provider refused this request."
|
||||
return None
|
||||
|
||||
|
||||
def _run_config_model(run_config: RunConfig) -> str | None:
|
||||
return run_config.model if isinstance(run_config.model, str) else None
|
||||
|
||||
@@ -89,15 +107,9 @@ async def _compact_session(
|
||||
)
|
||||
|
||||
|
||||
_GUARDRAIL_PARK_ERROR = (
|
||||
"Blocked by the model's content guardrail (flagged as a possible cybersecurity risk). "
|
||||
"Set STRIX_LLM to a model that isn't blocked and resume the scan to continue."
|
||||
)
|
||||
|
||||
_TRANSIENT_MODEL_STATUS_CODES = frozenset({408, 500, 502, 503, 504})
|
||||
_MAX_TRANSIENT_MODEL_RETRIES = 4
|
||||
_MAX_TRANSIENT_MODEL_RETRIES = 5
|
||||
_TRANSIENT_MODEL_RETRY_BASE_DELAY_S = 2.0
|
||||
_TRANSIENT_MODEL_RETRY_MAX_DELAY_S = 30.0
|
||||
_TRANSIENT_MODEL_RETRY_MAX_DELAY_S = 90.0
|
||||
|
||||
|
||||
def _model_error_status_code(exc: BaseException) -> int | None:
|
||||
@@ -106,15 +118,16 @@ def _model_error_status_code(exc: BaseException) -> int | None:
|
||||
|
||||
|
||||
def _is_transient_model_error(exc: BaseException) -> bool:
|
||||
if isinstance(exc, RateLimitError):
|
||||
if codex.is_content_guardrail_error(exc):
|
||||
return False
|
||||
if isinstance(exc, APITimeoutError | APIConnectionError):
|
||||
if isinstance(
|
||||
exc, APITimeoutError | APIConnectionError | TimeoutError | ConnectionError | OSError
|
||||
):
|
||||
return True
|
||||
if isinstance(exc, APIStatusError):
|
||||
return exc.status_code in _TRANSIENT_MODEL_STATUS_CODES
|
||||
if isinstance(exc, APIError):
|
||||
return _model_error_status_code(exc) is None
|
||||
return False
|
||||
code = _model_error_status_code(exc)
|
||||
if code is not None:
|
||||
return bool(litellm._should_retry(code))
|
||||
return isinstance(exc, APIError)
|
||||
|
||||
|
||||
def _transient_model_retry_delay(attempt: int) -> float:
|
||||
@@ -122,6 +135,40 @@ def _transient_model_retry_delay(attempt: int) -> float:
|
||||
return min(delay, _TRANSIENT_MODEL_RETRY_MAX_DELAY_S)
|
||||
|
||||
|
||||
async def _salvage_stream_to_session(
|
||||
session: Session,
|
||||
pre_run_items: list[Any],
|
||||
stream: Any,
|
||||
agent_id: str,
|
||||
) -> None:
|
||||
"""Persist a crashed run's full history so a revived agent loses no context."""
|
||||
if stream is None:
|
||||
return
|
||||
try:
|
||||
replay = list(stream.to_input_list())
|
||||
except Exception:
|
||||
logger.exception("could not build salvage history for %s", agent_id)
|
||||
return
|
||||
desired = list(pre_run_items) + replay
|
||||
if len(desired) <= len(pre_run_items):
|
||||
return
|
||||
try:
|
||||
await replace_session_items(session, desired)
|
||||
except Exception:
|
||||
logger.exception("salvaging crashed run history failed for %s", agent_id)
|
||||
|
||||
|
||||
async def _seed_and_prepare_first_input(
|
||||
session: Session | None, initial_input: Any, *, start_parked: bool
|
||||
) -> Any:
|
||||
"""Persist the opening input up front so it survives a first-turn crash."""
|
||||
if initial_input and session is not None and not start_parked:
|
||||
with contextlib.suppress(Exception):
|
||||
if await seed_initial_input(session, initial_input):
|
||||
return []
|
||||
return initial_input
|
||||
|
||||
|
||||
async def run_agent_loop(
|
||||
*,
|
||||
agent: Any,
|
||||
@@ -144,6 +191,10 @@ async def run_agent_loop(
|
||||
)
|
||||
result: RunResultBase | None = None
|
||||
|
||||
first_cycle_input = await _seed_and_prepare_first_input(
|
||||
session, initial_input, start_parked=start_parked
|
||||
)
|
||||
|
||||
budget_stopped = coordinator.budget_stopped
|
||||
reserve_stopped = coordinator.reserve_stopped
|
||||
if budget_stopped:
|
||||
@@ -157,31 +208,17 @@ async def run_agent_loop(
|
||||
await coordinator.send(agent_id, _reserve_notice())
|
||||
|
||||
if not (start_parked and interactive):
|
||||
if interactive:
|
||||
with contextlib.suppress(BudgetPausedError):
|
||||
result = await _run_cycle(
|
||||
agent,
|
||||
coordinator,
|
||||
agent_id,
|
||||
input_data=initial_input,
|
||||
run_config=run_config,
|
||||
context=context,
|
||||
max_turns=max_turns,
|
||||
session=session,
|
||||
interactive=interactive,
|
||||
event_sink=event_sink,
|
||||
hooks=hooks,
|
||||
)
|
||||
else:
|
||||
result = await _run_noninteractive_until_lifecycle(
|
||||
with contextlib.suppress(BudgetPausedError):
|
||||
result = await _run_until_lifecycle(
|
||||
agent,
|
||||
coordinator,
|
||||
agent_id,
|
||||
initial_input=initial_input,
|
||||
initial_input=first_cycle_input,
|
||||
run_config=run_config,
|
||||
context=context,
|
||||
max_turns=max_turns,
|
||||
session=session,
|
||||
interactive=interactive,
|
||||
event_sink=event_sink,
|
||||
hooks=hooks,
|
||||
)
|
||||
@@ -190,8 +227,9 @@ async def run_agent_loop(
|
||||
return result
|
||||
|
||||
while True:
|
||||
timeout = await _plain_waiting_timeout(coordinator, agent_id)
|
||||
try:
|
||||
await coordinator.wait_for_message(agent_id)
|
||||
woke = await coordinator.wait_for_message(agent_id, timeout=timeout)
|
||||
except asyncio.CancelledError:
|
||||
return result
|
||||
|
||||
@@ -203,18 +241,46 @@ async def run_agent_loop(
|
||||
await coordinator.set_status(agent_id, "stopped")
|
||||
raise SubagentBudgetReservedError("scan reached the sub-agent budget reserve")
|
||||
|
||||
if woke:
|
||||
# Real input is real progress, so the nudge budget starts over. A bare
|
||||
# auto-resume is not: it must not hand a wedged agent a fresh budget.
|
||||
await coordinator.reset_recovery(agent_id)
|
||||
await coordinator.reset_idle_resumes(agent_id)
|
||||
else:
|
||||
idle_resumes = await coordinator.record_idle_resume(agent_id)
|
||||
if idle_resumes >= _MAX_IDLE_AUTO_RESUMES:
|
||||
logger.warning(
|
||||
"agent %s auto-resumed %d times without hearing from anyone; "
|
||||
"leaving it parked until a real message arrives",
|
||||
agent_id,
|
||||
idle_resumes,
|
||||
)
|
||||
await coordinator.park_waiting(agent_id, wait_kind="stalled")
|
||||
await _notify_parent_on_stall(coordinator, agent_id)
|
||||
continue
|
||||
logger.info("agent %s reached its waiting timeout; auto-resuming", agent_id)
|
||||
await coordinator.send(
|
||||
agent_id,
|
||||
{
|
||||
"from": "system",
|
||||
"type": "auto_resume",
|
||||
"content": "Waiting timeout reached. Resuming execution.",
|
||||
},
|
||||
interrupt=False,
|
||||
)
|
||||
|
||||
await coordinator.consume_pending(agent_id)
|
||||
with contextlib.suppress(BudgetPausedError):
|
||||
result = await _run_cycle(
|
||||
result = await _run_until_lifecycle(
|
||||
agent,
|
||||
coordinator,
|
||||
agent_id,
|
||||
input_data=[],
|
||||
initial_input=[],
|
||||
run_config=run_config,
|
||||
context=context,
|
||||
max_turns=max_turns,
|
||||
session=session,
|
||||
interactive=interactive,
|
||||
interactive=True,
|
||||
event_sink=event_sink,
|
||||
hooks=hooks,
|
||||
)
|
||||
@@ -310,7 +376,6 @@ async def respawn_subagents(
|
||||
if coordinator.parent_of.get(aid) is None or aid == root_id:
|
||||
continue
|
||||
md["_restored_status"] = status
|
||||
md["_restored_error"] = coordinator.errors.get(aid)
|
||||
candidates.append(
|
||||
(
|
||||
aid,
|
||||
@@ -323,8 +388,7 @@ async def respawn_subagents(
|
||||
for child_id, name, parent_id, md in candidates:
|
||||
try:
|
||||
restored_status = str(md.get("_restored_status") or "running")
|
||||
recoverable_park = restored_status == "waiting" and bool(md.get("_restored_error"))
|
||||
start_parked = interactive and restored_status != "running" and not recoverable_park
|
||||
start_parked = interactive and restored_status != "running"
|
||||
|
||||
if start_parked:
|
||||
logger.warning(
|
||||
@@ -367,7 +431,10 @@ async def respawn_subagents(
|
||||
await coordinator.set_status(child_id, "crashed")
|
||||
|
||||
|
||||
async def _run_noninteractive_until_lifecycle(
|
||||
_INTERACTIVE_TOOL_RECOVERY_LIMIT = 3
|
||||
|
||||
|
||||
async def _run_until_lifecycle(
|
||||
agent: Any,
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
@@ -377,14 +444,20 @@ async def _run_noninteractive_until_lifecycle(
|
||||
context: dict[str, Any],
|
||||
max_turns: int,
|
||||
session: Session | None,
|
||||
interactive: bool,
|
||||
event_sink: StreamEventSink | None,
|
||||
hooks: RunHooks[dict[str, Any]] | None,
|
||||
) -> RunResultBase | None:
|
||||
"""Non-chat mode keeps running until finish_scan / agent_finish settles status."""
|
||||
"""Drive an agent until an explicit lifecycle tool settles its status.
|
||||
|
||||
A turn that ends without ``finish_scan``, ``agent_finish``,
|
||||
``respond_to_user``, or ``wait_for_agents`` leaves the agent ``running``:
|
||||
plain text never terminates a run and never yields to the user. Such a turn
|
||||
is nudged back into a tool call, bounded by a recovery limit.
|
||||
"""
|
||||
result: RunResultBase | None = None
|
||||
input_data: Any = initial_input
|
||||
invalid_final_outputs = 0
|
||||
invalid_final_output_limit = max(1, max_turns)
|
||||
recovery_limit = _INTERACTIVE_TOOL_RECOVERY_LIMIT if interactive else max(1, max_turns)
|
||||
|
||||
while True:
|
||||
if coordinator.budget_stopped:
|
||||
@@ -395,7 +468,143 @@ async def _run_noninteractive_until_lifecycle(
|
||||
await coordinator.set_status(agent_id, "stopped")
|
||||
raise SubagentBudgetReservedError("scan reached the sub-agent budget reserve")
|
||||
|
||||
result = await _run_cycle(
|
||||
if interactive:
|
||||
result = await _run_cycle_parked(
|
||||
agent,
|
||||
coordinator,
|
||||
agent_id,
|
||||
input_data=input_data,
|
||||
run_config=run_config,
|
||||
context=context,
|
||||
max_turns=max_turns,
|
||||
session=session,
|
||||
event_sink=event_sink,
|
||||
hooks=hooks,
|
||||
)
|
||||
else:
|
||||
result = await _run_cycle(
|
||||
agent,
|
||||
coordinator,
|
||||
agent_id,
|
||||
input_data=input_data,
|
||||
run_config=run_config,
|
||||
context=context,
|
||||
max_turns=max_turns,
|
||||
session=session,
|
||||
interactive=False,
|
||||
event_sink=event_sink,
|
||||
hooks=hooks,
|
||||
)
|
||||
|
||||
status = await _agent_status(coordinator, agent_id)
|
||||
if status != "running":
|
||||
await coordinator.reset_recovery(agent_id)
|
||||
return result
|
||||
|
||||
recoveries = await coordinator.record_recovery(agent_id)
|
||||
logger.warning(
|
||||
"agent %s ended a turn without a lifecycle tool call (interactive=%s); "
|
||||
"forcing tool continuation (%d/%d): %s",
|
||||
agent_id,
|
||||
interactive,
|
||||
recoveries,
|
||||
recovery_limit,
|
||||
_final_output_preview(result),
|
||||
)
|
||||
|
||||
if recoveries >= recovery_limit:
|
||||
return await _exhausted_recovery(coordinator, agent_id, result, interactive=interactive)
|
||||
|
||||
input_data = await _append_tool_required_message(
|
||||
session=session,
|
||||
context=context,
|
||||
attempt=recoveries,
|
||||
limit=recovery_limit,
|
||||
interactive=interactive,
|
||||
)
|
||||
|
||||
|
||||
async def _exhausted_recovery(
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
result: RunResultBase | None,
|
||||
*,
|
||||
interactive: bool,
|
||||
) -> RunResultBase | None:
|
||||
"""Settle an agent that never recovered into a tool call.
|
||||
|
||||
Interactive runs park instead of dying: a human is attached and can message
|
||||
any agent, so the scan stays resumable. Autonomous runs have nobody to
|
||||
resume them, so they fail loudly.
|
||||
"""
|
||||
if not interactive:
|
||||
await coordinator.set_status(agent_id, "crashed")
|
||||
await _notify_parent_on_terminal(coordinator, agent_id, "crashed")
|
||||
raise MaxTurnsExceeded(
|
||||
"Agent exhausted recovery attempts without calling finish_scan or agent_finish."
|
||||
)
|
||||
|
||||
logger.warning(
|
||||
"agent %s exhausted tool-call recovery attempts; parking until a message arrives",
|
||||
agent_id,
|
||||
)
|
||||
await coordinator.park_waiting(agent_id, wait_kind="stalled")
|
||||
# A parked child owes its parent a completion report it can no longer send. The
|
||||
# parent is an agent, not a watching human, so nothing else tells it to stop
|
||||
# waiting and it burns its full timeout on a message that is never coming.
|
||||
await _notify_parent_on_stall(coordinator, agent_id)
|
||||
return result
|
||||
|
||||
|
||||
_WAITING_AUTO_RESUME_TIMEOUT_S = 300.0
|
||||
|
||||
# An agent that parks again after every auto-resume makes no progress, so stop
|
||||
# spending a model turn per timeout and leave it parked for a real message.
|
||||
_MAX_IDLE_AUTO_RESUMES = 3
|
||||
|
||||
|
||||
async def _plain_waiting_timeout(
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
) -> float | None:
|
||||
"""Auto-resume timeout for a parked agent; None waits until a message arrives.
|
||||
|
||||
Driven by what the agent is waiting on, not by where it sits in the graph:
|
||||
the user can message any agent, so an agent awaiting a human parks
|
||||
indefinitely whether or not it is the root. Only an agent awaiting other
|
||||
agents is re-checked on a timer, and only until it has spent its idle
|
||||
budget re-parking without hearing anything.
|
||||
"""
|
||||
async with coordinator._lock:
|
||||
status = coordinator.statuses.get(agent_id)
|
||||
has_error = agent_id in coordinator.errors
|
||||
runtime = coordinator.runtimes.get(agent_id)
|
||||
gated = runtime.user_wake_required if runtime is not None else False
|
||||
wait_kind = coordinator.wait_kinds.get(agent_id)
|
||||
idle_resumes = coordinator.idle_resume_counts.get(agent_id, 0)
|
||||
if status != "waiting" or has_error or gated:
|
||||
return None
|
||||
if wait_kind != "agents" or idle_resumes >= _MAX_IDLE_AUTO_RESUMES:
|
||||
return None
|
||||
return _WAITING_AUTO_RESUME_TIMEOUT_S
|
||||
|
||||
|
||||
async def _run_cycle_parked(
|
||||
agent: Any,
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
*,
|
||||
input_data: Any,
|
||||
run_config: RunConfig,
|
||||
context: dict[str, Any],
|
||||
max_turns: int,
|
||||
session: Session | None,
|
||||
event_sink: StreamEventSink | None,
|
||||
hooks: RunHooks[dict[str, Any]] | None,
|
||||
) -> RunResultBase | None:
|
||||
"""Interactive run cycle that parks on any error instead of killing the runner."""
|
||||
try:
|
||||
return await _run_cycle(
|
||||
agent,
|
||||
coordinator,
|
||||
agent_id,
|
||||
@@ -404,39 +613,17 @@ async def _run_noninteractive_until_lifecycle(
|
||||
context=context,
|
||||
max_turns=max_turns,
|
||||
session=session,
|
||||
interactive=False,
|
||||
interactive=True,
|
||||
event_sink=event_sink,
|
||||
hooks=hooks,
|
||||
)
|
||||
|
||||
status = await _agent_status(coordinator, agent_id)
|
||||
if status != "running":
|
||||
return result
|
||||
|
||||
invalid_final_outputs += 1
|
||||
logger.warning(
|
||||
"agent %s produced non-lifecycle final output in non-interactive mode; "
|
||||
"forcing tool continuation (%d/%d): %s",
|
||||
agent_id,
|
||||
invalid_final_outputs,
|
||||
invalid_final_output_limit,
|
||||
_final_output_preview(result),
|
||||
)
|
||||
|
||||
if invalid_final_outputs >= invalid_final_output_limit:
|
||||
await coordinator.set_status(agent_id, "crashed")
|
||||
await _notify_parent_on_terminal(coordinator, agent_id, "crashed")
|
||||
raise MaxTurnsExceeded(
|
||||
"Agent exhausted non-interactive recovery attempts without calling "
|
||||
"finish_scan or agent_finish."
|
||||
)
|
||||
|
||||
input_data = await _append_noninteractive_tool_required_message(
|
||||
session=session,
|
||||
context=context,
|
||||
attempt=invalid_final_outputs,
|
||||
limit=invalid_final_output_limit,
|
||||
)
|
||||
except (BudgetExceededError, BudgetPausedError, SubagentBudgetReservedError):
|
||||
raise
|
||||
except Exception as exc:
|
||||
logger.exception("error escaped the run cycle for %s; parking as failed", agent_id)
|
||||
await coordinator.set_status(agent_id, "failed", error=str(exc) or type(exc).__name__)
|
||||
await _notify_parent_on_terminal(coordinator, agent_id, "failed")
|
||||
return None
|
||||
|
||||
|
||||
async def _run_cycle( # noqa: PLR0912, PLR0915
|
||||
@@ -457,6 +644,8 @@ async def _run_cycle( # noqa: PLR0912, PLR0915
|
||||
compactions = 0
|
||||
model_retries = 0
|
||||
while True:
|
||||
stream: Any = None
|
||||
pre_run_items: list[Any] = []
|
||||
try:
|
||||
await coordinator.mark_running(agent_id)
|
||||
if session is not None:
|
||||
@@ -470,6 +659,8 @@ async def _run_cycle( # noqa: PLR0912, PLR0915
|
||||
await _compact_session(agent, session, run_config, force=False)
|
||||
except Exception:
|
||||
logger.exception("proactive compaction failed for %s", agent_id)
|
||||
with contextlib.suppress(Exception):
|
||||
pre_run_items = list(await session.get_items())
|
||||
stream = Runner.run_streamed(
|
||||
agent,
|
||||
input=input_data,
|
||||
@@ -490,6 +681,8 @@ async def _run_cycle( # noqa: PLR0912, PLR0915
|
||||
logger.exception("stream event sink failed for %s", agent_id)
|
||||
if stream.run_loop_exception is not None:
|
||||
raise stream.run_loop_exception
|
||||
if refusal := _structured_provider_refusal(stream):
|
||||
raise ProviderRefusalError(refusal)
|
||||
except (BudgetExceededError, BudgetPausedError, SubagentBudgetReservedError):
|
||||
raise
|
||||
except RuntimeError as stream_exc:
|
||||
@@ -580,10 +773,13 @@ async def _run_cycle( # noqa: PLR0912, PLR0915
|
||||
if session is not None:
|
||||
input_data = []
|
||||
continue
|
||||
if codex.is_content_guardrail_error(exc):
|
||||
return await _handle_content_guardrail(
|
||||
coordinator, agent_id, exc, interactive=interactive
|
||||
)
|
||||
if session is not None:
|
||||
await _salvage_stream_to_session(session, pre_run_items, stream, agent_id)
|
||||
if isinstance(exc, ProviderRefusalError):
|
||||
logger.warning("agent %s refused by the model provider: %s", agent_id, exc)
|
||||
await coordinator.set_status(agent_id, "failed", error=str(exc))
|
||||
await _notify_parent_on_terminal(coordinator, agent_id, "failed")
|
||||
return None
|
||||
if not interactive:
|
||||
raise
|
||||
if isinstance(exc, MaxTurnsExceeded):
|
||||
@@ -597,41 +793,7 @@ async def _run_cycle( # noqa: PLR0912, PLR0915
|
||||
await _notify_parent_on_terminal(coordinator, agent_id, status)
|
||||
return None
|
||||
else:
|
||||
await _settle_run_result(coordinator, agent_id, interactive)
|
||||
return stream
|
||||
|
||||
|
||||
async def _handle_content_guardrail(
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
exc: BaseException,
|
||||
*,
|
||||
interactive: bool,
|
||||
) -> RunResultBase | None:
|
||||
logger.warning("agent %s blocked by the model's content guardrail: %s", agent_id, exc)
|
||||
if interactive:
|
||||
await coordinator.set_status(agent_id, "waiting", error=_GUARDRAIL_PARK_ERROR)
|
||||
return None
|
||||
await coordinator.set_status(agent_id, "failed", error=_GUARDRAIL_PARK_ERROR)
|
||||
await _notify_parent_on_terminal(coordinator, agent_id, "failed")
|
||||
return None
|
||||
|
||||
|
||||
async def _settle_run_result(
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
interactive: bool,
|
||||
) -> None:
|
||||
async with coordinator._lock:
|
||||
current_status = coordinator.statuses.get(agent_id)
|
||||
|
||||
if current_status != "running":
|
||||
return
|
||||
|
||||
if not interactive:
|
||||
return
|
||||
|
||||
await coordinator.set_status(agent_id, "waiting")
|
||||
return cast("RunResultBase | None", stream)
|
||||
|
||||
|
||||
async def _agent_status(coordinator: AgentCoordinator, agent_id: str) -> Status | None:
|
||||
@@ -649,23 +811,37 @@ def _final_output_preview(result: RunResultBase | None) -> str:
|
||||
return text[:300]
|
||||
|
||||
|
||||
async def _append_noninteractive_tool_required_message(
|
||||
async def _append_tool_required_message(
|
||||
*,
|
||||
session: Session | None,
|
||||
context: dict[str, Any],
|
||||
attempt: int,
|
||||
limit: int,
|
||||
interactive: bool,
|
||||
) -> list[dict[str, str]]:
|
||||
finish_tool = "finish_scan" if context.get("parent_id") is None else "agent_finish"
|
||||
message = (
|
||||
"Your previous response ended the autonomous Strix run without a lifecycle tool call. "
|
||||
"That is invalid in non-interactive mode; plain text final answers are ignored. "
|
||||
"Continue immediately and call exactly one tool. "
|
||||
f"If your work is complete, call {finish_tool}. "
|
||||
"If you are blocked waiting for another agent, call wait_for_message. "
|
||||
"Otherwise use the appropriate execution or planning tool. "
|
||||
f"This is recovery attempt {attempt}/{limit}."
|
||||
)
|
||||
if interactive:
|
||||
message = (
|
||||
"Your previous message ended a turn without a tool call. Plain text never ends "
|
||||
"execution and never hands control to the user: it is shown to the user, and the "
|
||||
"run continues. Continue immediately and call exactly one tool. "
|
||||
"If you have something to tell the user and nothing to do until they reply, "
|
||||
"call respond_to_user. "
|
||||
"If you are blocked waiting for another agent, call wait_for_agents. "
|
||||
f"If the whole engagement is complete, call {finish_tool}. "
|
||||
"Otherwise use the appropriate execution or planning tool. "
|
||||
f"This is recovery attempt {attempt}/{limit}."
|
||||
)
|
||||
else:
|
||||
message = (
|
||||
"Your previous response ended the autonomous Strix run without a lifecycle tool "
|
||||
"call. That is invalid in non-interactive mode; plain text final answers are "
|
||||
"ignored. Continue immediately and call exactly one tool. "
|
||||
f"If your work is complete, call {finish_tool}. "
|
||||
"If you are blocked waiting for another agent, call wait_for_agents. "
|
||||
"Otherwise use the appropriate execution or planning tool. "
|
||||
f"This is recovery attempt {attempt}/{limit}."
|
||||
)
|
||||
item = {"role": "user", "content": message}
|
||||
if session is None:
|
||||
return [item]
|
||||
@@ -692,6 +868,36 @@ _TERMINAL_NOTICE = {
|
||||
}
|
||||
|
||||
|
||||
_STALL_NOTICE = (
|
||||
"[Agent stalled] {name} ({agent_id}) kept ending turns without a tool call and is "
|
||||
"parked until it receives a message. It will not send a completion report on its "
|
||||
"own: either message it with a concrete next step to unblock it, or stop waiting on "
|
||||
"it and account for its unfinished subtask."
|
||||
)
|
||||
|
||||
|
||||
async def _notify_parent_on_stall(
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
) -> None:
|
||||
"""Tell the parent that a child parked mid-task, so it stops waiting blindly."""
|
||||
async with coordinator._lock:
|
||||
parent = coordinator.parent_of.get(agent_id)
|
||||
name = coordinator.names.get(agent_id, agent_id)
|
||||
if parent is None:
|
||||
return
|
||||
await coordinator.send(
|
||||
parent,
|
||||
{
|
||||
"from": agent_id,
|
||||
"type": "stalled",
|
||||
"priority": "high",
|
||||
"content": _STALL_NOTICE.format(name=name, agent_id=agent_id),
|
||||
},
|
||||
interrupt=False,
|
||||
)
|
||||
|
||||
|
||||
async def _notify_parent_on_terminal(
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
|
||||
@@ -377,12 +377,6 @@ async def run_strix_scan(
|
||||
|
||||
async with coordinator._lock:
|
||||
root_status = coordinator.statuses.get(root_id)
|
||||
root_error = coordinator.errors.get(root_id)
|
||||
|
||||
root_recoverable_park = root_status == "waiting" and bool(root_error)
|
||||
root_start_parked = bool(
|
||||
interactive and is_resume and root_status != "running" and not root_recoverable_park
|
||||
)
|
||||
|
||||
result = await run_agent_loop(
|
||||
agent=root_agent,
|
||||
@@ -394,7 +388,7 @@ async def run_strix_scan(
|
||||
agent_id=root_id,
|
||||
interactive=interactive,
|
||||
session=root_session,
|
||||
start_parked=root_start_parked,
|
||||
start_parked=bool(interactive and is_resume and root_status != "running"),
|
||||
event_sink=event_sink,
|
||||
hooks=hooks,
|
||||
)
|
||||
|
||||
@@ -7,6 +7,7 @@ import logging
|
||||
from typing import TYPE_CHECKING, Any, cast
|
||||
from weakref import WeakKeyDictionary
|
||||
|
||||
from agents.items import ItemHelpers
|
||||
from agents.memory import SQLiteSession
|
||||
|
||||
|
||||
@@ -26,6 +27,18 @@ def open_agent_session(agent_id: str, path: Path) -> SQLiteSession:
|
||||
return SQLiteSession(session_id=agent_id, db_path=path)
|
||||
|
||||
|
||||
async def seed_initial_input(session: Session, initial_input: Any) -> bool:
|
||||
"""Commit an agent's opening identity/task input before its first run cycle."""
|
||||
items = ItemHelpers.input_to_new_input_list(initial_input)
|
||||
if not items:
|
||||
return False
|
||||
async with session_write_lock(session):
|
||||
if await session.get_items():
|
||||
return False
|
||||
await session.add_items(items)
|
||||
return True
|
||||
|
||||
|
||||
_IMAGE_REJECTED_TEXT = "[image rejected by the model]"
|
||||
_IMAGE_ELIDED_TEXT = "[older screenshot elided to bound context memory]"
|
||||
_INHERITED_IMAGE_TEXT = "[screenshot omitted from inherited context]"
|
||||
|
||||
@@ -1041,9 +1041,9 @@ class StrixTUIApp(App): # type: ignore[misc]
|
||||
name=names.get(agent_id, agent_id),
|
||||
parent_id=parent_of.get(agent_id),
|
||||
status=status,
|
||||
error_message=error,
|
||||
error_message=error or "",
|
||||
)
|
||||
if status in {"failed", "crashed"} and error:
|
||||
if error:
|
||||
if agent_id not in self._error_noted_agents:
|
||||
self._error_noted_agents.add(agent_id)
|
||||
self.live_view.record_agent_error(agent_id, error)
|
||||
@@ -1293,6 +1293,10 @@ class StrixTUIApp(App): # type: ignore[misc]
|
||||
text.append("Send a message to continue", style="dim")
|
||||
keymap = keymap_styled([("ctrl-q", "quit")])
|
||||
else:
|
||||
error_msg = agent_data.get("error_message") or ""
|
||||
if error_msg:
|
||||
text.append(error_msg, style="red")
|
||||
text.append(" \u00b7 ", style="dim")
|
||||
text.append("Send message to resume", style="dim")
|
||||
return (text, keymap, False)
|
||||
|
||||
|
||||
@@ -82,7 +82,7 @@ class TuiLiveView:
|
||||
current["parent_id"] = parent_id
|
||||
if status is not None:
|
||||
current["status"] = status
|
||||
if error_message:
|
||||
if error_message is not None:
|
||||
current["error_message"] = error_message
|
||||
current["updated_at"] = now
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ from . import (
|
||||
notes_renderer,
|
||||
proxy_renderer,
|
||||
reporting_renderer,
|
||||
respond_renderer,
|
||||
shell_renderer,
|
||||
thinking_renderer,
|
||||
todo_renderer,
|
||||
@@ -23,6 +24,7 @@ __all__ = [
|
||||
"proxy_renderer",
|
||||
"render_tool_widget",
|
||||
"reporting_renderer",
|
||||
"respond_renderer",
|
||||
"shell_renderer",
|
||||
"thinking_renderer",
|
||||
"todo_renderer",
|
||||
|
||||
@@ -117,8 +117,8 @@ class AgentFinishRenderer(BaseToolRenderer):
|
||||
|
||||
|
||||
@register_tool_renderer
|
||||
class WaitForMessageRenderer(BaseToolRenderer):
|
||||
tool_name: ClassVar[str] = "wait_for_message"
|
||||
class WaitForAgentsRenderer(BaseToolRenderer):
|
||||
tool_name: ClassVar[str] = "wait_for_agents"
|
||||
css_classes: ClassVar[list[str]] = ["tool-call", "agents-graph-tool"]
|
||||
|
||||
@classmethod
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
from typing import Any, ClassVar
|
||||
|
||||
from rich.text import Text
|
||||
from textual.widgets import Static
|
||||
|
||||
from .agent_message_renderer import AgentMessageRenderer
|
||||
from .base_renderer import BaseToolRenderer
|
||||
from .registry import register_tool_renderer
|
||||
|
||||
|
||||
@register_tool_renderer
|
||||
class RespondToUserRenderer(BaseToolRenderer):
|
||||
"""Render a reply as the agent's own prose, not as a tool call.
|
||||
|
||||
``respond_to_user`` carries the message the user is meant to read, so it
|
||||
gets the same markdown treatment as a plain assistant turn.
|
||||
"""
|
||||
|
||||
tool_name: ClassVar[str] = "respond_to_user"
|
||||
css_classes: ClassVar[list[str]] = ["tool-call", "respond-tool"]
|
||||
|
||||
@classmethod
|
||||
def render(cls, tool_data: dict[str, Any]) -> Static:
|
||||
args = tool_data.get("args", {})
|
||||
message = args.get("message", "")
|
||||
|
||||
text = Text()
|
||||
if message:
|
||||
text.append_text(AgentMessageRenderer.render_simple(message))
|
||||
text.append("\n\n")
|
||||
text.append("○ ", style="#6b7280")
|
||||
text.append("waiting for your reply", style="dim")
|
||||
|
||||
css_classes = cls.get_css_classes(tool_data.get("status", "unknown"))
|
||||
return Static(text, classes=css_classes)
|
||||
+1
-1
@@ -54,7 +54,7 @@ export default function AgentCommsRenderer({ toolName, args }: ToolRendererProps
|
||||
);
|
||||
}
|
||||
|
||||
if (toolName === "wait_for_message") {
|
||||
if (toolName === "wait_for_agents") {
|
||||
const reason = (args.reason as string) ?? "";
|
||||
return (
|
||||
<div className="flex items-center gap-2">
|
||||
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
"use client";
|
||||
|
||||
import type { ToolRendererProps } from "@/types/events";
|
||||
import Markdown from "./Markdown";
|
||||
|
||||
/**
|
||||
* `respond_to_user` carries the message the user is meant to read, so it renders
|
||||
* as the agent's own prose rather than as a tool call.
|
||||
*/
|
||||
export default function RespondRenderer({ args }: ToolRendererProps) {
|
||||
const message = (args.message as string) ?? "";
|
||||
if (!message) return null;
|
||||
|
||||
return (
|
||||
<div>
|
||||
<Markdown text={message} />
|
||||
<div className="mt-1.5 text-[#888] text-[13px]">waiting for your reply</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -24,6 +24,7 @@ import NotesRenderer from "./NotesRenderer";
|
||||
import TodoRenderer from "./TodoRenderer";
|
||||
import FallbackRenderer from "./FallbackRenderer";
|
||||
import LoadSkillRenderer from "./LoadSkillRenderer";
|
||||
import RespondRenderer from "./RespondRenderer";
|
||||
|
||||
/**
|
||||
* Tool-renderer mapping — data-driven, keyed by the engine's tool *family*.
|
||||
@@ -104,10 +105,10 @@ const CATEGORY_TOOLS: Record<ToolCategory, readonly string[]> = {
|
||||
proxy: ["list_requests", "view_request", "repeat_request", "list_sitemap", "view_sitemap_entry", "scope_rules", "send_request"],
|
||||
reporting: ["create_vulnerability_report", "list_reports", "get_report"],
|
||||
thinking: ["think"],
|
||||
agents: ["create_agent", "agent_finish", "send_message_to_agent", "wait_for_message", "view_agent_graph", "stop_agent"],
|
||||
agents: ["create_agent", "agent_finish", "send_message_to_agent", "wait_for_agents", "view_agent_graph", "stop_agent"],
|
||||
search: ["web_search"],
|
||||
// scan_start_info / subagent_start_info are strix-app synthetic events; finish_scan is the engine's
|
||||
lifecycle: ["scan_start_info", "subagent_start_info", "finish_scan"],
|
||||
lifecycle: ["scan_start_info", "subagent_start_info", "finish_scan", "respond_to_user"],
|
||||
notes: ["create_note", "delete_note", "update_note", "list_notes", "get_note"],
|
||||
skills: ["load_skill"],
|
||||
todos: ["create_todo", "list_todos", "update_todo", "mark_todo_done", "mark_todo_pending", "delete_todo"],
|
||||
@@ -127,6 +128,7 @@ const TOOL_CATEGORY: Record<string, ToolCategory> = Object.fromEntries(
|
||||
*/
|
||||
const RENDERER_OVERRIDES: Partial<Record<string, ComponentType<ToolRendererProps>>> = {
|
||||
finish_scan: FinishRenderer,
|
||||
respond_to_user: RespondRenderer,
|
||||
apply_patch: ApplyPatchRenderer,
|
||||
view_image: ViewImageRenderer,
|
||||
list_reports: ReportListRenderer,
|
||||
@@ -140,7 +142,8 @@ const RENDERER_OVERRIDES: Partial<Record<string, ComponentType<ToolRendererProps
|
||||
const ICON_OVERRIDES: Partial<Record<string, ToolIconMeta>> = {
|
||||
agent_finish: { icon: Flag, color: "text-cyan-400" },
|
||||
send_message_to_agent: { icon: MessageCircle, color: "text-cyan-400" },
|
||||
wait_for_message: { icon: MessageCircle, color: "text-cyan-400" },
|
||||
wait_for_agents: { icon: MessageCircle, color: "text-cyan-400" },
|
||||
respond_to_user: { icon: MessageCircle, color: "text-emerald-400" },
|
||||
view_agent_graph: { icon: Eye, color: "text-cyan-400" },
|
||||
stop_agent: { icon: Ban, color: "text-red-400" },
|
||||
scan_start_info: { icon: Crosshair, color: "text-emerald-400" },
|
||||
|
||||
+50
-50
File diff suppressed because one or more lines are too long
@@ -6,7 +6,7 @@
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<meta name="color-scheme" content="dark" />
|
||||
<title>Strix Results</title>
|
||||
<script type="module" crossorigin src="./assets/index-DzvI_0HX.js"></script>
|
||||
<script type="module" crossorigin src="./assets/index-CGvQq6oe.js"></script>
|
||||
<link rel="stylesheet" crossorigin href="./assets/index-C3kQ5kk8.css">
|
||||
</head>
|
||||
<body>
|
||||
|
||||
@@ -218,25 +218,43 @@ def _session_items_payload(items: list[Any]) -> list[dict[str, Any]]:
|
||||
return payload
|
||||
|
||||
|
||||
@function_tool(timeout=601)
|
||||
async def wait_for_message( # noqa: PLR0911
|
||||
_WAIT_DEFAULT_TIMEOUT_S = 300
|
||||
# Enforced by the SDK around the whole tool call, so it caps an oversized
|
||||
# ``timeout_seconds`` the model asks for. One second of headroom lets the
|
||||
# tool's own timeout fire first and return a clean result.
|
||||
_WAIT_HARD_CEILING_S = _WAIT_DEFAULT_TIMEOUT_S + 1
|
||||
|
||||
|
||||
@function_tool(timeout=_WAIT_HARD_CEILING_S)
|
||||
async def wait_for_agents( # noqa: PLR0911
|
||||
ctx: RunContextWrapper,
|
||||
reason: str = "Waiting for messages from other agents",
|
||||
timeout_seconds: int = 600,
|
||||
timeout_seconds: int = _WAIT_DEFAULT_TIMEOUT_S,
|
||||
) -> str:
|
||||
"""Pause this agent until a message lands in its inbox (or timeout).
|
||||
"""Pause until another AGENT messages you (or the timeout elapses).
|
||||
|
||||
Use when you have nothing useful to do until a child/peer responds
|
||||
— typically after spawning subagents and you want to wait for
|
||||
their completion reports. The agent automatically resumes when any
|
||||
message arrives, so pick a ``timeout_seconds`` proportional to the
|
||||
work you're awaiting.
|
||||
Use when you have nothing useful to do until a child or peer
|
||||
responds — typically after spawning subagents and you want their
|
||||
completion reports. You resume the instant any message arrives, so
|
||||
size ``timeout_seconds`` to the work you're awaiting.
|
||||
|
||||
**This tool is only for waiting on other agents.** Two things it is
|
||||
NOT for:
|
||||
|
||||
- **Talking to the user.** Use ``respond_to_user``, which delivers
|
||||
your message and hands control back in one call.
|
||||
- **Waiting for a long-running command.** This tool does not watch
|
||||
processes at all — it sleeps until a *message* arrives, so it
|
||||
burns the full timeout even if your command finished a second
|
||||
later. Poll the process instead: ``exec_command`` returns a
|
||||
session/process id, and ``write_stdin`` with ``chars=""`` returns
|
||||
as soon as there is new output or the process exits.
|
||||
|
||||
**Critical caveats:**
|
||||
|
||||
- **Never** call this if you finished your own task and have **no**
|
||||
child agents running — that's a permanent stall. Call
|
||||
``finish_scan`` (root) or ``agent_finish`` (subagent) instead.
|
||||
- **Never** call this if you have no agents left to hear from —
|
||||
that just strands you until the timeout. Call ``finish_scan``
|
||||
(root) or ``agent_finish`` (subagent) instead.
|
||||
- If you're waiting on an agent that **isn't your child**, message
|
||||
it first asking it to ping you when done — otherwise it has no
|
||||
reason to send to your inbox and you'll wait the full timeout.
|
||||
@@ -247,7 +265,8 @@ async def wait_for_message( # noqa: PLR0911
|
||||
reason: One-line note shown in graph snapshots while you're
|
||||
waiting (helps a human or sibling agent debug who's stuck
|
||||
on what).
|
||||
timeout_seconds: Max seconds to wait (default 600). This is only
|
||||
timeout_seconds: Max seconds to wait (default 300, and values above
|
||||
that are cut short by a hard ceiling). This is only
|
||||
a cap — the tool returns the INSTANT a message arrives, so a
|
||||
larger value never makes you wait longer when the reply does
|
||||
come. Right-size it to what you're waiting on: a short wait
|
||||
@@ -257,9 +276,7 @@ async def wait_for_message( # noqa: PLR0911
|
||||
bites when the expected message never arrives — so an oversized
|
||||
timeout on a trivial wait just strands you idle until it
|
||||
elapses. On timeout the tool returns and you decide whether to
|
||||
keep working or wait again. (Applies to autonomous multi-agent
|
||||
runs; in interactive/chat sessions the agent instead parks until
|
||||
a message arrives and this cap is not enforced.)
|
||||
keep working or wait again.
|
||||
"""
|
||||
inner = _ctx(ctx)
|
||||
coordinator = coordinator_from_context(inner)
|
||||
@@ -302,7 +319,7 @@ async def wait_for_message( # noqa: PLR0911
|
||||
)
|
||||
|
||||
if interactive:
|
||||
await coordinator.park_waiting(me)
|
||||
await coordinator.park_waiting(me, wait_kind="agents")
|
||||
return json.dumps(
|
||||
{
|
||||
"success": True,
|
||||
@@ -314,7 +331,7 @@ async def wait_for_message( # noqa: PLR0911
|
||||
default=str,
|
||||
)
|
||||
|
||||
await coordinator.park_waiting(me)
|
||||
await coordinator.park_waiting(me, wait_kind="agents")
|
||||
try:
|
||||
await asyncio.wait_for(coordinator.wait_for_message(me), timeout_seconds)
|
||||
except TimeoutError:
|
||||
@@ -373,7 +390,7 @@ async def create_agent(
|
||||
|
||||
Decompose complex pentests by handing focused subtasks to dedicated
|
||||
children. The child runs asynchronously — the parent continues
|
||||
immediately and can ``wait_for_message`` later (or just keep
|
||||
immediately and can ``wait_for_agents`` later (or just keep
|
||||
working in parallel). When the child calls ``agent_finish``, its
|
||||
completion report lands in the parent's inbox.
|
||||
|
||||
|
||||
@@ -101,7 +101,7 @@ async def finish_scan(
|
||||
execution stops. There is no draft mode and no second chance: never
|
||||
submit placeholder, provisional, or "checking if done" text in any
|
||||
field, and never call ``finish_scan`` to poll whether subagents are
|
||||
done (use ``view_agent_graph`` / ``wait_for_message`` for that).
|
||||
done (use ``view_agent_graph`` / ``wait_for_agents`` for that).
|
||||
Call it exactly ONCE, only when every field holds genuine, finished
|
||||
assessment prose.
|
||||
|
||||
@@ -111,7 +111,7 @@ async def finish_scan(
|
||||
summary. If ANY agent is in ``running`` / ``waiting`` state,
|
||||
you MUST NOT call ``finish_scan`` yet —
|
||||
wrap them up first via ``send_message_to_agent`` (ask them to
|
||||
finish), ``wait_for_message`` (block until their report
|
||||
finish), ``wait_for_agents`` (block until their report
|
||||
arrives), or ``stop_agent`` (graceful cancel). Only ``completed``
|
||||
/ ``crashed`` / ``stopped`` agents are safe to leave behind.
|
||||
Calling ``finish_scan`` while children are alive orphans their
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
"""User-facing reply tool for interactive sessions."""
|
||||
|
||||
from strix.tools.respond.tool import respond_to_user
|
||||
|
||||
|
||||
__all__ = ["respond_to_user"]
|
||||
@@ -0,0 +1,110 @@
|
||||
"""``respond_to_user`` — deliver a reply and hand control back to the user."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from agents import RunContextWrapper, function_tool
|
||||
|
||||
from strix.core.agents import coordinator_from_context
|
||||
|
||||
|
||||
def _ctx(ctx: RunContextWrapper) -> dict[str, Any]:
|
||||
return ctx.context if isinstance(ctx.context, dict) else {}
|
||||
|
||||
|
||||
@function_tool
|
||||
async def respond_to_user(ctx: RunContextWrapper, message: str) -> str:
|
||||
"""Answer the user and hand control back to them.
|
||||
|
||||
This is the ONLY way to yield to the user. Delivering the message and
|
||||
yielding are the same call on purpose: there is no way to answer and
|
||||
then forget to stop, and no way to stop without having answered.
|
||||
|
||||
Call it when you have something for the user and nothing to do until
|
||||
they reply — you answered their question, you need a decision or a
|
||||
credential only they can give, or you finished a chunk of work and
|
||||
want direction. You resume exactly where you left off when they
|
||||
reply, with everything you have done so far intact.
|
||||
|
||||
Do NOT call it to narrate progress or to think out loud. Plain text
|
||||
is still shown to the user as you work, so say whatever you like
|
||||
mid-task without stopping; ``respond_to_user`` is specifically the
|
||||
act of *waiting* for them. Every call costs the user their attention.
|
||||
|
||||
Not for these:
|
||||
|
||||
- **Waiting on another agent** (a child's report, a peer's reply) —
|
||||
use ``wait_for_agents``.
|
||||
- **Ending the engagement** — use ``finish_scan`` (root) or
|
||||
``agent_finish`` (subagent). Those are terminal; this is a pause.
|
||||
|
||||
Args:
|
||||
message: What to say to the user. Self-contained: they may not
|
||||
have followed the tool calls that led here. Lead with the
|
||||
answer or the decision you need, and if you are blocked, say
|
||||
exactly what you need from them.
|
||||
"""
|
||||
inner = _ctx(ctx)
|
||||
coordinator = coordinator_from_context(inner)
|
||||
me = inner.get("agent_id")
|
||||
interactive = bool(inner.get("interactive", False))
|
||||
|
||||
if coordinator is None or me is None:
|
||||
return json.dumps(
|
||||
{"success": False, "error": "Agent coordinator or agent_id missing in context"},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
|
||||
if not interactive:
|
||||
return json.dumps(
|
||||
{
|
||||
"success": False,
|
||||
"error": (
|
||||
"No user is attached to an autonomous run. Keep working, and call "
|
||||
"finish_scan (root) or agent_finish (subagent) when the task is done."
|
||||
),
|
||||
},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
|
||||
async with coordinator._lock:
|
||||
stopped = coordinator.statuses.get(me) == "stopped"
|
||||
if stopped:
|
||||
return json.dumps(
|
||||
{"success": True, "wait_outcome": "stopped", "message": message},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
|
||||
# A message that arrived while this turn was running is the user already
|
||||
# talking: take it now instead of parking for one they have sent.
|
||||
pending, _ = await coordinator.consume_pending(me)
|
||||
if pending > 0:
|
||||
await coordinator.mark_running(me)
|
||||
return json.dumps(
|
||||
{
|
||||
"success": True,
|
||||
"wait_outcome": "message_arrived",
|
||||
"pending_messages": pending,
|
||||
"message": message,
|
||||
"note": "Your reply was delivered; the user had already sent a new message.",
|
||||
},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
|
||||
await coordinator.park_waiting(me, wait_kind="user")
|
||||
return json.dumps(
|
||||
{
|
||||
"success": True,
|
||||
"wait_outcome": "waiting",
|
||||
"message": message,
|
||||
"note": "Reply delivered; parked until the user responds.",
|
||||
},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
@@ -2,18 +2,29 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import pytest
|
||||
from agents.tool import FunctionTool
|
||||
|
||||
from strix.agents import factory
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from agents.tool_context import ToolContext
|
||||
|
||||
|
||||
def _tool(name: str) -> FunctionTool:
|
||||
# A per-tool closure keeps two same-named tools unequal, which is what the
|
||||
# duplicate-name tests exercise.
|
||||
async def invoke(_ctx: ToolContext[Any], _input: str) -> str:
|
||||
return "ok"
|
||||
|
||||
return FunctionTool(
|
||||
name=name,
|
||||
description="test tool",
|
||||
params_json_schema={"type": "object", "properties": {}, "additionalProperties": False},
|
||||
on_invoke_tool=lambda _ctx, _inp: "ok",
|
||||
on_invoke_tool=invoke,
|
||||
)
|
||||
|
||||
|
||||
@@ -86,3 +97,18 @@ def test_no_override_renders_builtin_prompt() -> None:
|
||||
|
||||
assert isinstance(agent.instructions, str)
|
||||
assert agent.instructions != ""
|
||||
|
||||
|
||||
def test_respond_to_user_is_interactive_only() -> None:
|
||||
"""Yielding to the user is meaningless when no user is attached."""
|
||||
interactive = factory.build_strix_agent(is_root=True, interactive=True)
|
||||
autonomous = factory.build_strix_agent(is_root=True, interactive=False)
|
||||
|
||||
assert "respond_to_user" in [t.name for t in interactive.tools]
|
||||
assert "respond_to_user" not in [t.name for t in autonomous.tools]
|
||||
|
||||
|
||||
def test_wait_for_agents_is_available_in_both_modes() -> None:
|
||||
for interactive in (True, False):
|
||||
agent = factory.build_strix_agent(is_root=True, interactive=interactive)
|
||||
assert "wait_for_agents" in [t.name for t in agent.tools]
|
||||
|
||||
@@ -7,7 +7,7 @@ from unittest.mock import MagicMock, patch
|
||||
import pytest
|
||||
|
||||
from strix.core import execution
|
||||
from strix.core.agents import AgentCoordinator
|
||||
from strix.core.agents import AgentCoordinator, WaitKind
|
||||
from strix.core.execution import _start_child_runner, run_agent_loop
|
||||
from strix.core.hooks import BudgetExceededError, ReportUsageHooks
|
||||
from strix.core.sessions import open_agent_session
|
||||
@@ -42,23 +42,32 @@ class _FakeStream:
|
||||
hooks: ReportUsageHooks,
|
||||
context: dict[str, Any],
|
||||
agent: Any,
|
||||
coordinator: AgentCoordinator,
|
||||
) -> None:
|
||||
self._ledger = ledger
|
||||
self._hooks = hooks
|
||||
self._context = context
|
||||
self._agent = agent
|
||||
self._coordinator = coordinator
|
||||
self.run_loop_exception: BaseException | None = None
|
||||
self.final_output = None
|
||||
|
||||
async def stream_events(self) -> AsyncIterator[Any]:
|
||||
agent_id = str(self._context.get("agent_id"))
|
||||
self._ledger.cost += COST_PER_CALL
|
||||
self._ledger.calls.append(str(self._context.get("agent_id")))
|
||||
self._ledger.calls.append(agent_id)
|
||||
ctx_wrapper = MagicMock()
|
||||
ctx_wrapper.context = self._context
|
||||
try:
|
||||
await self._hooks.on_llm_end(ctx_wrapper, self._agent, MagicMock())
|
||||
except Exception as exc: # noqa: BLE001
|
||||
self.run_loop_exception = exc
|
||||
# Stand in for the explicit yield tool a real turn ends with. Without it
|
||||
# every turn looks like a forgotten tool call and burns the recovery
|
||||
# budget, which is a different scenario from the one under test here.
|
||||
if self._coordinator.statuses.get(agent_id) == "running":
|
||||
wait_kind: WaitKind = "user" if self._context.get("parent_id") is None else "agents"
|
||||
await self._coordinator.park_waiting(agent_id, wait_kind=wait_kind)
|
||||
items: tuple[Any, ...] = ()
|
||||
for item in items:
|
||||
yield item
|
||||
@@ -67,7 +76,7 @@ class _FakeStream:
|
||||
return
|
||||
|
||||
|
||||
def _fake_runner(ledger: _FakeLedger) -> Any:
|
||||
def _fake_runner(ledger: _FakeLedger, coordinator: AgentCoordinator) -> Any:
|
||||
class _FakeRunner:
|
||||
@staticmethod
|
||||
def run_streamed(
|
||||
@@ -80,7 +89,13 @@ def _fake_runner(ledger: _FakeLedger) -> Any:
|
||||
session: Any, # noqa: ARG004
|
||||
hooks: ReportUsageHooks,
|
||||
) -> _FakeStream:
|
||||
return _FakeStream(ledger=ledger, hooks=hooks, context=context, agent=agent)
|
||||
return _FakeStream(
|
||||
ledger=ledger,
|
||||
hooks=hooks,
|
||||
context=context,
|
||||
agent=agent,
|
||||
coordinator=coordinator,
|
||||
)
|
||||
|
||||
return _FakeRunner
|
||||
|
||||
@@ -103,10 +118,10 @@ async def test_full_budget_lifecycle_reserve_then_cap( # noqa: PLR0915
|
||||
) -> None:
|
||||
ledger = _FakeLedger()
|
||||
hooks = ReportUsageHooks(model="test-model", max_budget_usd=MAX_BUDGET)
|
||||
monkeypatch.setattr(execution, "Runner", _fake_runner(ledger))
|
||||
coordinator = AgentCoordinator()
|
||||
monkeypatch.setattr(execution, "Runner", _fake_runner(ledger, coordinator))
|
||||
monkeypatch.setattr(execution, "_compact_session", _noop_compact)
|
||||
|
||||
coordinator = AgentCoordinator()
|
||||
db_path = tmp_path / "agents.sqlite"
|
||||
sessions: list[Any] = []
|
||||
run_config = MagicMock()
|
||||
@@ -218,7 +233,6 @@ async def test_respawned_children_after_reserve_never_spend(
|
||||
ledger = _FakeLedger()
|
||||
ledger.cost = 9.5
|
||||
hooks = ReportUsageHooks(model="test-model", max_budget_usd=MAX_BUDGET)
|
||||
monkeypatch.setattr(execution, "Runner", _fake_runner(ledger))
|
||||
monkeypatch.setattr(execution, "_compact_session", _noop_compact)
|
||||
|
||||
coordinator = AgentCoordinator()
|
||||
@@ -230,6 +244,7 @@ async def test_respawned_children_after_reserve_never_spend(
|
||||
restored = AgentCoordinator()
|
||||
await restored.restore(snap)
|
||||
assert restored.reserve_stopped is True
|
||||
monkeypatch.setattr(execution, "Runner", _fake_runner(ledger, restored))
|
||||
|
||||
sessions: list[Any] = []
|
||||
with patch("strix.core.hooks.get_global_report_state", return_value=ledger):
|
||||
@@ -264,7 +279,6 @@ async def test_resumed_parked_root_after_reserve_is_renotified_and_finalizes(
|
||||
ledger = _FakeLedger()
|
||||
ledger.cost = 9.0
|
||||
hooks = ReportUsageHooks(model="test-model", max_budget_usd=MAX_BUDGET)
|
||||
monkeypatch.setattr(execution, "Runner", _fake_runner(ledger))
|
||||
monkeypatch.setattr(execution, "_compact_session", _noop_compact)
|
||||
|
||||
coordinator = AgentCoordinator()
|
||||
@@ -276,6 +290,7 @@ async def test_resumed_parked_root_after_reserve_is_renotified_and_finalizes(
|
||||
restored = AgentCoordinator()
|
||||
await restored.restore(snap)
|
||||
assert restored.reserve_stopped is True
|
||||
monkeypatch.setattr(execution, "Runner", _fake_runner(ledger, restored))
|
||||
|
||||
root_session = open_agent_session("root", tmp_path / "agents.sqlite")
|
||||
with patch("strix.core.hooks.get_global_report_state", return_value=ledger):
|
||||
@@ -312,10 +327,10 @@ async def test_interactive_budget_pause_then_user_message_extends_and_resumes(
|
||||
ledger = _FakeLedger()
|
||||
ledger.cost = 9.0
|
||||
hooks = ReportUsageHooks(model="test-model", max_budget_usd=MAX_BUDGET, interactive=True)
|
||||
monkeypatch.setattr(execution, "Runner", _fake_runner(ledger))
|
||||
coordinator = AgentCoordinator()
|
||||
monkeypatch.setattr(execution, "Runner", _fake_runner(ledger, coordinator))
|
||||
monkeypatch.setattr(execution, "_compact_session", _noop_compact)
|
||||
|
||||
coordinator = AgentCoordinator()
|
||||
coordinator.set_budget_extender(hooks.extend_budget)
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
root_session = open_agent_session("root", tmp_path / "agents.sqlite")
|
||||
|
||||
+556
-44
@@ -5,25 +5,53 @@ from __future__ import annotations
|
||||
import asyncio
|
||||
import contextlib
|
||||
import json
|
||||
from typing import Any
|
||||
from typing import Any, cast
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
from agents.exceptions import MaxTurnsExceeded
|
||||
from agents.items import MessageOutputItem
|
||||
from agents.memory import SQLiteSession
|
||||
from agents.tool_context import ToolContext
|
||||
from openai.types.responses import ResponseOutputMessage, ResponseOutputRefusal
|
||||
|
||||
from strix.config import codex
|
||||
from strix.core import execution
|
||||
from strix.core.agents import AgentCoordinator
|
||||
from strix.core.execution import (
|
||||
_handle_content_guardrail,
|
||||
_notify_parent_on_terminal,
|
||||
_notify_root_on_budget_reserve,
|
||||
respawn_subagents,
|
||||
)
|
||||
from strix.core.sessions import seed_initial_input
|
||||
from strix.tools.finish.tool import finish_scan
|
||||
|
||||
|
||||
_NO_STREAM_EVENTS: list[Any] = []
|
||||
|
||||
|
||||
class _StructuredRefusalStream:
|
||||
def __init__(self, refusal: str) -> None:
|
||||
self.run_loop_exception: BaseException | None = None
|
||||
self.new_items = [
|
||||
MessageOutputItem(
|
||||
agent=MagicMock(),
|
||||
raw_item=ResponseOutputMessage(
|
||||
id="msg-refusal",
|
||||
content=[ResponseOutputRefusal(type="refusal", refusal=refusal)],
|
||||
role="assistant",
|
||||
status="completed",
|
||||
type="message",
|
||||
),
|
||||
)
|
||||
]
|
||||
|
||||
async def stream_events(self) -> Any:
|
||||
for event in _NO_STREAM_EVENTS:
|
||||
yield event
|
||||
|
||||
def cancel(self, mode: str = "immediate") -> None: # noqa: ARG002
|
||||
return
|
||||
|
||||
|
||||
async def _call_finish_scan(
|
||||
coordinator: AgentCoordinator, agent_id: str, parent_id: str | None
|
||||
) -> dict[str, Any]:
|
||||
@@ -503,76 +531,560 @@ async def test_terminal_notice_does_not_cancel_parent_stream(tmp_path: Any) -> N
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_guardrail_interactive_parks_agent_wakeable(tmp_path: Any) -> None:
|
||||
async def test_send_queues_without_session_and_drains_on_consume(tmp_path: Any) -> None:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
await coordinator.register("child", "recon", parent_id="root")
|
||||
exc = codex.CodexContentGuardrailError("chatgpt/gpt-5.6-sol")
|
||||
|
||||
result = await _handle_content_guardrail(coordinator, "child", exc, interactive=True)
|
||||
assert await coordinator.send("root", {"from": "user", "content": "hello"}) is True
|
||||
assert coordinator.pending_counts["root"] == 1
|
||||
|
||||
assert result is None
|
||||
assert coordinator.statuses["child"] == "waiting"
|
||||
assert "STRIX_LLM" in coordinator.errors["child"]
|
||||
session = SQLiteSession("root", tmp_path / "agents.db")
|
||||
await coordinator.attach_runtime("root", session=session)
|
||||
|
||||
waiter = asyncio.create_task(coordinator.wait_for_message("child"))
|
||||
await asyncio.sleep(0)
|
||||
assert not waiter.done()
|
||||
session = SQLiteSession("child", tmp_path / "agents.db")
|
||||
await coordinator.attach_runtime("child", session=session)
|
||||
await coordinator.send("child", {"from": "user", "content": "switched model, resume"})
|
||||
await asyncio.wait_for(waiter, timeout=1.0)
|
||||
count, items = await coordinator.consume_pending("root", include_items=True)
|
||||
assert count == 1
|
||||
assert items[0]["content"] == "hello"
|
||||
stored = await session.get_items()
|
||||
last = cast("dict[str, Any]", stored[-1])
|
||||
assert last["content"] == "hello"
|
||||
session.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_guardrail_noninteractive_fails_only_blocked_agent(tmp_path: Any) -> None:
|
||||
async def test_error_parked_agent_only_released_by_user_message(tmp_path: Any) -> None:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
await coordinator.register("child", "recon", parent_id="root")
|
||||
session = SQLiteSession("child", tmp_path / "agents.db")
|
||||
await coordinator.attach_runtime("child", session=session)
|
||||
await coordinator.set_status("child", "crashed", error="boom")
|
||||
|
||||
await coordinator.send("child", {"from": "root", "content": "peer nudge"})
|
||||
waiter = asyncio.create_task(coordinator.wait_for_message("child"))
|
||||
await asyncio.sleep(0.05)
|
||||
assert not waiter.done()
|
||||
|
||||
await coordinator.send("child", {"from": "user", "content": "wake up"})
|
||||
assert await asyncio.wait_for(waiter, timeout=1.0) is True
|
||||
|
||||
count, items = await coordinator.consume_pending("child", include_items=True)
|
||||
assert count == 2
|
||||
assert items[0]["content"].endswith("peer nudge")
|
||||
assert items[1]["content"] == "wake up"
|
||||
session.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_wait_for_message_timeout_returns_false() -> None:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("child", "recon", parent_id="root")
|
||||
|
||||
assert await coordinator.wait_for_message("child", timeout=0.05) is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_snapshot_round_trip_preserves_mailboxes() -> None:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
await coordinator.send("root", {"from": "user", "content": "queued"})
|
||||
|
||||
snap = await coordinator.snapshot()
|
||||
restored = AgentCoordinator()
|
||||
await restored.restore(snap)
|
||||
|
||||
assert restored.pending_counts["root"] == 1
|
||||
assert restored.runtimes["root"].mailbox == [{"from": "user", "content": "queued"}]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_run_cycle_parked_parks_instead_of_raising(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
async def _boom(*_args: Any, **_kwargs: Any) -> Any:
|
||||
raise RuntimeError("unexpected explosion")
|
||||
|
||||
monkeypatch.setattr(execution, "_run_cycle", _boom)
|
||||
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
|
||||
result = await execution._run_cycle_parked(
|
||||
object(),
|
||||
coordinator,
|
||||
"root",
|
||||
input_data=[],
|
||||
run_config=None, # type: ignore[arg-type]
|
||||
context={},
|
||||
max_turns=5,
|
||||
session=None,
|
||||
event_sink=None,
|
||||
hooks=None,
|
||||
)
|
||||
|
||||
assert result is None
|
||||
assert coordinator.statuses["root"] == "failed"
|
||||
assert coordinator.errors["root"] == "unexpected explosion"
|
||||
|
||||
|
||||
class _SalvageStream:
|
||||
def __init__(self, replay: list[dict[str, Any]]) -> None:
|
||||
self._replay = replay
|
||||
|
||||
def to_input_list(self) -> list[dict[str, Any]]:
|
||||
return self._replay
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_salvage_stream_to_session_preserves_full_history(tmp_path: Any) -> None:
|
||||
session = SQLiteSession("child", tmp_path / "agents.db")
|
||||
await session.add_items([{"role": "user", "content": "identity + task"}])
|
||||
pre_run = list(await session.get_items())
|
||||
|
||||
# A crash mid-run: the stream produced two turns the SDK never committed.
|
||||
stream = _SalvageStream(
|
||||
[
|
||||
{"role": "assistant", "content": "recon turn 1"},
|
||||
{"role": "assistant", "content": "recon turn 2"},
|
||||
]
|
||||
)
|
||||
await execution._salvage_stream_to_session(session, pre_run, stream, "child")
|
||||
|
||||
stored = [cast("dict[str, Any]", i) for i in await session.get_items()]
|
||||
assert [i["content"] for i in stored] == [
|
||||
"identity + task",
|
||||
"recon turn 1",
|
||||
"recon turn 2",
|
||||
]
|
||||
|
||||
# A crash with nothing new to salvage leaves the session untouched.
|
||||
await execution._salvage_stream_to_session(
|
||||
session, list(await session.get_items()), _SalvageStream([]), "child"
|
||||
)
|
||||
assert len(await session.get_items()) == 3
|
||||
session.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_seed_initial_input_persists_and_is_idempotent(tmp_path: Any) -> None:
|
||||
session = SQLiteSession("child", tmp_path / "agents.db")
|
||||
identity = [{"role": "user", "content": "You are agent recon (abc); do X."}]
|
||||
|
||||
assert await seed_initial_input(session, identity) is True
|
||||
assert len(await session.get_items()) == 1
|
||||
|
||||
# A populated session is left untouched (no duplicate identity message).
|
||||
assert await seed_initial_input(session, identity) is False
|
||||
assert len(await session.get_items()) == 1
|
||||
|
||||
assert await seed_initial_input(session, []) is False
|
||||
session.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_structured_provider_refusal_fails_interactive_agent(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
refusal = "This request was blocked under the provider's usage policy."
|
||||
stream = _StructuredRefusalStream(refusal)
|
||||
monkeypatch.setattr(
|
||||
"strix.core.execution.Runner.run_streamed", lambda *_args, **_kwargs: stream
|
||||
)
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
|
||||
result = await execution._run_cycle(
|
||||
MagicMock(),
|
||||
coordinator,
|
||||
"root",
|
||||
input_data="task",
|
||||
run_config=MagicMock(),
|
||||
context={},
|
||||
max_turns=5,
|
||||
session=None,
|
||||
interactive=True,
|
||||
event_sink=None,
|
||||
hooks=None,
|
||||
)
|
||||
|
||||
assert result is None
|
||||
assert coordinator.statuses["root"] == "failed"
|
||||
assert coordinator.errors["root"] == refusal
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_structured_provider_refusal_fails_noninteractive_child(
|
||||
tmp_path: Any,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
refusal = "This request was blocked under the provider's usage policy."
|
||||
stream = _StructuredRefusalStream(refusal)
|
||||
monkeypatch.setattr(
|
||||
"strix.core.execution.Runner.run_streamed", lambda *_args, **_kwargs: stream
|
||||
)
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
await coordinator.register("child", "recon", parent_id="root")
|
||||
session = SQLiteSession("root", tmp_path / "agents.db")
|
||||
await coordinator.attach_runtime("root", session=session)
|
||||
exc = codex.CodexContentGuardrailError("chatgpt/gpt-5.6-sol")
|
||||
|
||||
result = await _handle_content_guardrail(coordinator, "child", exc, interactive=False)
|
||||
result = await execution._run_cycle(
|
||||
MagicMock(),
|
||||
coordinator,
|
||||
"child",
|
||||
input_data="task",
|
||||
run_config=MagicMock(),
|
||||
context={"parent_id": "root"},
|
||||
max_turns=5,
|
||||
session=None,
|
||||
interactive=False,
|
||||
event_sink=None,
|
||||
hooks=None,
|
||||
)
|
||||
|
||||
assert result is None
|
||||
assert coordinator.statuses["child"] == "failed"
|
||||
assert "STRIX_LLM" in coordinator.errors["child"]
|
||||
assert coordinator.statuses["root"] == "running"
|
||||
assert coordinator.errors["child"] == refusal
|
||||
assert coordinator.pending_counts.get("root", 0) > 0
|
||||
session.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resume_revives_guardrail_parked_child_but_not_plain_waiting(
|
||||
async def test_run_agent_loop_seeds_identity_before_first_cycle(
|
||||
tmp_path: Any, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
await coordinator.register("blocked", "recon", parent_id="root")
|
||||
await coordinator.register("peer_waiter", "recon", parent_id="root")
|
||||
await coordinator.set_status("blocked", "waiting", error="STRIX_LLM guardrail")
|
||||
await coordinator.set_status("peer_waiter", "waiting")
|
||||
await coordinator.register("child", "recon", parent_id="root")
|
||||
session = SQLiteSession("child", tmp_path / "agents.db")
|
||||
|
||||
parked: dict[str, bool] = {}
|
||||
captured: dict[str, Any] = {}
|
||||
|
||||
async def _fake_start_child_runner(**kwargs: Any) -> None:
|
||||
parked[kwargs["child_id"]] = bool(kwargs["start_parked"])
|
||||
async def _crash_first_turn(*_args: Any, **kwargs: Any) -> Any:
|
||||
captured["input_data"] = kwargs.get("input_data")
|
||||
captured["items_at_start"] = await session.get_items()
|
||||
raise RuntimeError("first-turn crash")
|
||||
|
||||
monkeypatch.setattr(execution, "_start_child_runner", _fake_start_child_runner)
|
||||
monkeypatch.setattr(execution, "_run_cycle", _crash_first_turn)
|
||||
|
||||
await respawn_subagents(
|
||||
coordinator=coordinator,
|
||||
factory=lambda **_kwargs: object(),
|
||||
agents_db_path=tmp_path / "agents.db",
|
||||
sessions_to_close=[],
|
||||
identity = [{"role": "user", "content": "You are agent recon (abc); maintain your identity."}]
|
||||
with pytest.raises(RuntimeError, match="first-turn crash"):
|
||||
await execution.run_agent_loop(
|
||||
agent=object(),
|
||||
initial_input=identity,
|
||||
run_config=None, # type: ignore[arg-type]
|
||||
context={"agent_id": "child", "parent_id": "root"},
|
||||
max_turns=5,
|
||||
coordinator=coordinator,
|
||||
agent_id="child",
|
||||
interactive=False,
|
||||
session=session,
|
||||
)
|
||||
|
||||
# The first cycle ran with an empty input against the pre-seeded session.
|
||||
assert captured["input_data"] == []
|
||||
assert captured["items_at_start"]
|
||||
# The identity/task survives the first-turn crash, so a revival can resume it.
|
||||
stored = await session.get_items()
|
||||
assert any("recon" in str(cast("dict[str, Any]", i).get("content", "")) for i in stored)
|
||||
session.close()
|
||||
|
||||
|
||||
def _scripted_cycle(
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
statuses: list[str],
|
||||
calls: list[Any],
|
||||
) -> Any:
|
||||
"""Fake run cycle that leaves ``agent_id`` in a scripted status per call."""
|
||||
|
||||
async def _cycle(*_args: Any, **kwargs: Any) -> Any:
|
||||
calls.append(kwargs.get("input_data"))
|
||||
status = statuses[min(len(calls) - 1, len(statuses) - 1)]
|
||||
await coordinator.set_status(agent_id, status)
|
||||
return MagicMock(final_output="plain text, no tool call")
|
||||
|
||||
return _cycle
|
||||
|
||||
|
||||
async def _drive(
|
||||
coordinator: AgentCoordinator,
|
||||
agent_id: str,
|
||||
*,
|
||||
interactive: bool,
|
||||
max_turns: int = 5,
|
||||
) -> Any:
|
||||
return await execution._run_until_lifecycle(
|
||||
MagicMock(),
|
||||
coordinator,
|
||||
agent_id,
|
||||
initial_input=[],
|
||||
run_config=MagicMock(),
|
||||
max_turns=10,
|
||||
interactive=True,
|
||||
parent_ctx={"agent_id": "root", "parent_id": None},
|
||||
root_id="root",
|
||||
context={"agent_id": agent_id, "parent_id": None},
|
||||
max_turns=max_turns,
|
||||
session=None,
|
||||
interactive=interactive,
|
||||
event_sink=None,
|
||||
hooks=None,
|
||||
)
|
||||
|
||||
assert parked["blocked"] is False
|
||||
assert parked["peer_waiter"] is True
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_interactive_text_only_turn_is_nudged_instead_of_parking(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""A no-tool-call turn must not silently hand control back to the user."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_scripted_cycle(coordinator, "root", ["running", "completed"], calls),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert len(calls) == 2
|
||||
# The retry carries an explicit "call a tool" nudge rather than empty input.
|
||||
nudge = calls[1][0]["content"]
|
||||
assert "without a tool call" in nudge
|
||||
assert "respond_to_user" in nudge
|
||||
assert coordinator.statuses["root"] == "completed"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_interactive_explicit_park_gets_no_nudge(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""``waiting`` is only reachable via respond_to_user / wait_for_agents."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_scripted_cycle(coordinator, "root", ["waiting"], calls),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert len(calls) == 1
|
||||
assert coordinator.statuses["root"] == "waiting"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_interactive_recovery_exhaustion_parks_instead_of_crashing(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""A human can resume an interactive scan, so exhaustion parks rather than dies."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_scripted_cycle(coordinator, "root", ["running"], calls),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert len(calls) == execution._INTERACTIVE_TOOL_RECOVERY_LIMIT
|
||||
assert coordinator.statuses["root"] == "waiting"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_interactive_subagent_exhaustion_tells_its_parent(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""A parked child must report up so its parent stops waiting on it.
|
||||
|
||||
The parent is an agent, not a watching human, so a parent blocked in
|
||||
wait_for_agents otherwise burns its whole timeout on a completion
|
||||
report the child can no longer send.
|
||||
"""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
await coordinator.register("child", "recon", parent_id="root")
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_scripted_cycle(coordinator, "child", ["running"], calls),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "child", interactive=True)
|
||||
|
||||
assert coordinator.statuses["child"] == "waiting"
|
||||
pending, items = await coordinator.consume_pending("root", include_items=True)
|
||||
assert pending == 1
|
||||
notice = str(items[0])
|
||||
assert "child" in notice
|
||||
assert "parked" in notice
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_interactive_root_exhaustion_notifies_nobody(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""The root has no parent to report to, so parking stays silent."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_scripted_cycle(coordinator, "root", ["running"], []),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
pending, _ = await coordinator.consume_pending("root")
|
||||
assert pending == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_noninteractive_recovery_exhaustion_crashes(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""No user is present to resume an autonomous run, so it still fails loudly."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle",
|
||||
_scripted_cycle(coordinator, "root", ["running"], calls),
|
||||
)
|
||||
|
||||
with pytest.raises(MaxTurnsExceeded):
|
||||
await _drive(coordinator, "root", interactive=False, max_turns=2)
|
||||
|
||||
assert len(calls) == 2
|
||||
assert coordinator.statuses["root"] == "crashed"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_tool_required_message_is_persisted_to_the_session(tmp_path: Any) -> None:
|
||||
session = SQLiteSession("root", tmp_path / "agents.db")
|
||||
|
||||
assert (
|
||||
await execution._append_tool_required_message(
|
||||
session=session,
|
||||
context={"parent_id": None},
|
||||
attempt=1,
|
||||
limit=3,
|
||||
interactive=True,
|
||||
)
|
||||
== []
|
||||
)
|
||||
|
||||
stored = [cast("dict[str, Any]", i) for i in await session.get_items()]
|
||||
assert "finish_scan" in stored[0]["content"]
|
||||
assert "respond_to_user" in stored[0]["content"]
|
||||
session.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_recovery_count_survives_a_snapshot_round_trip(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""A resumed agent must not earn a fresh nudge budget and loop forever."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_scripted_cycle(coordinator, "root", ["running"], calls),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
assert coordinator.recovery_counts["root"] == execution._INTERACTIVE_TOOL_RECOVERY_LIMIT
|
||||
|
||||
restored = AgentCoordinator()
|
||||
await restored.restore(await coordinator.snapshot())
|
||||
assert restored.recovery_counts["root"] == execution._INTERACTIVE_TOOL_RECOVERY_LIMIT
|
||||
|
||||
# The restored agent is already at its cap, so it parks after a single
|
||||
# further text-only cycle instead of starting the whole budget over.
|
||||
resumed_calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_scripted_cycle(restored, "root", ["running"], resumed_calls),
|
||||
)
|
||||
await _drive(restored, "root", interactive=True)
|
||||
|
||||
assert len(resumed_calls) == 1
|
||||
assert restored.statuses["root"] == "waiting"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_recovery_count_is_cleared_by_a_lifecycle_tool(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
calls: list[Any] = []
|
||||
monkeypatch.setattr(
|
||||
execution,
|
||||
"_run_cycle_parked",
|
||||
_scripted_cycle(coordinator, "root", ["running", "completed"], calls),
|
||||
)
|
||||
|
||||
await _drive(coordinator, "root", interactive=True)
|
||||
|
||||
assert "root" not in coordinator.recovery_counts
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_agent_awaiting_a_human_is_never_auto_resumed() -> None:
|
||||
"""The user can message any agent, so parking for one is not root-only."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
await coordinator.register("child", "recon", parent_id="root")
|
||||
|
||||
for agent_id in ("root", "child"):
|
||||
await coordinator.park_waiting(agent_id, wait_kind="user")
|
||||
assert await execution._plain_waiting_timeout(coordinator, agent_id) is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_agent_awaiting_other_agents_is_re_checked_on_a_timer() -> None:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
await coordinator.park_waiting("root", wait_kind="agents")
|
||||
|
||||
timeout = await execution._plain_waiting_timeout(coordinator, "root")
|
||||
assert timeout == execution._WAITING_AUTO_RESUME_TIMEOUT_S
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_idle_auto_resumes_stop_after_their_budget() -> None:
|
||||
"""A wedged agent must not burn a model turn per timeout for the whole scan."""
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("child", "recon", parent_id="root")
|
||||
await coordinator.park_waiting("child", wait_kind="agents")
|
||||
|
||||
for _ in range(execution._MAX_IDLE_AUTO_RESUMES):
|
||||
assert await execution._plain_waiting_timeout(coordinator, "child") is not None
|
||||
await coordinator.record_idle_resume("child")
|
||||
|
||||
assert await execution._plain_waiting_timeout(coordinator, "child") is None
|
||||
|
||||
# A real message is real progress, so the budget starts over.
|
||||
await coordinator.reset_idle_resumes("child")
|
||||
assert await execution._plain_waiting_timeout(coordinator, "child") is not None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_wait_kind_survives_a_snapshot_round_trip() -> None:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
await coordinator.park_waiting("root", wait_kind="user")
|
||||
await coordinator.record_idle_resume("root")
|
||||
|
||||
restored = AgentCoordinator()
|
||||
await restored.restore(await coordinator.snapshot())
|
||||
|
||||
assert restored.wait_kinds["root"] == "user"
|
||||
assert restored.idle_resume_counts["root"] == 1
|
||||
assert await execution._plain_waiting_timeout(restored, "root") is None
|
||||
|
||||
@@ -15,6 +15,7 @@ from openai import (
|
||||
RateLimitError,
|
||||
)
|
||||
|
||||
from strix.config import codex
|
||||
from strix.core import execution
|
||||
from strix.core.agents import AgentCoordinator
|
||||
|
||||
@@ -55,11 +56,27 @@ def test_server_errors_are_transient() -> None:
|
||||
assert execution._is_transient_model_error(_status_error(status)) is True
|
||||
|
||||
|
||||
def test_rate_limit_is_not_retried_here() -> None:
|
||||
def test_rate_limit_is_retried() -> None:
|
||||
rate_limited = RateLimitError(
|
||||
"slow down", response=httpx.Response(429, request=_request()), body=None
|
||||
)
|
||||
assert execution._is_transient_model_error(rate_limited) is False
|
||||
assert execution._is_transient_model_error(rate_limited) is True
|
||||
|
||||
|
||||
def test_dns_and_connection_errors_are_transient() -> None:
|
||||
assert execution._is_transient_model_error(OSError("nodename nor servname provided")) is True
|
||||
assert execution._is_transient_model_error(ConnectionError("reset")) is True
|
||||
assert execution._is_transient_model_error(TimeoutError("timed out")) is True
|
||||
|
||||
|
||||
def test_content_guardrail_is_not_retried() -> None:
|
||||
guardrail = APIError(
|
||||
"This content was flagged for possible cybersecurity risk",
|
||||
_request(),
|
||||
body=None,
|
||||
)
|
||||
assert codex.is_content_guardrail_error(guardrail) is True
|
||||
assert execution._is_transient_model_error(guardrail) is False
|
||||
|
||||
|
||||
def test_client_errors_are_not_transient() -> None:
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
"""Tests for the ``respond_to_user`` yield tool."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from agents.tool_context import ToolContext
|
||||
|
||||
from strix.core.agents import AgentCoordinator
|
||||
from strix.tools.respond.tool import respond_to_user
|
||||
|
||||
|
||||
async def _call(context: dict[str, Any], message: str = "here is what I found") -> dict[str, Any]:
|
||||
ctx = ToolContext(
|
||||
context=context,
|
||||
tool_name="respond_to_user",
|
||||
tool_call_id="call-1",
|
||||
tool_arguments="{}",
|
||||
)
|
||||
raw = await respond_to_user.on_invoke_tool(ctx, json.dumps({"message": message}))
|
||||
return json.loads(raw) # type: ignore[no-any-return]
|
||||
|
||||
|
||||
async def _context(*, interactive: bool, agent_id: str = "root") -> dict[str, Any]:
|
||||
coordinator = AgentCoordinator()
|
||||
await coordinator.register("root", "strix", parent_id=None)
|
||||
return {"coordinator": coordinator, "agent_id": agent_id, "interactive": interactive}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_parks_the_agent_and_carries_the_message() -> None:
|
||||
context = await _context(interactive=True)
|
||||
result = await _call(context)
|
||||
|
||||
coordinator = context["coordinator"]
|
||||
assert result["success"] is True
|
||||
assert result["wait_outcome"] == "waiting"
|
||||
assert result["message"] == "here is what I found"
|
||||
assert coordinator.statuses["root"] == "waiting"
|
||||
# Recorded as a human wait, so the driver never auto-resumes it.
|
||||
assert coordinator.wait_kinds["root"] == "user"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_rejected_in_an_autonomous_run() -> None:
|
||||
context = await _context(interactive=False)
|
||||
result = await _call(context)
|
||||
|
||||
assert result["success"] is False
|
||||
assert "finish_scan" in result["error"]
|
||||
assert context["coordinator"].statuses["root"] == "running"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_message_that_already_arrived_is_taken_instead_of_parking() -> None:
|
||||
context = await _context(interactive=True)
|
||||
coordinator = context["coordinator"]
|
||||
await coordinator.send("root", {"from": "user", "content": "wait, one more thing"})
|
||||
|
||||
result = await _call(context)
|
||||
|
||||
assert result["wait_outcome"] == "message_arrived"
|
||||
assert result["pending_messages"] == 1
|
||||
assert coordinator.statuses["root"] == "running"
|
||||
Reference in New Issue
Block a user