mirror of
https://github.com/usestrix/strix.git
synced 2026-08-16 09:26:39 +02:00
* let an agent wait on what it already said An agent that answers in plain text is nudged to call a tool, and the only tool that hands control back takes a required message. So it says the same thing twice: once as text the user has already read, once as the argument it had to supply to stop. Seen on a run whose whole instruction was "hi" - a greeting, then the same greeting again through respond_to_user. message is optional now. The nudge arms the tool with the text that was delivered and says not to repeat it, so an agent that has said its piece can park on it with an empty call. Anything it does want to add it passes normally. Parking still cannot leave the user on silence: an empty call is refused unless something was actually said, and the arming is single use - execution clears it as soon as a turn ends any other way. The interactive prompt now also says to answer and stop in one respond_to_user call, which is what avoids the nudge in the first place. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> * drop the worked example from the interactive prompt "the user greeted you, asked something you can answer outright, or you need a decision" was the run I had been reading, written into a rule that holds whatever the reason. The rule is that replying and stopping is one call; listing occasions only invites the model to check whether this is one of them. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> * drop the arming flag; an empty message just waits Passing the delivered text from execution into the tool, and refusing an empty call without it, was machinery guarding against an agent parking having said nothing. That leaves the user looking at "waiting for your reply" with a cursor in front of them - they type. It does not need a mechanism. What is left is the default on message, and the nudge saying the text already landed. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> * only offer waiting on words that were written The nudge told every agent its text had already been delivered, but it fires whenever a turn leaves the agent running, and a turn can end with no tool call and no text at all - _final_output_preview has carried <none> and <empty> branches all along. An agent that said nothing was being invited to wait on an answer the user never received, leaving them at a bare prompt. It now reads the turn: waiting on what was said is offered only when something was, and otherwise the agent is told plainly that the user has read nothing and to send its message. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> * leave the continuation nudge alone Rewording it meant asserting from the outside whether the agent had spoken, and the nudge fires whenever a turn leaves the agent running - text or no text. The agent knows which it did without being told, so the guidance belongs in its prompt, where the condition is its own to read. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> * say it in the nudge, where the agent is reading An agent stranded by the nudge reasons off the nudge. Told only to call respond_to_user, it supplies a message, and since it has just answered in plain text that message is the same answer again. The system prompt saying otherwise sits thousands of tokens earlier and loses. The clause goes on the line the agent acts on: call respond_to_user, with no message if it has already said it. That reads true whatever the turn did, including one that produced no text, because the agent is the one who knows which — nothing here has to work it out from the outside. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
95 lines
3.1 KiB
Python
95 lines
3.1 KiB
Python
"""Tests for the ``respond_to_user`` yield tool."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from typing import Any
|
|
|
|
import pytest
|
|
from agents.tool_context import ToolContext
|
|
|
|
from strix.core.agents import AgentCoordinator
|
|
from strix.tools.respond.tool import respond_to_user
|
|
|
|
|
|
async def _call(context: dict[str, Any], message: str = "here is what I found") -> dict[str, Any]:
|
|
ctx = ToolContext(
|
|
context=context,
|
|
tool_name="respond_to_user",
|
|
tool_call_id="call-1",
|
|
tool_arguments="{}",
|
|
)
|
|
raw = await respond_to_user.on_invoke_tool(ctx, json.dumps({"message": message}))
|
|
return json.loads(raw) # type: ignore[no-any-return]
|
|
|
|
|
|
async def _context(*, interactive: bool, agent_id: str = "root") -> dict[str, Any]:
|
|
coordinator = AgentCoordinator()
|
|
await coordinator.register("root", "strix", parent_id=None)
|
|
return {"coordinator": coordinator, "agent_id": agent_id, "interactive": interactive}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_parks_the_agent_and_carries_the_message() -> None:
|
|
context = await _context(interactive=True)
|
|
result = await _call(context)
|
|
|
|
coordinator = context["coordinator"]
|
|
assert result["success"] is True
|
|
assert result["wait_outcome"] == "waiting"
|
|
assert result["message"] == "here is what I found"
|
|
assert coordinator.statuses["root"] == "waiting"
|
|
# Recorded as a human wait, so the driver never auto-resumes it.
|
|
assert coordinator.wait_kinds["root"] == "user"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_rejected_in_an_autonomous_run() -> None:
|
|
context = await _context(interactive=False)
|
|
result = await _call(context)
|
|
|
|
assert result["success"] is False
|
|
assert "finish_scan" in result["error"]
|
|
assert context["coordinator"].statuses["root"] == "running"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_a_message_that_already_arrived_is_taken_instead_of_parking() -> None:
|
|
context = await _context(interactive=True)
|
|
coordinator = context["coordinator"]
|
|
await coordinator.send("root", {"from": "user", "content": "wait, one more thing"})
|
|
|
|
result = await _call(context)
|
|
|
|
assert result["wait_outcome"] == "message_arrived"
|
|
assert result["pending_messages"] == 1
|
|
assert coordinator.statuses["root"] == "running"
|
|
|
|
|
|
async def _call_without_message(context: dict[str, Any]) -> dict[str, Any]:
|
|
ctx = ToolContext(
|
|
context=context,
|
|
tool_name="respond_to_user",
|
|
tool_call_id="call-1",
|
|
tool_arguments="{}",
|
|
)
|
|
raw = await respond_to_user.on_invoke_tool(ctx, "{}")
|
|
return json.loads(raw) # type: ignore[no-any-return]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_parks_without_a_message() -> None:
|
|
"""An agent that has already said its piece as plain text can just wait.
|
|
|
|
The nudge is what leaves it here, and while a message was required the only
|
|
way to stop was to send the same answer a second time.
|
|
"""
|
|
context = await _context(interactive=True)
|
|
|
|
result = await _call_without_message(context)
|
|
|
|
assert result["success"] is True
|
|
assert result["wait_outcome"] == "waiting"
|
|
assert result["message"] == ""
|
|
assert context["coordinator"].statuses["root"] == "waiting"
|