mirror of
https://github.com/usestrix/strix.git
synced 2026-08-23 03:12:37 +02:00
Merge remote-tracking branch 'origin/main' into grok-subscription-oauth
This commit is contained in:
@@ -70,7 +70,6 @@ async def test_encoded_list_is_decoded_for_an_array_parameter(schema: dict[str,
|
||||
"auth",
|
||||
"Endpoint /admin leaks user data, and session tokens never expire",
|
||||
'"auth"',
|
||||
"",
|
||||
],
|
||||
)
|
||||
async def test_free_form_strings_are_never_split_into_an_array(value: str) -> None:
|
||||
@@ -79,6 +78,29 @@ async def test_free_form_strings_are_never_split_into_an_array(value: str) -> No
|
||||
assert parsed["tags"] == value
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("schema", [_ARRAY, _NULLABLE_ARRAY])
|
||||
@pytest.mark.parametrize("value", ["", " "])
|
||||
async def test_empty_string_becomes_an_empty_array(schema: dict[str, Any], value: str) -> None:
|
||||
parsed = await _roundtrip(schema, {"tags": value})
|
||||
|
||||
assert parsed["tags"] == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_empty_string_becomes_an_empty_object() -> None:
|
||||
parsed = await _roundtrip(_OBJECT, {"modifications": ""})
|
||||
|
||||
assert parsed["modifications"] == {}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_empty_string_for_a_string_parameter_is_untouched() -> None:
|
||||
parsed = await _roundtrip(_STRING, {"todos": ""})
|
||||
|
||||
assert parsed["todos"] == ""
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_encoded_mapping_is_decoded_for_an_object_parameter() -> None:
|
||||
parsed = await _roundtrip(_OBJECT, {"modifications": '{"method": "POST"}'})
|
||||
|
||||
@@ -143,7 +143,7 @@ def test_cost_callback_estimates_cost_with_bare_model_fallback() -> None:
|
||||
}
|
||||
|
||||
def fake_completion_cost(**kwargs: object) -> float:
|
||||
if kwargs["model"] == "gpt-4o-mini":
|
||||
if kwargs["model"] == "openai/gpt-4o-mini":
|
||||
return 0.025
|
||||
raise ValueError(kwargs["model"])
|
||||
|
||||
|
||||
@@ -1228,3 +1228,40 @@ async def test_wait_kind_survives_a_snapshot_round_trip() -> None:
|
||||
assert restored.wait_kinds["root"] == "user"
|
||||
assert restored.idle_resume_counts["root"] == 1
|
||||
assert await execution._plain_waiting_timeout(restored, "root") is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_interactive_nudge_offers_waiting_without_repeating() -> None:
|
||||
"""The nudge is the instruction an agent reads when it is stranded here.
|
||||
|
||||
It is where the option to wait on what was already said has to be, not only
|
||||
in the system prompt: an agent that ended a turn on plain text reasons off
|
||||
this text, and without the clause it restates its answer to reach a tool
|
||||
call, so the user reads it twice.
|
||||
|
||||
The clause holds whatever the turn did, because the agent is the one who
|
||||
knows whether it spoke — this fires for a turn that produced no text at all.
|
||||
"""
|
||||
items = await execution._append_tool_required_message(
|
||||
session=None,
|
||||
context={"parent_id": None},
|
||||
attempt=1,
|
||||
limit=3,
|
||||
interactive=True,
|
||||
)
|
||||
|
||||
assert "with no message if you have already said it" in items[0]["content"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_autonomous_nudge_does_not_offer_the_user() -> None:
|
||||
"""There is nobody attached to an autonomous run to wait for."""
|
||||
items = await execution._append_tool_required_message(
|
||||
session=None,
|
||||
context={"parent_id": None},
|
||||
attempt=1,
|
||||
limit=3,
|
||||
interactive=False,
|
||||
)
|
||||
|
||||
assert "respond_to_user" not in items[0]["content"]
|
||||
|
||||
@@ -299,6 +299,16 @@ def test_make_model_settings_forces_required_for_anyllm_routed_openai_model() ->
|
||||
assert settings.tool_choice == "required"
|
||||
|
||||
|
||||
def test_make_model_settings_disables_parallel_tool_calls_by_default() -> None:
|
||||
assert make_model_settings("none", model_name="gpt-4o").parallel_tool_calls is False
|
||||
|
||||
|
||||
def test_make_model_settings_omits_parallel_tool_calls_without_tools() -> None:
|
||||
settings = make_model_settings("none", model_name="gpt-4o", has_tools=False)
|
||||
|
||||
assert settings.parallel_tool_calls is None
|
||||
|
||||
|
||||
def test_make_model_settings_sets_request_timeout() -> None:
|
||||
settings = make_model_settings(
|
||||
"none",
|
||||
@@ -351,3 +361,32 @@ def test_make_model_settings_timeout_survives_reasoning_resolve() -> None:
|
||||
|
||||
assert settings.extra_args is not None
|
||||
assert settings.extra_args["timeout"] == 120.0
|
||||
|
||||
|
||||
def test_openrouter_attribution_rides_on_the_request_headers() -> None:
|
||||
# litellm.headers is ignored once a request carries any header of its own,
|
||||
# so the attribution must be part of the per-request headers.
|
||||
headers = make_model_settings(
|
||||
None, model_name="openrouter/anthropic/claude-sonnet-4-5"
|
||||
).extra_headers
|
||||
assert headers == {
|
||||
"HTTP-Referer": "https://strix.ai",
|
||||
"X-Title": "Strix",
|
||||
"X-OpenRouter-Categories": "cli-agent",
|
||||
}
|
||||
|
||||
|
||||
def test_openrouter_attribution_absent_for_other_providers() -> None:
|
||||
assert make_model_settings(None, model_name="anthropic/claude-sonnet-4-5").extra_headers is None
|
||||
|
||||
|
||||
def test_user_headers_override_openrouter_attribution() -> None:
|
||||
headers = make_model_settings(
|
||||
None,
|
||||
model_name="openrouter/anthropic/claude-sonnet-4-5",
|
||||
extra_headers={"X-Title": "Custom", "X-Tenant": "acme"},
|
||||
).extra_headers
|
||||
assert headers is not None
|
||||
assert headers["X-Title"] == "Custom"
|
||||
assert headers["X-Tenant"] == "acme"
|
||||
assert headers["HTTP-Referer"] == "https://strix.ai"
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
import litellm
|
||||
from agents.usage import Usage
|
||||
|
||||
from strix.report.pricing import resolve_litellm_model
|
||||
from strix.report.usage import LLMUsageLedger
|
||||
|
||||
|
||||
def test_resolves_common_bare_model_names() -> None:
|
||||
resolve_litellm_model.cache_clear()
|
||||
assert resolve_litellm_model("deepseek-v4-flash") == "deepseek/deepseek-v4-flash"
|
||||
assert resolve_litellm_model("openai/deepseek-v4-flash") == "deepseek/deepseek-v4-flash"
|
||||
assert resolve_litellm_model("grok-4.5") == "xai/grok-4.5"
|
||||
assert resolve_litellm_model("MiniMax-M3") == "minimax/MiniMax-M3"
|
||||
|
||||
|
||||
def test_resolver_returns_none_for_unresolvable_model() -> None:
|
||||
resolve_litellm_model.cache_clear()
|
||||
assert resolve_litellm_model("provider/not-a-real-model") is None
|
||||
|
||||
|
||||
def test_ledger_uses_estimate_when_routed_provider_reports_no_cost() -> None:
|
||||
usage = Usage()
|
||||
usage.requests = 1
|
||||
usage.input_tokens = 1000
|
||||
usage.output_tokens = 200
|
||||
usage.total_tokens = 1200
|
||||
ledger = LLMUsageLedger()
|
||||
|
||||
with patch("litellm.completion_cost", return_value=0.42):
|
||||
ledger.record(agent_id="a", usage=usage, model="openai/deepseek-v4-flash")
|
||||
|
||||
assert ledger.total_cost == 0.42
|
||||
|
||||
|
||||
def test_ledger_prefers_observed_cost_over_estimate() -> None:
|
||||
usage = Usage()
|
||||
usage.requests = 1
|
||||
usage.input_tokens = 1000
|
||||
usage.output_tokens = 200
|
||||
usage.total_tokens = 1200
|
||||
ledger = LLMUsageLedger()
|
||||
|
||||
with patch("litellm.completion_cost", return_value=0.42):
|
||||
ledger.record(agent_id="a", usage=usage, model="openai/deepseek-v4-flash")
|
||||
ledger.record_observed_cost(0.17)
|
||||
|
||||
assert ledger.total_cost == 0.17
|
||||
|
||||
|
||||
def test_hydrated_estimate_continues_accumulating_new_estimates() -> None:
|
||||
usage = Usage()
|
||||
usage.requests = 1
|
||||
usage.input_tokens = 1000
|
||||
usage.output_tokens = 200
|
||||
usage.total_tokens = 1200
|
||||
ledger = LLMUsageLedger()
|
||||
ledger.hydrate({"cost": 0.42})
|
||||
|
||||
with patch("litellm.completion_cost", return_value=0.17):
|
||||
ledger.record(agent_id="a", usage=usage, model="openai/deepseek-v4-flash")
|
||||
|
||||
assert ledger.total_cost == 0.59
|
||||
|
||||
|
||||
def test_zero_cost_disables_both_observed_and_estimated_costs() -> None:
|
||||
usage = Usage()
|
||||
usage.requests = 1
|
||||
usage.input_tokens = 1000
|
||||
usage.output_tokens = 200
|
||||
usage.total_tokens = 1200
|
||||
ledger = LLMUsageLedger()
|
||||
ledger.zero_cost = True
|
||||
|
||||
with patch("litellm.completion_cost", return_value=0.42) as estimate:
|
||||
ledger.record(agent_id="a", usage=usage, model="deepseek-v4-flash")
|
||||
ledger.record_observed_cost(1.0)
|
||||
|
||||
estimate.assert_not_called()
|
||||
assert ledger.total_cost == 0.0
|
||||
|
||||
|
||||
def test_resolver_uses_provider_when_bare_entry_has_one() -> None:
|
||||
original = litellm.model_cost
|
||||
litellm.model_cost = {
|
||||
"example": {
|
||||
"litellm_provider": "example-provider",
|
||||
"input_cost_per_token": 1.0,
|
||||
"output_cost_per_token": 2.0,
|
||||
}
|
||||
}
|
||||
try:
|
||||
resolve_litellm_model.cache_clear()
|
||||
assert resolve_litellm_model("example") == "example-provider/example"
|
||||
finally:
|
||||
litellm.model_cost = original
|
||||
resolve_litellm_model.cache_clear()
|
||||
|
||||
|
||||
def test_resolver_does_not_guess_between_differently_priced_providers() -> None:
|
||||
original = litellm.model_cost
|
||||
litellm.model_cost = {
|
||||
"provider-a/example": {
|
||||
"input_cost_per_token": 1.0,
|
||||
"output_cost_per_token": 2.0,
|
||||
},
|
||||
"provider-b/example": {
|
||||
"input_cost_per_token": 3.0,
|
||||
"output_cost_per_token": 4.0,
|
||||
},
|
||||
}
|
||||
try:
|
||||
resolve_litellm_model.cache_clear()
|
||||
assert resolve_litellm_model("example") is None
|
||||
finally:
|
||||
litellm.model_cost = original
|
||||
resolve_litellm_model.cache_clear()
|
||||
@@ -64,3 +64,31 @@ async def test_a_message_that_already_arrived_is_taken_instead_of_parking() -> N
|
||||
assert result["wait_outcome"] == "message_arrived"
|
||||
assert result["pending_messages"] == 1
|
||||
assert coordinator.statuses["root"] == "running"
|
||||
|
||||
|
||||
async def _call_without_message(context: dict[str, Any]) -> dict[str, Any]:
|
||||
ctx = ToolContext(
|
||||
context=context,
|
||||
tool_name="respond_to_user",
|
||||
tool_call_id="call-1",
|
||||
tool_arguments="{}",
|
||||
)
|
||||
raw = await respond_to_user.on_invoke_tool(ctx, "{}")
|
||||
return json.loads(raw) # type: ignore[no-any-return]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_parks_without_a_message() -> None:
|
||||
"""An agent that has already said its piece as plain text can just wait.
|
||||
|
||||
The nudge is what leaves it here, and while a message was required the only
|
||||
way to stop was to send the same answer a second time.
|
||||
"""
|
||||
context = await _context(interactive=True)
|
||||
|
||||
result = await _call_without_message(context)
|
||||
|
||||
assert result["success"] is True
|
||||
assert result["wait_outcome"] == "waiting"
|
||||
assert result["message"] == ""
|
||||
assert context["coordinator"].statuses["root"] == "waiting"
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import types
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from agents import ModelSettings
|
||||
|
||||
import strix.tools.notes.tools as notes_tools
|
||||
import strix.tools.todo.tools as todo_tools
|
||||
from strix.core import runner
|
||||
from strix.core.agents import AgentCoordinator
|
||||
from strix.runtime import session_manager
|
||||
|
||||
|
||||
def _wire_runner(monkeypatch: pytest.MonkeyPatch, tmp_path: Any) -> None:
|
||||
monkeypatch.setattr(runner, "run_dir_for", lambda _scan_id: tmp_path)
|
||||
monkeypatch.setattr(runner, "runtime_state_dir", lambda _run_dir: tmp_path)
|
||||
monkeypatch.setattr(runner, "setup_scan_logging", lambda _run_dir: lambda: None)
|
||||
monkeypatch.setattr(runner, "set_scan_id", lambda _scan_id: None)
|
||||
|
||||
settings = types.SimpleNamespace(
|
||||
llm=types.SimpleNamespace(
|
||||
model="openai/gpt-4o",
|
||||
reasoning_effort="high",
|
||||
force_required_tool_choice=False,
|
||||
timeout=300,
|
||||
prompt_cache=True,
|
||||
extra_headers=None,
|
||||
),
|
||||
runtime=types.SimpleNamespace(max_context_images=3),
|
||||
)
|
||||
monkeypatch.setattr(runner, "load_settings", lambda: settings)
|
||||
monkeypatch.setattr(runner, "configure_sdk_model_defaults", lambda _settings: None)
|
||||
monkeypatch.setattr(
|
||||
runner, "uses_chat_completions_tool_schema", lambda _model, _settings: False
|
||||
)
|
||||
monkeypatch.setattr(todo_tools, "hydrate_todos_from_disk", lambda _state_dir: None)
|
||||
monkeypatch.setattr(notes_tools, "hydrate_notes_from_disk", lambda _state_dir: None)
|
||||
|
||||
async def _create_or_reuse(*_args: Any, **_kwargs: Any) -> dict[str, Any]:
|
||||
return {"client": object(), "session": object(), "caido_client": None}
|
||||
|
||||
async def _cleanup(*_args: Any, **_kwargs: Any) -> None:
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(session_manager, "create_or_reuse", _create_or_reuse)
|
||||
monkeypatch.setattr(session_manager, "cleanup", _cleanup)
|
||||
monkeypatch.setattr(runner, "build_root_task", lambda _scan_config: "task")
|
||||
monkeypatch.setattr(runner, "build_scope_context", lambda _scan_config: "")
|
||||
monkeypatch.setattr(runner, "make_model_settings", lambda *_a, **_k: ModelSettings())
|
||||
monkeypatch.setattr(runner, "build_strix_agent", lambda **_kwargs: object())
|
||||
monkeypatch.setattr(runner, "make_child_factory", lambda **_kwargs: lambda **_k: object())
|
||||
monkeypatch.setattr(runner, "open_agent_session", lambda _root_id, _db: object())
|
||||
|
||||
|
||||
def _root_status(coordinator: AgentCoordinator) -> str:
|
||||
roots = [aid for aid, parent in coordinator.parent_of.items() if parent is None]
|
||||
assert len(roots) == 1
|
||||
return coordinator.statuses[roots[0]]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("interrupt", [KeyboardInterrupt, asyncio.CancelledError])
|
||||
@pytest.mark.asyncio
|
||||
async def test_user_interrupt_leaves_the_root_running_for_resume(
|
||||
monkeypatch: pytest.MonkeyPatch, tmp_path: Any, interrupt: type[BaseException]
|
||||
) -> None:
|
||||
_wire_runner(monkeypatch, tmp_path)
|
||||
|
||||
async def _interrupt(*_args: Any, **_kwargs: Any) -> None:
|
||||
raise interrupt()
|
||||
|
||||
monkeypatch.setattr(runner, "run_agent_loop", _interrupt)
|
||||
coordinator = AgentCoordinator()
|
||||
|
||||
with pytest.raises(interrupt):
|
||||
await runner.run_strix_scan(
|
||||
scan_config={"targets": [], "scan_mode": "deep"},
|
||||
scan_id="scan-test",
|
||||
image="img",
|
||||
coordinator=coordinator,
|
||||
)
|
||||
|
||||
assert _root_status(coordinator) == "running"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_real_crash_still_marks_root_failed(
|
||||
monkeypatch: pytest.MonkeyPatch, tmp_path: Any
|
||||
) -> None:
|
||||
_wire_runner(monkeypatch, tmp_path)
|
||||
|
||||
async def _boom(*_args: Any, **_kwargs: Any) -> None:
|
||||
raise RuntimeError("boom")
|
||||
|
||||
monkeypatch.setattr(runner, "run_agent_loop", _boom)
|
||||
coordinator = AgentCoordinator()
|
||||
|
||||
with pytest.raises(RuntimeError, match="boom"):
|
||||
await runner.run_strix_scan(
|
||||
scan_config={"targets": [], "scan_mode": "deep"},
|
||||
scan_id="scan-test",
|
||||
image="img",
|
||||
coordinator=coordinator,
|
||||
)
|
||||
|
||||
assert _root_status(coordinator) == "failed"
|
||||
@@ -0,0 +1,93 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import types
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from agents import ModelSettings
|
||||
|
||||
import strix.tools.notes.tools as notes_tools
|
||||
import strix.tools.todo.tools as todo_tools
|
||||
from strix.core import runner
|
||||
from strix.core.agents import AgentCoordinator
|
||||
from strix.runtime import session_manager
|
||||
|
||||
|
||||
def _wire_runner(monkeypatch: pytest.MonkeyPatch, tmp_path: Any) -> None:
|
||||
monkeypatch.setattr(runner, "run_dir_for", lambda _scan_id: tmp_path)
|
||||
monkeypatch.setattr(runner, "runtime_state_dir", lambda _run_dir: tmp_path)
|
||||
monkeypatch.setattr(runner, "setup_scan_logging", lambda _run_dir: lambda: None)
|
||||
monkeypatch.setattr(runner, "set_scan_id", lambda _scan_id: None)
|
||||
|
||||
settings = _settings()
|
||||
monkeypatch.setattr(runner, "load_settings", lambda: settings)
|
||||
monkeypatch.setattr(runner, "configure_sdk_model_defaults", lambda _s: None)
|
||||
monkeypatch.setattr(runner, "uses_chat_completions_tool_schema", lambda _m, _s: False)
|
||||
monkeypatch.setattr(todo_tools, "hydrate_todos_from_disk", lambda _d: None)
|
||||
monkeypatch.setattr(notes_tools, "hydrate_notes_from_disk", lambda _d: None)
|
||||
|
||||
async def _create_or_reuse(*_a: Any, **_k: Any) -> dict[str, Any]:
|
||||
return {"client": object(), "session": object(), "caido_client": None}
|
||||
|
||||
async def _cleanup(*_a: Any, **_k: Any) -> None:
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(session_manager, "create_or_reuse", _create_or_reuse)
|
||||
monkeypatch.setattr(session_manager, "cleanup", _cleanup)
|
||||
monkeypatch.setattr(runner, "build_root_task", lambda _c: "task")
|
||||
monkeypatch.setattr(runner, "build_scope_context", lambda _c: "")
|
||||
monkeypatch.setattr(runner, "make_model_settings", lambda *_a, **_k: ModelSettings())
|
||||
monkeypatch.setattr(runner, "build_strix_agent", lambda **_k: object())
|
||||
monkeypatch.setattr(runner, "make_child_factory", lambda **_k: lambda **_kk: object())
|
||||
monkeypatch.setattr(runner, "open_agent_session", lambda _root_id, _db: object())
|
||||
|
||||
|
||||
def _settings() -> Any:
|
||||
return types.SimpleNamespace(
|
||||
llm=types.SimpleNamespace(
|
||||
model="openai/gpt-4o",
|
||||
reasoning_effort="high",
|
||||
force_required_tool_choice=False,
|
||||
timeout=300,
|
||||
prompt_cache=True,
|
||||
extra_headers=None,
|
||||
),
|
||||
runtime=types.SimpleNamespace(max_context_images=3),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_live_child_is_settled_before_sessions_close(
|
||||
monkeypatch: pytest.MonkeyPatch, tmp_path: Any
|
||||
) -> None:
|
||||
_wire_runner(monkeypatch, tmp_path)
|
||||
coordinator = AgentCoordinator()
|
||||
child_started = asyncio.Event()
|
||||
child_task: dict[str, asyncio.Task[None]] = {}
|
||||
|
||||
async def _root_finishes(**kwargs: Any) -> None:
|
||||
root_id = kwargs["agent_id"]
|
||||
|
||||
async def _child_mid_turn() -> None:
|
||||
child_started.set()
|
||||
await asyncio.sleep(3600)
|
||||
|
||||
await coordinator.register("child", "Child", parent_id=root_id)
|
||||
task = asyncio.create_task(_child_mid_turn())
|
||||
child_task["t"] = task
|
||||
await coordinator.attach_runtime("child", task=task)
|
||||
await child_started.wait()
|
||||
|
||||
monkeypatch.setattr(runner, "run_agent_loop", _root_finishes)
|
||||
|
||||
await runner.run_strix_scan(
|
||||
scan_config={"targets": [], "scan_mode": "deep"},
|
||||
scan_id="scan-test",
|
||||
image="img",
|
||||
coordinator=coordinator,
|
||||
)
|
||||
|
||||
task = child_task["t"]
|
||||
assert task.done(), "the child task was left running past scan teardown"
|
||||
assert task.cancelled(), "the child was not cancelled cleanly on a finish"
|
||||
@@ -0,0 +1,117 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from pathlib import Path
|
||||
from typing import Any, cast
|
||||
|
||||
import pytest
|
||||
|
||||
from strix.core.sessions import open_agent_session
|
||||
|
||||
|
||||
def _count_open_fds() -> int | None:
|
||||
for path in (Path("/proc/self/fd"), Path("/dev/fd")):
|
||||
if path.is_dir():
|
||||
return len(list(path.iterdir()))
|
||||
return None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_sessions_hold_no_descriptors_while_parked(tmp_path: Path) -> None:
|
||||
"""Descriptor use must track live operations, not the number of sessions.
|
||||
|
||||
The SDK keeps a connection per (session, pool thread) open for the session's
|
||||
whole life. An agent parks rather than exits, so its session lives for the
|
||||
scan, and fan-out multiplies those handles until the process runs out of file
|
||||
descriptors (#1018). A session that is not mid-operation should hold none.
|
||||
"""
|
||||
baseline = _count_open_fds()
|
||||
if baseline is None:
|
||||
pytest.skip("no /proc/self/fd or /dev/fd on this platform")
|
||||
|
||||
sessions = [open_agent_session(f"a{i}", tmp_path / f"s{i}.db") for i in range(60)]
|
||||
try:
|
||||
for _ in range(4):
|
||||
await asyncio.gather(
|
||||
*(s.add_items([{"role": "user", "content": "x"}]) for s in sessions)
|
||||
)
|
||||
await asyncio.gather(*(s.get_items() for s in sessions))
|
||||
parked = _count_open_fds()
|
||||
assert parked is not None
|
||||
# 60 parked sessions, yet descriptors are back at the baseline.
|
||||
assert parked - baseline <= 5, f"parked fds grew by {parked - baseline}"
|
||||
finally:
|
||||
for s in sessions:
|
||||
s.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_in_flight_descriptors_track_concurrency_not_session_count(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
baseline = _count_open_fds()
|
||||
if baseline is None:
|
||||
pytest.skip("no /proc/self/fd or /dev/fd on this platform")
|
||||
|
||||
sessions = [open_agent_session(f"a{i}", tmp_path / f"s{i}.db") for i in range(200)]
|
||||
peak = baseline
|
||||
try:
|
||||
|
||||
async def sample() -> None:
|
||||
nonlocal peak
|
||||
for _ in range(500):
|
||||
current = _count_open_fds()
|
||||
if current is not None:
|
||||
peak = max(peak, current)
|
||||
await asyncio.sleep(0)
|
||||
|
||||
async def load() -> None:
|
||||
for _ in range(4):
|
||||
await asyncio.gather(
|
||||
*(s.add_items([{"role": "user", "content": "x"}]) for s in sessions)
|
||||
)
|
||||
|
||||
await asyncio.gather(load(), sample())
|
||||
# 200 sessions, but peak is bounded by the thread pool, well under 200.
|
||||
assert peak - baseline < 100, f"in-flight fds peaked at +{peak - baseline}"
|
||||
finally:
|
||||
for s in sessions:
|
||||
s.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_history_survives_the_per_operation_connection(tmp_path: Path) -> None:
|
||||
session = open_agent_session("agent-1", tmp_path / "agents.db")
|
||||
try:
|
||||
for i in range(30):
|
||||
await session.add_items([{"role": "user", "content": f"m{i}"}])
|
||||
items = [cast("dict[str, Any]", i) for i in await session.get_items()]
|
||||
assert [i["content"] for i in items] == [f"m{i}" for i in range(30)]
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_concurrent_sessions_sharing_one_file_stay_consistent(tmp_path: Path) -> None:
|
||||
db = tmp_path / "shared.db"
|
||||
sessions = [open_agent_session(f"a{i}", db) for i in range(10)]
|
||||
try:
|
||||
await asyncio.gather(
|
||||
*(s.add_items([{"role": "user", "content": s.session_id}]) for s in sessions)
|
||||
)
|
||||
# Each session sees only its own row despite sharing the file.
|
||||
for s in sessions:
|
||||
items = [cast("dict[str, Any]", i) for i in await s.get_items()]
|
||||
assert [i["content"] for i in items] == [s.session_id]
|
||||
finally:
|
||||
for s in sessions:
|
||||
s.close()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_closed_session_refuses_operations(tmp_path: Path) -> None:
|
||||
session = open_agent_session("agent-1", tmp_path / "agents.db")
|
||||
await session.add_items([{"role": "user", "content": "x"}])
|
||||
session.close()
|
||||
with pytest.raises(RuntimeError, match="closed"):
|
||||
await session.add_items([{"role": "user", "content": "y"}])
|
||||
@@ -0,0 +1,105 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from agents.tool_context import ToolContext
|
||||
|
||||
from strix.tools.todo import tools
|
||||
from strix.tools.todo.tools import _coerce_priority, create_todo
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_store() -> Any:
|
||||
tools._todos_storage.clear()
|
||||
yield
|
||||
tools._todos_storage.clear()
|
||||
|
||||
|
||||
async def _create(todos: list[Any], agent_id: str = "root") -> dict[str, Any]:
|
||||
ctx = ToolContext(
|
||||
context={"agent_id": agent_id},
|
||||
tool_name="create_todo",
|
||||
tool_call_id="call-1",
|
||||
tool_arguments="{}",
|
||||
)
|
||||
raw = await create_todo.on_invoke_tool(ctx, json.dumps({"todos": json.dumps(todos)}))
|
||||
return json.loads(raw) # type: ignore[no-any-return]
|
||||
|
||||
|
||||
def test_unknown_priority_falls_back_to_normal() -> None:
|
||||
assert _coerce_priority("medium") == "normal"
|
||||
assert _coerce_priority("urgent") == "normal"
|
||||
assert _coerce_priority("high") == "high"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_one_bad_priority_no_longer_discards_the_batch() -> None:
|
||||
result = await _create(
|
||||
[
|
||||
{"title": "Recon", "priority": "medium"},
|
||||
{"title": "Probe /admin", "priority": "sky-high"},
|
||||
{"title": "Report"},
|
||||
]
|
||||
)
|
||||
|
||||
assert result["success"] is True
|
||||
assert result["created_count"] == 3
|
||||
by_title = {c["title"]: c["priority"] for c in result["created"]}
|
||||
assert by_title["Recon"] == "normal"
|
||||
assert by_title["Probe /admin"] == "normal"
|
||||
assert by_title["Report"] == "normal"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_duplicate_titles_within_a_batch_are_skipped() -> None:
|
||||
result = await _create(
|
||||
[
|
||||
{"title": "Subdomain enumeration"},
|
||||
{"title": "Content discovery"},
|
||||
{"title": "Subdomain enumeration"},
|
||||
{"title": "content discovery"},
|
||||
]
|
||||
)
|
||||
|
||||
assert result["created_count"] == 2
|
||||
assert {c["title"] for c in result["created"]} == {
|
||||
"Subdomain enumeration",
|
||||
"Content discovery",
|
||||
}
|
||||
assert len(result["skipped"]) == 2
|
||||
assert all(s["reason"] == "duplicate title" for s in result["skipped"])
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_title_already_on_the_list_is_not_created_again() -> None:
|
||||
await _create([{"title": "Crawl with katana"}])
|
||||
result = await _create([{"title": "crawl with katana"}, {"title": "JS analysis"}])
|
||||
|
||||
assert [c["title"] for c in result["created"]] == ["JS analysis"]
|
||||
assert [s["title"] for s in result["skipped"]] == ["crawl with katana"]
|
||||
assert result["total_count"] == 2
|
||||
|
||||
|
||||
def test_coerce_never_raises() -> None:
|
||||
assert _coerce_priority("nonsense") == "normal"
|
||||
assert _coerce_priority(None) == "normal"
|
||||
assert _coerce_priority("high") == "high"
|
||||
for value in (2, ["high"], {"p": 1}, True):
|
||||
assert _coerce_priority(value) == "normal" # type: ignore[arg-type]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_non_string_priority_does_not_fail_the_batch() -> None:
|
||||
result = await _create(
|
||||
[
|
||||
{"title": "Recon", "priority": 2},
|
||||
{"title": "Probe", "priority": ["high"]},
|
||||
{"title": "Report"},
|
||||
]
|
||||
)
|
||||
|
||||
assert result["success"] is True
|
||||
assert result["created_count"] == 3
|
||||
assert {c["priority"] for c in result["created"]} == {"normal"}
|
||||
@@ -234,12 +234,11 @@ async def test_confirming_the_mount_starts_the_scan_without_a_target() -> None:
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_declining_the_mount_returns_to_the_start_screen() -> None:
|
||||
started = False
|
||||
async def test_declining_the_mount_runs_without_one() -> None:
|
||||
started: list[bool] = []
|
||||
|
||||
async def start(_verify: bool = True) -> None:
|
||||
nonlocal started
|
||||
started = True
|
||||
async def start(verify: bool = True) -> None:
|
||||
started.append(verify)
|
||||
|
||||
os.environ["STRIX_LLM"] = "anthropic/claude-sonnet-4"
|
||||
os.environ["ANTHROPIC_API_KEY"] = "test-key"
|
||||
@@ -250,14 +249,34 @@ async def test_declining_the_mount_returns_to_the_start_screen() -> None:
|
||||
result = await controller.handle("setup.confirm_mount", {"approved": False})
|
||||
|
||||
assert result == {"approved": False}
|
||||
# Nothing was prepared, so the session goes back to the start screen and can
|
||||
# be launched again.
|
||||
assert started is False
|
||||
# Declining skips the directory; it does not abandon the scan.
|
||||
assert started == [False]
|
||||
assert controller.workspace_mount is None
|
||||
assert controller.pending_workspace_mount is None
|
||||
assert controller.setup_mode is True
|
||||
assert controller.scan_started is False
|
||||
assert controller.scan_state == "setup"
|
||||
assert controller.setup_mode is False
|
||||
assert controller.scan_started is True
|
||||
assert controller.scan_state == "running"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_approving_the_mount_runs_with_it() -> None:
|
||||
started: list[bool] = []
|
||||
|
||||
async def start(verify: bool = True) -> None:
|
||||
started.append(verify)
|
||||
|
||||
os.environ["STRIX_LLM"] = "anthropic/claude-sonnet-4"
|
||||
os.environ["ANTHROPIC_API_KEY"] = "test-key"
|
||||
loader._cached = None
|
||||
controller = TuiController(args(), on_start=start)
|
||||
await controller.handle("setup.start", {"verify": False, "mount_working_dir": True})
|
||||
|
||||
result = await controller.handle("setup.confirm_mount", {"approved": True})
|
||||
|
||||
assert result == {"approved": True}
|
||||
assert started == [False]
|
||||
assert controller.workspace_mount == str(Path.cwd())
|
||||
assert controller.scan_state == "running"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
||||
@@ -7,19 +7,26 @@ shows what the user actually typed; resuming has to match that.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import json
|
||||
import sqlite3
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import pytest
|
||||
|
||||
from strix.core import execution
|
||||
from strix.core.paths import runtime_state_dir
|
||||
from strix.interface.tui.backend.live_view import TuiLiveView as GoTuiLiveView
|
||||
from strix.interface.tui.live_view import TuiLiveView, _is_internal_agent_turn
|
||||
from strix.interface.tui.live_view import (
|
||||
_INTERNAL_TURN_PREFIXES,
|
||||
TuiLiveView,
|
||||
_is_internal_agent_turn,
|
||||
)
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
from types import ModuleType
|
||||
|
||||
|
||||
def _write_run(run_dir: Path, items: list[dict[str, Any]], agent_id: str = "root") -> None:
|
||||
@@ -176,11 +183,60 @@ def test_internal_turn_classifier_matches_every_injected_form() -> None:
|
||||
"[CRITICAL] Turn budget: 480/500 used (96%).",
|
||||
"== Inherited context from parent (background only) ==",
|
||||
"Your previous message ended a turn without a tool call.",
|
||||
"Your previous response ended the autonomous Strix run without a lifecycle tool call.",
|
||||
"Your previous response ended the autonomous run without a lifecycle tool call.",
|
||||
):
|
||||
assert _is_internal_agent_turn(content), content
|
||||
|
||||
|
||||
def _injected_strings(module: ModuleType) -> list[str]:
|
||||
"""Every string a module can inject, and nothing it merely mentions.
|
||||
|
||||
Parsing rather than searching the text keeps comments out of it, so a stale
|
||||
copy of a message left in a comment cannot pass for the message itself. It
|
||||
also joins adjacent literals for free, which the line wrapping needs, and
|
||||
docstrings are dropped because they describe the code rather than run in it.
|
||||
"""
|
||||
tree = ast.parse(Path(module.__file__ or "").read_text(encoding="utf-8"))
|
||||
docstrings = set()
|
||||
for node in ast.walk(tree):
|
||||
if not isinstance(node, ast.Module | ast.ClassDef | ast.FunctionDef | ast.AsyncFunctionDef):
|
||||
continue
|
||||
first = node.body[0] if node.body else None
|
||||
if isinstance(first, ast.Expr) and isinstance(first.value, ast.Constant):
|
||||
docstrings.add(id(first.value))
|
||||
|
||||
literals: list[str] = []
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Constant):
|
||||
if isinstance(node.value, str) and id(node) not in docstrings:
|
||||
literals.append(node.value)
|
||||
elif isinstance(node, ast.JoinedStr):
|
||||
literals.append(
|
||||
"".join(
|
||||
part.value
|
||||
for part in node.values
|
||||
if isinstance(part, ast.Constant) and isinstance(part.value, str)
|
||||
)
|
||||
)
|
||||
return literals
|
||||
|
||||
|
||||
def test_internal_turn_prefixes_still_match_what_is_injected() -> None:
|
||||
"""The classifier copies sentences out of another module, so they can drift.
|
||||
|
||||
Both nudges are written inline in strix.core.execution, so there is nothing to
|
||||
import and compare against. Read them back out of what that module can inject.
|
||||
"""
|
||||
injected = _injected_strings(execution)
|
||||
nudges = [prefix for prefix in _INTERNAL_TURN_PREFIXES if prefix.startswith("Your previous")]
|
||||
assert nudges, "the no-tool-call nudges are no longer in the classifier"
|
||||
for nudge in nudges:
|
||||
assert any(nudge in literal for literal in injected), (
|
||||
f"the classifier expects {nudge!r}, which strix.core.execution no longer "
|
||||
f"injects. A resumed scan would show that nudge as the user's own message."
|
||||
)
|
||||
|
||||
|
||||
def test_internal_turn_classifier_keeps_bracketed_user_text() -> None:
|
||||
"""A leading bracket is not enough: typed text often starts with one."""
|
||||
for content in (
|
||||
|
||||
Reference in New Issue
Block a user