diff --git a/strix/agents/prompts/system_prompt.jinja b/strix/agents/prompts/system_prompt.jinja index 5ea11f14..0e2cf924 100644 --- a/strix/agents/prompts/system_prompt.jinja +++ b/strix/agents/prompts/system_prompt.jinja @@ -61,7 +61,9 @@ ACTION SAFETY POLICY: - Safety mode is {{ system_prompt_context.safety_mode }} and is enforced before tool execution - Target authorization does not grant permission to bypass action safety restrictions - If a command is blocked, follow the returned guidance; do not retry it through alternate quoting, scripts, subprocesses, direct CDP, or another tool -- Browser interactions must be issued as individual direct ``agent-browser`` commands; browser automation embedded in scripts is blocked +- Browser interactions must be issued as individual direct ``agent-browser`` commands; browser automation embedded in scripts, command chains, aliases, or subprocess wrappers is blocked +- Strix assigns the browser session for you; do not override ``--session``, ``--profile``, ``--state``, or CDP connection flags +- If an element-reference action is blocked as stale, take a new snapshot and retry the direct command - Commands that create code and execute it in the same shell call must be split into a creation call and a later execution call so the exact artifact can be inspected {% if system_prompt_context.safety_mode == "observe" %} - Observe mode permits passive target interaction only: do not submit forms, authenticate, upload, click state-changing controls, or send mutating requests diff --git a/strix/skills/tooling/agent_browser.md b/strix/skills/tooling/agent_browser.md index 16c55a45..db074e86 100644 --- a/strix/skills/tooling/agent_browser.md +++ b/strix/skills/tooling/agent_browser.md @@ -6,14 +6,6 @@ description: agent-browser CLI for headless Chrome via shell. Snapshot-and-ref w # agent-browser core -When an action safety mode is active, issue browser interactions as individual -direct `agent-browser` commands. Do not hide browser automation in Python, -shell scripts, command chains, aliases, or subprocess wrappers: embedded -browser control is blocked so each action can be reviewed against the current -snapshot. If a referenced action is blocked as stale, run a new snapshot and -retry the direct command. Strix assigns the live browser session automatically; -do not override `--session`, `--profile`, or CDP connection flags. - Fast browser automation CLI for AI agents. Chrome/Chromium via CDP, no Playwright or Puppeteer dependency. Accessibility-tree snapshots with compact `@eN` refs let agents interact with pages in ~200-400 tokens instead of diff --git a/tests/test_config_loader.py b/tests/test_config_loader.py index f66df4ab..80433593 100644 --- a/tests/test_config_loader.py +++ b/tests/test_config_loader.py @@ -166,27 +166,13 @@ def test_aliases_for_no_alias() -> None: def test_apply_override_and_load_settings_round_trip(tmp_path: Path) -> None: - path = tmp_path / "cli-config.json" - path.write_text( - json.dumps({"env": {"STRIX_LLM": "round-trip-model", "PERPLEXITY_API_KEY": "pk"}}), - encoding="utf-8", - ) - - loader.apply_config_override(path) - settings = loader.load_settings() - - assert settings.llm.model == "round-trip-model" - assert settings.integrations.perplexity_api_key == "pk" - # Second call is memoized -> same object. - assert loader.load_settings() is settings - - -def test_safety_settings_load_from_config(tmp_path: Path) -> None: path = tmp_path / "cli-config.json" path.write_text( json.dumps( { "env": { + "STRIX_LLM": "round-trip-model", + "PERPLEXITY_API_KEY": "pk", "STRIX_SAFETY_MODE": "guarded", "STRIX_SAFETY_MODEL": "openai/safety-model", "STRIX_SAFETY_TIMEOUT": "12", @@ -197,11 +183,15 @@ def test_safety_settings_load_from_config(tmp_path: Path) -> None: ) loader.apply_config_override(path) - settings = loader.load_settings().safety + settings = loader.load_settings() - assert settings.mode == "guarded" - assert settings.model == "openai/safety-model" - assert settings.timeout == 12 + assert settings.llm.model == "round-trip-model" + assert settings.integrations.perplexity_api_key == "pk" + assert settings.safety.mode == "guarded" + assert settings.safety.model == "openai/safety-model" + assert settings.safety.timeout == 12 + # Second call is memoized -> same object. + assert loader.load_settings() is settings def test_apply_config_override_invalidates_cache(tmp_path: Path) -> None: diff --git a/tests/test_safety_evidence.py b/tests/test_safety_evidence.py index 1df19222..fe5aecdd 100644 --- a/tests/test_safety_evidence.py +++ b/tests/test_safety_evidence.py @@ -250,6 +250,7 @@ async def test_browser_ref_uses_prior_snapshot_output() -> None: bundle.cleanup() +@pytest.mark.asyncio @pytest.mark.parametrize( "command", [ @@ -258,19 +259,11 @@ async def test_browser_ref_uses_prior_snapshot_output() -> None: "ls -la; rm -rf /workspace/app", ], ) -def test_separators_beyond_double_operators_are_compound(command: str) -> None: +async def test_destructive_command_chained_to_a_read_command_is_blocked(command: str) -> None: plan = parse_command(command) - assert plan.compound is True assert plan.read_only is False - -@pytest.mark.asyncio -@pytest.mark.parametrize( - "command", - ["ls -la\nrm -rf /workspace/app", "ls & rm -rf /workspace/app"], -) -async def test_destructive_command_chained_to_a_read_command_is_blocked(command: str) -> None: bundle = await _compile(command) try: assert bundle.deterministic_allow is None @@ -537,3 +530,97 @@ def test_mutating_http_requests_are_recognized(command: str, expected: str) -> N def test_passive_http_requests_are_not_flagged() -> None: assert parse_command("curl https://example.test/users").mutating_request is None assert parse_command("curl -X GET https://example.test/users").mutating_request is None + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("command", "expected"), + [ + ('bash -c "rm -rf /workspace/app"', "destructive"), + ('sh -c "rm -rf /workspace/app"', "destructive"), + ('bash -lc "rm -rf /workspace/app"', "destructive"), + ('bash -c "agent-browser click @e3"', "Browser automation embedded"), + ], +) +async def test_shell_inline_source_is_parsed_not_just_stored( + command: str, + expected: str, +) -> None: + """`-c` source is the obvious place to hide a command, so the inner string is parsed + and the same deterministic rules applied to it.""" + bundle = await _compile(command) + try: + assert bundle.deterministic_block is not None + assert expected in bundle.deterministic_block + finally: + bundle.cleanup() + + +@pytest.mark.asyncio +async def test_shell_inline_source_is_recorded_as_an_artifact() -> None: + bundle = await _compile('bash -c "echo hello"') + try: + [artifact] = bundle.packet["artifacts"] + assert artifact["path"] == "" + assert artifact["source"] == "echo hello" + assert artifact["inner_executable"] == "echo" + finally: + bundle.cleanup() + + +@pytest.mark.asyncio +async def test_module_execution_cannot_be_resolved_to_a_script() -> None: + bundle = await _compile("python -m http.server") + try: + assert bundle.complete is False + assert any("-m execution" in reason for reason in bundle.incomplete_reasons) + finally: + bundle.cleanup() + + +@pytest.mark.asyncio +async def test_dependency_count_limit_makes_evidence_incomplete() -> None: + bundle = await compile_evidence( + case_id="case-dependency-limit", + ctx=_ctx( + { + "/workspace/run.py": "import first\nimport second\n", + "/workspace/first.py": "value = 1\n", + "/workspace/second.py": "value = 2\n", + } + ), + arguments={"cmd": "python /workspace/run.py"}, + mode="guarded", + scope={}, + user_instruction="", + settings=SafetySettings(max_dependencies=1), + ) + try: + assert bundle.complete is False + assert any("dependency count" in reason for reason in bundle.incomplete_reasons) + finally: + bundle.cleanup() + + +@pytest.mark.asyncio +async def test_oversized_dependency_closure_makes_evidence_incomplete() -> None: + filler = "#" * 8000 + bundle = await compile_evidence( + case_id="case-byte-limit", + ctx=_ctx( + { + "/workspace/run.py": f"import first\n{filler}", + "/workspace/first.py": filler, + } + ), + arguments={"cmd": "python /workspace/run.py"}, + mode="guarded", + scope={}, + user_instruction="", + settings=SafetySettings(max_total_artifact_bytes=10_000), + ) + try: + assert bundle.complete is False + assert any("total byte limit" in reason for reason in bundle.incomplete_reasons) + finally: + bundle.cleanup() diff --git a/tests/test_safety_prompt.py b/tests/test_safety_prompt.py new file mode 100644 index 00000000..5c79dbec --- /dev/null +++ b/tests/test_safety_prompt.py @@ -0,0 +1,47 @@ +"""Action-safety guidance reaches the agent only when a safety mode is active.""" + +from __future__ import annotations + +import pytest + +from strix.agents.prompt import render_system_prompt + + +# Phrased as prohibitions, so they misdescribe the tools an `off`-mode agent actually has. +_SAFETY_ONLY_PHRASES = [ + "ACTION SAFETY POLICY", + "do not override", + "blocked as stale", + "must be split into a creation call", +] + + +@pytest.mark.parametrize("phrase", _SAFETY_ONLY_PHRASES) +@pytest.mark.parametrize("context", [None, {}, {"safety_mode": "off"}]) +def test_safety_guidance_is_absent_without_a_safety_mode( + phrase: str, + context: dict[str, str] | None, +) -> None: + assert phrase not in render_system_prompt(system_prompt_context=context) + + +@pytest.mark.parametrize("phrase", _SAFETY_ONLY_PHRASES) +@pytest.mark.parametrize("mode", ["guarded", "observe"]) +def test_safety_guidance_is_present_in_a_safety_mode(phrase: str, mode: str) -> None: + assert phrase in render_system_prompt(system_prompt_context={"safety_mode": mode}) + + +def test_browser_skill_carries_no_safety_prohibitions() -> None: + """The browser skill is always loaded, so mode-specific rules do not belong in it.""" + prompt = render_system_prompt(skills=["agent_browser"], system_prompt_context={}) + + assert "agent-browser snapshot" in prompt + for phrase in _SAFETY_ONLY_PHRASES: + assert phrase not in prompt + + +def test_observe_mode_states_its_passive_only_contract() -> None: + prompt = render_system_prompt(system_prompt_context={"safety_mode": "observe"}) + + assert "passive target interaction only" in prompt + assert "Guarded mode permits" not in prompt diff --git a/tests/test_safety_reviewer.py b/tests/test_safety_reviewer.py index 0424431c..3bbcc99c 100644 --- a/tests/test_safety_reviewer.py +++ b/tests/test_safety_reviewer.py @@ -182,3 +182,198 @@ async def test_reviewer_failure_blocks(tmp_path: Path, monkeypatch: MonkeyPatch) assert decision.allowed is False assert decision.source == "review_error" + + +@pytest.fixture +def _patched_sdk(monkeypatch: MonkeyPatch) -> None: + monkeypatch.setattr(reviewer_module, "load_settings", _settings) + monkeypatch.setattr(reviewer_module, "configure_sdk_model_defaults", lambda _settings: None) + monkeypatch.setattr( + reviewer_module.StrixProvider, "get_model", lambda _self, _name: "test-model" + ) + monkeypatch.setattr(reviewer_module, "get_global_report_state", lambda: None) + + +def _bundle(tmp_path: Path, case_id: str) -> EvidenceBundle: + return EvidenceBundle( + case_id=case_id, + root=tmp_path, + packet={"completeness": {"status": "complete"}}, + complete=True, + incomplete_reasons=[], + ) + + +def _verdict_run(verdict: SafetyVerdict) -> Any: + async def fake_run(_agent: Any, **_kwargs: Any) -> _Result: + return _Result(verdict) + + return fake_run + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("_patched_sdk") +async def test_low_confidence_allow_is_refused(tmp_path: Path, monkeypatch: MonkeyPatch) -> None: + """An allow the reviewer is unsure of is the case the threshold exists for.""" + monkeypatch.setattr( + reviewer_module.Runner, + "run", + _verdict_run( + SafetyVerdict( + decision="allow", + risk="medium", + categories=["target_mutation"], + reason="probably fine", + confidence=0.5, + ) + ), + ) + + decision = await SafetyReviewer(inspection_runner=_InspectionRunner()).review( + _bundle(tmp_path, "case-low-confidence") + ) + + assert decision.allowed is False + assert decision.source == "reviewer" + assert "below the 0.75 allow threshold" in decision.reason + assert decision.categories == ("target_mutation",) + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("_patched_sdk") +async def test_confident_allow_passes(tmp_path: Path, monkeypatch: MonkeyPatch) -> None: + monkeypatch.setattr( + reviewer_module.Runner, + "run", + _verdict_run( + SafetyVerdict( + decision="allow", + risk="low", + categories=[], + reason="read only", + confidence=0.8, + ) + ), + ) + + decision = await SafetyReviewer(inspection_runner=_InspectionRunner()).review( + _bundle(tmp_path, "case-confident") + ) + + assert decision.allowed is True + assert decision.source == "reviewer" + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("_patched_sdk") +async def test_block_verdict_is_returned_as_a_block( + tmp_path: Path, + monkeypatch: MonkeyPatch, +) -> None: + monkeypatch.setattr( + reviewer_module.Runner, + "run", + _verdict_run( + SafetyVerdict( + decision="block", + risk="high", + categories=["state_mutation"], + reason="deletes a record", + confidence=0.99, + ) + ), + ) + + decision = await SafetyReviewer(inspection_runner=_InspectionRunner()).review( + _bundle(tmp_path, "case-block") + ) + + assert decision.allowed is False + assert decision.source == "reviewer" + assert decision.reason == "deletes a record" + + +@pytest.mark.asyncio +async def test_missing_model_configuration_blocks( + tmp_path: Path, + monkeypatch: MonkeyPatch, +) -> None: + monkeypatch.setattr( + reviewer_module, + "load_settings", + lambda: SimpleNamespace( + safety=SafetySettings(model=None), + llm=SimpleNamespace(model="", extra_headers=None), + ), + ) + + decision = await SafetyReviewer(inspection_runner=_InspectionRunner()).review( + _bundle(tmp_path, "case-no-model") + ) + + assert decision.allowed is False + assert decision.source == "review_error" + assert decision.categories == ("review_unavailable",) + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("_patched_sdk") +async def test_allow_after_a_failed_inspection_is_refused( + tmp_path: Path, + monkeypatch: MonkeyPatch, +) -> None: + """The reviewer decides from the inspection's own output, so an inspection that failed + must not be able to underwrite an allow.""" + + async def fake_run(_agent: Any, *, context: Any, **_kwargs: Any) -> _Result: + context.incomplete = True + return _Result( + SafetyVerdict( + decision="allow", + risk="low", + categories=[], + reason="looked fine", + confidence=0.99, + ) + ) + + monkeypatch.setattr(reviewer_module.Runner, "run", fake_run) + + decision = await SafetyReviewer(inspection_runner=_InspectionRunner()).review( + _bundle(tmp_path, "case-bad-inspection") + ) + + assert decision.allowed is False + assert decision.categories == ("inspection_incomplete",) + + +@pytest.mark.parametrize( + "output", + [ + "Inspection failed: frozen evidence directory is unavailable.", + "Inspection exit code: 1", + "... output truncated ...", + ], +) +@pytest.mark.asyncio +async def test_inspection_failure_output_is_recognized(tmp_path: Path, output: str) -> None: + """These strings are produced in inspection.py and matched by substring here, so a + reword on either side silently stops marking failed inspections.""" + + class _Failing: + async def run(self, *, evidence_dir: str, script: str) -> str: # noqa: ARG002 + return output + + state = InspectionContext(evidence_dir=str(tmp_path), runner=_Failing()) + ctx = ToolContext( + context=state, + tool_name="run_inspection", + tool_call_id="inspect-1", + tool_arguments="{}", + ) + + await run_inspection.on_invoke_tool( + ctx, json.dumps({"reason": "check", "script": "print('x')"}) + ) + + assert state.incomplete is True diff --git a/tests/test_safety_runtime.py b/tests/test_safety_runtime.py index 0c619cdc..b142fdf3 100644 --- a/tests/test_safety_runtime.py +++ b/tests/test_safety_runtime.py @@ -19,11 +19,45 @@ if TYPE_CHECKING: from pathlib import Path +_SNAPSHOT_HISTORY: list[dict[str, Any]] = [ + { + "type": "function_call", + "name": "exec_command", + "call_id": "snapshot-1", + "arguments": '{"cmd":"agent-browser snapshot -i"}', + }, + { + "type": "function_call_output", + "call_id": "snapshot-1", + "output": '@e3 [button type="submit"] "Search"', + }, +] + + class _InspectionRunner: async def run(self, *, evidence_dir: str, script: str) -> str: return f"unused: {evidence_dir} {script}" +class _StubReviewer: + """Stands in for the model review so a decision's source can be asserted.""" + + def __init__(self, on_review: Any = None) -> None: + self.on_review = on_review + self.calls = 0 + + async def review(self, bundle: Any) -> SafetyDecision: + self.calls += 1 + if self.on_review is not None: + await self.on_review() + return SafetyDecision( + allowed=True, + source="reviewer", + reason="allowed", + case_id=bundle.case_id, + ) + + def _runtime(tmp_path: Path, mode: str) -> SafetyRuntime: return SafetyRuntime( scan_id="scan-1", @@ -37,11 +71,11 @@ def _runtime(tmp_path: Path, mode: str) -> SafetyRuntime: ) -def _ctx() -> Any: +def _ctx(*, agent_id: str = "agent-1", turn_input: list[dict[str, Any]] | None = None) -> Any: return SimpleNamespace( - context={"agent_id": "agent-1", "sandbox_session": object()}, + context={"agent_id": agent_id, "sandbox_session": object()}, tool_call_id="call-1", - turn_input=[], + turn_input=turn_input or [], ) @@ -64,25 +98,37 @@ async def test_known_read_command_executes_without_model_review(tmp_path: Path) @pytest.mark.asyncio -async def test_direct_browser_command_gets_per_agent_session(tmp_path: Path) -> None: +async def test_browser_sessions_are_disjoint_per_agent(tmp_path: Path) -> None: seen: list[str] = [] async def invoke(_ctx: Any, raw_input: str) -> str: seen.append(json.loads(raw_input)["cmd"]) return "snapshot" - result = await _runtime(tmp_path, "guarded").invoke_exec( - ctx=_ctx(), - arguments={"cmd": "agent-browser snapshot -i"}, - invoke_tool=invoke, - ) + runtime = _runtime(tmp_path, "guarded") + for agent_id in ("agent-1", "agent-2"): + result = await runtime.invoke_exec( + ctx=_ctx(agent_id=agent_id), + arguments={"cmd": "agent-browser snapshot -i"}, + invoke_tool=invoke, + ) + assert result == "snapshot" - assert result == "snapshot" - assert seen == ["AGENT_BROWSER_SESSION=strix-scan-1-agent-1 agent-browser snapshot -i"] + assert seen == [ + "AGENT_BROWSER_SESSION=strix-scan-1-agent-1 agent-browser snapshot -i", + "AGENT_BROWSER_SESSION=strix-scan-1-agent-2 agent-browser snapshot -i", + ] @pytest.mark.asyncio -async def test_observe_mode_blocks_browser_click(tmp_path: Path) -> None: +async def test_observe_mode_blocks_a_browser_click_against_a_fresh_snapshot( + tmp_path: Path, +) -> None: + """Without the snapshot the click blocks as incomplete evidence in every mode, so the + observe-mode rule itself would never be exercised.""" + runtime = _runtime(tmp_path, "observe") + reviewer = _StubReviewer() + runtime._reviewer = reviewer invoked = False async def invoke(_ctx: Any, _raw_input: str) -> str: @@ -90,17 +136,35 @@ async def test_observe_mode_blocks_browser_click(tmp_path: Path) -> None: invoked = True return "bad" - result = await _runtime(tmp_path, "observe").invoke_exec( - ctx=_ctx(), + result = await runtime.invoke_exec( + ctx=_ctx(turn_input=_SNAPSHOT_HISTORY), arguments={"cmd": "agent-browser click @e3"}, invoke_tool=invoke, ) payload = json.loads(result) assert payload["status"] == "blocked" + assert payload["safety"]["source"] == "deterministic" + assert payload["safety"]["categories"] == ["target_mutation"] + assert "not passive" in payload["safety"]["reason"] + assert reviewer.calls == 0 assert invoked is False +@pytest.mark.asyncio +async def test_observe_mode_allows_a_passive_browser_read(tmp_path: Path) -> None: + async def invoke(_ctx: Any, _raw_input: str) -> str: + return "snapshot" + + result = await _runtime(tmp_path, "observe").invoke_exec( + ctx=_ctx(), + arguments={"cmd": "agent-browser snapshot -i"}, + invoke_tool=invoke, + ) + + assert result == "snapshot" + + @pytest.mark.asyncio async def test_guarded_repeat_request_fails_closed(tmp_path: Path) -> None: async def invoke(_ctx: Any, _raw_input: str) -> str: @@ -133,23 +197,6 @@ def _script_ctx() -> Any: ) -class _StubReviewer: - def __init__(self, on_review: Any = None) -> None: - self.on_review = on_review - self.calls = 0 - - async def review(self, bundle: Any) -> SafetyDecision: - self.calls += 1 - if self.on_review is not None: - await self.on_review() - return SafetyDecision( - allowed=True, - source="reviewer", - reason="allowed", - case_id=bundle.case_id, - ) - - @pytest.mark.asyncio async def test_write_stdin_is_blocked_in_guarded_mode(tmp_path: Path) -> None: invoked = False @@ -301,3 +348,132 @@ async def test_unchanged_workspace_executes_after_review(tmp_path: Path) -> None ) assert result == "ran" + + +@pytest.mark.asyncio +async def test_observe_mode_blocks_a_workspace_patch(tmp_path: Path) -> None: + invoked = False + + async def invoke(_ctx: Any, _raw_input: str) -> str: + nonlocal invoked + invoked = True + return "bad" + + result = await _runtime(tmp_path, "observe").invoke_mutating_tool( + ctx=_ctx(), + tool_name="apply_patch", + raw_input="{}", + invoke_tool=invoke, + ) + + payload = json.loads(result) + assert payload["status"] == "blocked" + assert payload["safety"]["categories"] == ["state_mutation"] + assert invoked is False + + +@pytest.mark.asyncio +async def test_mutating_tool_is_untouched_when_safety_is_off(tmp_path: Path) -> None: + async def invoke(_ctx: Any, _raw_input: str) -> str: + return "patched" + + result = await _runtime(tmp_path, "off").invoke_mutating_tool( + ctx=_ctx(), + tool_name="apply_patch", + raw_input="{}", + invoke_tool=invoke, + ) + + assert result == "patched" + + +@pytest.mark.asyncio +async def test_guarded_patch_runs_and_advances_the_workspace_epoch(tmp_path: Path) -> None: + """The epoch is what makes a script decision go stale, so the write that invalidates + inspected sources has to advance it.""" + runtime = _runtime(tmp_path, "guarded") + + async def invoke(_ctx: Any, _raw_input: str) -> str: + return "patched" + + before = runtime._workspace_epoch + result = await runtime.invoke_mutating_tool( + ctx=_ctx(), + tool_name="apply_patch", + raw_input="{}", + invoke_tool=invoke, + ) + after = runtime._workspace_epoch + + assert result == "patched" + assert after > before + + +@pytest.mark.asyncio +async def test_a_patch_during_review_invalidates_a_script_decision(tmp_path: Path) -> None: + """End-to-end pairing of the two halves: apply_patch bumps the epoch, and a decision + compiled before it is refused rather than executed against changed sources.""" + runtime = _runtime(tmp_path, "guarded") + + async def patch_during_review() -> None: + await runtime.invoke_mutating_tool( + ctx=_ctx(agent_id="agent-2"), + tool_name="apply_patch", + raw_input="{}", + invoke_tool=_noop_invoke, + ) + + runtime._reviewer = _StubReviewer(patch_during_review) + invoked = False + + async def invoke(_ctx: Any, _raw_input: str) -> str: + nonlocal invoked + invoked = True + return "bad" + + result = await runtime.invoke_exec( + ctx=_script_ctx(), + arguments={"cmd": "python /workspace/app.py"}, + invoke_tool=invoke, + ) + + payload = json.loads(result) + assert payload["safety"]["categories"] == ["stale_evidence"] + assert invoked is False + + +async def _noop_invoke(_ctx: Any, _raw_input: str) -> str: + return "patched" + + +@pytest.mark.asyncio +async def test_a_workspace_command_advances_the_epoch(tmp_path: Path) -> None: + runtime = _runtime(tmp_path, "guarded") + runtime._reviewer = _StubReviewer() + + before = runtime._workspace_epoch + result = await runtime.invoke_exec( + ctx=_script_ctx(), + arguments={"cmd": "python /workspace/app.py"}, + invoke_tool=_noop_invoke, + ) + after = runtime._workspace_epoch + + assert result == "patched" + assert after > before + + +@pytest.mark.asyncio +async def test_a_read_only_command_leaves_the_epoch_alone(tmp_path: Path) -> None: + """A read cannot invalidate another agent's inspected sources, so it must not bump the + epoch; if it did, concurrent reads would spuriously stale each other's decisions.""" + runtime = _runtime(tmp_path, "guarded") + + before = runtime._workspace_epoch + await runtime.invoke_exec( + ctx=_ctx(), + arguments={"cmd": "ls /workspace"}, + invoke_tool=_noop_invoke, + ) + + assert runtime._workspace_epoch == before