mirror of
https://github.com/usestrix/strix.git
synced 2026-08-16 09:26:39 +02:00
Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1e56e1c010 | ||
|
|
ed6ca662f0 | ||
|
|
907ea79714 |
@@ -81,14 +81,6 @@ Protocol-specific testing techniques.
|
||||
| --------- | ------------------------------------------------ |
|
||||
| `graphql` | GraphQL introspection, batching, resolver issues |
|
||||
|
||||
### Reconnaissance
|
||||
|
||||
Passive discovery and attack-surface mapping techniques.
|
||||
|
||||
| Skill | Coverage |
|
||||
| ----------------- | --------------------------------------------------------------- |
|
||||
| `asset_discovery` | CT, TLS SAN pivoting, passive DNS, and ASN/IP asset enumeration |
|
||||
|
||||
### Tooling
|
||||
|
||||
Sandbox CLI playbooks for core recon and scanning tools.
|
||||
|
||||
@@ -201,7 +201,7 @@ EFFICIENCY TACTICS:
|
||||
VALIDATION REQUIREMENTS:
|
||||
- Full validation required - no assumptions
|
||||
- Demonstrate concrete impact with evidence
|
||||
- Consider business context for severity assessment — check whether the target is a demo/sandbox environment or content meant to be public, and factor that in
|
||||
- Consider business context for severity assessment
|
||||
- Independent verification through subagent
|
||||
- Document complete attack chain
|
||||
- Keep going until you find something that matters
|
||||
@@ -434,7 +434,6 @@ SPECIALIZED TOOLS:
|
||||
PROXY & INTERCEPTION:
|
||||
- Caido CLI - Modern web proxy (already running). Use the proxy tools
|
||||
directly, or import `caido_api` from sandbox Python scripts.
|
||||
- HTTPQL filters (for `list_requests`): quote string values, leave integers unquoted (`resp.code.eq:200`, not `"200"`); combine terms with `AND`/`OR` (there is no `NOT` — use the negated operator `ne`/`ncont`/`nregex`). Numeric fields (`resp.code`, `req.port`) use `eq`/`ne`/`gt`/`gte`/`lt`/`lte`; text fields (`req.host`, `req.path`, `req.method`, `req.raw`) use `cont`/`ncont`/`eq`/`regex`. Example: `resp.code.gte:200 AND resp.code.lt:300 AND req.host.cont:"api"`.
|
||||
- NOTE: If you are seeing proxy errors when sending requests, it usually means you are not sending requests to a correct url/host/port.
|
||||
- Ignore Caido proxy-generated 50x HTML error pages; these are proxy issues (might happen when requesting a wrong host or SSL/TLS issues, etc).
|
||||
|
||||
|
||||
@@ -10,7 +10,6 @@ from agents.models.multi_provider import MultiProvider
|
||||
from agents.retry import (
|
||||
ModelRetryBackoffSettings,
|
||||
ModelRetrySettings,
|
||||
RetryPolicyContext,
|
||||
retry_policies,
|
||||
)
|
||||
|
||||
@@ -21,21 +20,6 @@ if TYPE_CHECKING:
|
||||
from strix.config.settings import Settings
|
||||
|
||||
|
||||
def request_timeout_extra_args(timeout_s: float | None) -> dict[str, float] | None:
|
||||
"""Per-request model timeout; a plain float so ``ModelSettings.to_json_dict()`` stays serializable.""" # noqa: E501
|
||||
if not timeout_s or timeout_s <= 0:
|
||||
return None
|
||||
return {"timeout": timeout_s}
|
||||
|
||||
|
||||
def _retry_statusless_provider_errors(context: RetryPolicyContext) -> bool:
|
||||
"""Retry statusless provider errors (e.g. mid-stream quota/billing), but not aborts."""
|
||||
normalized = context.normalized
|
||||
if normalized.is_abort:
|
||||
return False
|
||||
return normalized.status_code is None
|
||||
|
||||
|
||||
class StrixProvider(MultiProvider):
|
||||
"""Route any non-OpenAI prefix through LiteLLM with the prefix preserved,
|
||||
so users type ``deepseek/deepseek-chat`` rather than
|
||||
@@ -72,7 +56,6 @@ DEFAULT_MODEL_RETRY = ModelRetrySettings(
|
||||
retry_policies.provider_suggested(),
|
||||
retry_policies.network_error(),
|
||||
retry_policies.http_status((429, 500, 502, 503, 504)),
|
||||
_retry_statusless_provider_errors,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
+2
-1
@@ -3,7 +3,6 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from agents.lifecycle import RunHooks
|
||||
@@ -28,6 +27,8 @@ class ReportUsageHooks(RunHooks[dict[str, Any]]):
|
||||
"""Persist SDK-native usage after every model response."""
|
||||
|
||||
def __init__(self, *, model: str, max_budget_usd: float | None = None) -> None:
|
||||
import math
|
||||
|
||||
if max_budget_usd is not None and (
|
||||
not math.isfinite(max_budget_usd) or max_budget_usd <= 0
|
||||
):
|
||||
|
||||
@@ -12,7 +12,6 @@ from strix.config.models import (
|
||||
DEFAULT_MODEL_RETRY,
|
||||
is_known_openai_bare_model,
|
||||
model_supports_reasoning,
|
||||
request_timeout_extra_args,
|
||||
)
|
||||
from strix.core.sessions import scrub_images_from_items
|
||||
|
||||
@@ -127,13 +126,11 @@ def make_model_settings(
|
||||
*,
|
||||
model_name: str,
|
||||
force_required_tool_choice: bool = False,
|
||||
request_timeout: float | None = None,
|
||||
) -> ModelSettings:
|
||||
model_settings = ModelSettings(
|
||||
parallel_tool_calls=False,
|
||||
retry=DEFAULT_MODEL_RETRY,
|
||||
include_usage=True,
|
||||
extra_args=request_timeout_extra_args(request_timeout),
|
||||
)
|
||||
if (
|
||||
reasoning_effort is not None
|
||||
|
||||
@@ -215,7 +215,6 @@ async def run_strix_scan(
|
||||
settings.llm.reasoning_effort,
|
||||
model_name=resolved_model,
|
||||
force_required_tool_choice=settings.llm.force_required_tool_choice,
|
||||
request_timeout=settings.llm.timeout,
|
||||
)
|
||||
run_config = RunConfig(
|
||||
model=resolved_model,
|
||||
@@ -283,6 +282,10 @@ async def run_strix_scan(
|
||||
context: dict[str, Any] = {
|
||||
"coordinator": coordinator,
|
||||
"sandbox_session": bundle["session"],
|
||||
# One ``SharedCaidoClient`` is reused by every agent in the scan
|
||||
# (child contexts are shallow copies via ``dict(parent_ctx)``). It
|
||||
# serializes access to the non-concurrency-safe GraphQL transport
|
||||
# and rebuilds it if it dies mid-scan.
|
||||
"caido_client": bundle["caido_client"],
|
||||
"agent_id": root_id,
|
||||
"parent_id": None,
|
||||
|
||||
@@ -16,7 +16,6 @@ from strix.config.models import (
|
||||
DEFAULT_MODEL_RETRY,
|
||||
StrixProvider,
|
||||
configure_sdk_model_defaults,
|
||||
request_timeout_extra_args,
|
||||
)
|
||||
from strix.report.state import get_global_report_state
|
||||
|
||||
@@ -311,11 +310,7 @@ async def check_duplicate(
|
||||
response = await model.get_response(
|
||||
system_instructions=DEDUPE_SYSTEM_PROMPT,
|
||||
input=user_msg,
|
||||
model_settings=ModelSettings(
|
||||
retry=DEFAULT_MODEL_RETRY,
|
||||
include_usage=True,
|
||||
extra_args=request_timeout_extra_args(settings.llm.timeout),
|
||||
),
|
||||
model_settings=ModelSettings(retry=DEFAULT_MODEL_RETRY, include_usage=True),
|
||||
tools=[],
|
||||
output_schema=None,
|
||||
handoffs=[],
|
||||
|
||||
+10
-102
@@ -534,10 +534,16 @@ def litellm_cost_callback(
|
||||
cost = value
|
||||
|
||||
if cost is None:
|
||||
cost = _usage_reported_cost(completion_response)
|
||||
|
||||
if cost is None:
|
||||
cost = _estimate_response_cost(kwargs, completion_response)
|
||||
usage: Any = getattr(completion_response, "usage", None)
|
||||
if usage is None and isinstance(completion_response, dict):
|
||||
usage = cast("dict[str, Any]", completion_response).get("usage")
|
||||
usage_cost: Any
|
||||
if isinstance(usage, dict):
|
||||
usage_cost = cast("dict[str, Any]", usage).get("cost")
|
||||
else:
|
||||
usage_cost = getattr(usage, "cost", None)
|
||||
if isinstance(usage_cost, int | float) and usage_cost > 0:
|
||||
cost = float(usage_cost)
|
||||
|
||||
if cost is None or cost <= 0:
|
||||
return
|
||||
@@ -548,101 +554,3 @@ def litellm_cost_callback(
|
||||
report_state.record_observed_llm_cost(cost)
|
||||
except Exception:
|
||||
logger.exception("Failed to record observed LiteLLM cost")
|
||||
|
||||
|
||||
def _usage_reported_cost(completion_response: Any) -> float | None:
|
||||
"""Provider-reported cost from the ``usage`` block (e.g. OpenRouter).
|
||||
|
||||
Non-BYOK responses charge everything to ``usage.cost``. BYOK responses
|
||||
charge only the OpenRouter fee to ``usage.cost`` (often 0) and report the
|
||||
provider charge in ``usage.cost_details.upstream_inference_cost``, so the
|
||||
true BYOK total is the sum of the two.
|
||||
"""
|
||||
usage: Any = getattr(completion_response, "usage", None)
|
||||
if usage is None and isinstance(completion_response, dict):
|
||||
usage = cast("dict[str, Any]", completion_response).get("usage")
|
||||
if usage is None:
|
||||
return None
|
||||
|
||||
def _field(container: Any, name: str) -> Any:
|
||||
if isinstance(container, dict):
|
||||
return cast("dict[str, Any]", container).get(name)
|
||||
return getattr(container, name, None)
|
||||
|
||||
total = 0.0
|
||||
usage_cost = _field(usage, "cost")
|
||||
if isinstance(usage_cost, int | float) and usage_cost > 0:
|
||||
total += float(usage_cost)
|
||||
|
||||
if bool(_field(usage, "is_byok")):
|
||||
upstream = _field(_field(usage, "cost_details"), "upstream_inference_cost")
|
||||
if isinstance(upstream, int | float) and upstream > 0:
|
||||
total += float(upstream)
|
||||
|
||||
return total if total > 0 else None
|
||||
|
||||
|
||||
def _estimate_response_cost(kwargs: Any, completion_response: Any) -> float | None:
|
||||
"""Best-effort LiteLLM cost-map estimate when no provider-reported cost exists.
|
||||
|
||||
LiteLLM strips provider cost fields when rebuilding streamed responses and
|
||||
returns no ``response_cost`` for models missing from its cost map, so try
|
||||
the provider-prefixed name, the raw name, and the bare model name.
|
||||
"""
|
||||
from litellm import completion_cost
|
||||
|
||||
model = kwargs.get("model") if isinstance(kwargs, dict) else None
|
||||
if not isinstance(model, str) or not model:
|
||||
if isinstance(completion_response, dict):
|
||||
model = cast("dict[str, Any]", completion_response).get("model")
|
||||
else:
|
||||
model = getattr(completion_response, "model", None)
|
||||
if not isinstance(model, str) or not model:
|
||||
return None
|
||||
|
||||
provider = None
|
||||
litellm_params = kwargs.get("litellm_params") if isinstance(kwargs, dict) else None
|
||||
if isinstance(litellm_params, dict):
|
||||
provider = litellm_params.get("custom_llm_provider")
|
||||
|
||||
usage_payload = _usage_payload(completion_response)
|
||||
if usage_payload is None:
|
||||
return None
|
||||
|
||||
candidates: list[str] = []
|
||||
if isinstance(provider, str) and provider and not model.startswith(f"{provider}/"):
|
||||
candidates.append(f"{provider}/{model}")
|
||||
candidates.append(model)
|
||||
if "/" in model:
|
||||
candidates.append(model.rsplit("/", 1)[-1])
|
||||
|
||||
for candidate in candidates:
|
||||
try:
|
||||
value = completion_cost(
|
||||
completion_response={"model": candidate, "usage": usage_payload},
|
||||
model=candidate,
|
||||
)
|
||||
except Exception: # nosec B112 # noqa: BLE001, S112
|
||||
continue
|
||||
if isinstance(value, int | float) and value > 0:
|
||||
return float(value)
|
||||
return None
|
||||
|
||||
|
||||
def _usage_payload(completion_response: Any) -> dict[str, Any] | None:
|
||||
"""Token counts as a plain dict, detached from the response's provider metadata."""
|
||||
usage: Any = getattr(completion_response, "usage", None)
|
||||
if usage is None and isinstance(completion_response, dict):
|
||||
usage = cast("dict[str, Any]", completion_response).get("usage")
|
||||
if usage is None:
|
||||
return None
|
||||
if hasattr(usage, "model_dump"):
|
||||
usage = usage.model_dump()
|
||||
if not isinstance(usage, dict):
|
||||
return None
|
||||
payload = cast("dict[str, Any]", usage)
|
||||
if not payload.get("total_tokens") and not (
|
||||
payload.get("prompt_tokens") or payload.get("completion_tokens")
|
||||
):
|
||||
return None
|
||||
return payload
|
||||
|
||||
@@ -80,30 +80,72 @@ async def _login_as_guest(
|
||||
raise RuntimeError(f"loginAsGuest failed after {attempts} attempts: {last_err}")
|
||||
|
||||
|
||||
async def bootstrap_caido(
|
||||
async def _aclose_quietly(client: Client) -> None:
|
||||
"""Best-effort close of a client whose setup failed; never raises."""
|
||||
with contextlib.suppress(Exception):
|
||||
await client.aclose()
|
||||
|
||||
|
||||
async def _connect_client(
|
||||
session: BaseSandboxSession,
|
||||
*,
|
||||
host_url: str,
|
||||
container_url: str,
|
||||
) -> Client:
|
||||
"""Connect to the in-container Caido sidecar and select a fresh project."""
|
||||
logger.info("Bootstrapping Caido client (host=%s, container=%s)", host_url, container_url)
|
||||
|
||||
access_token = await _login_as_guest(session, container_url=container_url)
|
||||
|
||||
client = Client(host_url, auth=TokenAuthOptions(token=access_token))
|
||||
await client.connect()
|
||||
return client
|
||||
|
||||
|
||||
async def bootstrap_caido(
|
||||
session: BaseSandboxSession,
|
||||
*,
|
||||
host_url: str,
|
||||
container_url: str,
|
||||
) -> tuple[Client, str]:
|
||||
"""Connect to the in-container Caido sidecar and select a fresh project.
|
||||
|
||||
Returns the connected client and the id of the temporary project it
|
||||
selected. The project id lets :func:`reconnect_caido` rebuild a dead
|
||||
transport while staying on the *same* project (and its captured traffic)
|
||||
instead of creating a new empty one.
|
||||
"""
|
||||
logger.info("Bootstrapping Caido client (host=%s, container=%s)", host_url, container_url)
|
||||
|
||||
client = await _connect_client(session, host_url=host_url, container_url=container_url)
|
||||
try:
|
||||
project = await client.project.create(
|
||||
CreateProjectOptions(name="sandbox", temporary=True),
|
||||
)
|
||||
await client.project.select(project.id)
|
||||
except BaseException:
|
||||
# The connected client never reaches the session bundle if project
|
||||
# setup fails, so close it here to avoid leaking the transport.
|
||||
with contextlib.suppress(Exception):
|
||||
await client.aclose()
|
||||
# Don't leak the connected transport if project setup fails.
|
||||
await _aclose_quietly(client)
|
||||
raise
|
||||
logger.info("Caido project selected: %s", project.id)
|
||||
return client, str(project.id)
|
||||
|
||||
|
||||
async def reconnect_caido(
|
||||
session: BaseSandboxSession,
|
||||
*,
|
||||
host_url: str,
|
||||
container_url: str,
|
||||
project_id: str,
|
||||
) -> Client:
|
||||
"""Rebuild a Caido client after its transport died, keeping the project.
|
||||
|
||||
Re-authenticates, reconnects, and re-selects the existing project so the
|
||||
caller keeps access to the traffic captured before the disconnect.
|
||||
"""
|
||||
logger.info("Reconnecting Caido client (host=%s, project=%s)", host_url, project_id)
|
||||
client = await _connect_client(session, host_url=host_url, container_url=container_url)
|
||||
try:
|
||||
await client.project.select(project_id)
|
||||
except BaseException:
|
||||
# A missing/unavailable project must not leave the freshly-connected
|
||||
# transport dangling — otherwise every retry leaks another one.
|
||||
await _aclose_quietly(client)
|
||||
raise
|
||||
return client
|
||||
|
||||
@@ -5,15 +5,20 @@ from __future__ import annotations
|
||||
import logging
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from agents.sandbox.entries import BaseEntry, LocalDir
|
||||
from agents.sandbox.manifest import Environment, Manifest
|
||||
|
||||
from strix.config import load_settings
|
||||
from strix.runtime.backends import get_backend
|
||||
from strix.runtime.caido_bootstrap import bootstrap_caido
|
||||
from strix.runtime.caido_bootstrap import bootstrap_caido, reconnect_caido
|
||||
from strix.runtime.local_dir_staging import stage_symlink_safe_dir
|
||||
from strix.tools.proxy.caido_api import SharedCaidoClient
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from caido_sdk_client import Client as CaidoClient
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -131,16 +136,24 @@ async def create_or_reuse(
|
||||
host_caido_url = f"{scheme}://{caido_endpoint.host}:{caido_endpoint.port}"
|
||||
logger.debug("Caido host endpoint resolved: %s", host_caido_url)
|
||||
|
||||
caido_client = await bootstrap_caido(
|
||||
caido_client, caido_project_id = await bootstrap_caido(
|
||||
session,
|
||||
host_url=host_caido_url,
|
||||
container_url=container_caido_url,
|
||||
)
|
||||
|
||||
async def _reconnect_caido() -> CaidoClient:
|
||||
return await reconnect_caido(
|
||||
session,
|
||||
host_url=host_caido_url,
|
||||
container_url=container_caido_url,
|
||||
project_id=caido_project_id,
|
||||
)
|
||||
|
||||
bundle = {
|
||||
"client": client,
|
||||
"session": session,
|
||||
"caido_client": caido_client,
|
||||
"caido_client": SharedCaidoClient(caido_client, _reconnect_caido),
|
||||
}
|
||||
_SESSION_CACHE[scan_id] = bundle
|
||||
logger.info("Sandbox session for scan %s ready and cached", scan_id)
|
||||
|
||||
@@ -1,151 +0,0 @@
|
||||
---
|
||||
name: asset-discovery
|
||||
description: Passive asset and attack-surface discovery via certificate transparency, TLS SAN pivoting, passive DNS, and ASN/IP enumeration to find hosts beyond subdomain brute force
|
||||
---
|
||||
|
||||
# Asset Discovery
|
||||
|
||||
Most engagements start from a small seed (one domain, one org name) but the real attack surface is far larger: forgotten hosts, staging/internal-named services, acquisitions, and infrastructure that never appears in a wordlist. Build a broad, deduplicated inventory using passive intelligence — certificate transparency, TLS certificate metadata, passive DNS, and ASN/IP data — then collapse it into a probed, classified attack surface. The aim is coverage and pivoting: every certificate, DNS record, and IP is a lead to more assets.
|
||||
|
||||
Only use this skill when all subdomains and related assets of the target are in scope — broad discovery pulls in hosts far beyond the seed.
|
||||
|
||||
## Attack Surface
|
||||
|
||||
- Hosts discoverable via issued certificates (CT logs) but absent from DNS brute force
|
||||
- Internal/staging/pre-prod hostnames leaked in certificate SAN lists
|
||||
- Sibling and acquisition domains sharing certificates, ASNs, or IP ranges with the seed
|
||||
- Wildcard and short-lived certs revealing naming conventions (`*.internal.example.com`, `k8s-*`, `argocd.*`)
|
||||
- ASN-owned IP ranges hosting services with no DNS name at all
|
||||
- Virtual hosts co-located on shared IPs (multiple apps behind one address)
|
||||
- Non-HTTP services on discovered hosts (databases, brokers, admin ports)
|
||||
|
||||
## High-Value Sources
|
||||
|
||||
### Certificate Transparency (CT)
|
||||
|
||||
CT logs record nearly every publicly-trusted certificate. Query by domain (matches SAN/CN) and by organization name.
|
||||
|
||||
- **crt.sh** (free, no key):
|
||||
- By domain incl. subdomains: `curl -s 'https://crt.sh/?q=%25.example.com&output=json' | jq -r '.[].name_value' | sed 's/^\*\.//' | sort -u`
|
||||
- By organization: `https://crt.sh/?O=Example+Inc&output=json`
|
||||
- **Censys / Shodan / Fofa** (API keys): search certs by `parsed.names`, `parsed.subject.organization`, or a specific `fingerprint_sha256`, then pivot to every host serving that cert.
|
||||
- Cross-check multiple indexes (`certspotter`, Google CT, `chaos`) — no single log is complete.
|
||||
- **Wildcards** (`*.corp.example.com`) reveal internal naming schemes even when individual hosts resolve privately; use them to seed targeted guesses (`grafana.corp`, `ci.corp`, `vault.corp`).
|
||||
|
||||
### TLS Certificate SAN/CN
|
||||
|
||||
- **SAN expansion**: one cert often lists many hostnames (marketing + api + admin + internal) — extract every SAN, not just the queried name.
|
||||
- **Shared-cert pivot**: the same cert fingerprint served on multiple IPs ties disparate assets to one owner.
|
||||
- **Issuer/org pivot**: certs sharing `subject.organization`/`organizationalUnit` frequently belong to the same target.
|
||||
- **Active read** catches names never submitted to public CT: `echo | openssl s_client -connect HOST:443 -servername HOST 2>/dev/null | openssl x509 -noout -text | grep -A1 'Subject Alternative Name'`
|
||||
- **Internal leak signal**: SANs like `localhost`, `*.internal`, `*.svc.cluster.local`, `*.local`, or RFC1918-style names on a public cert expose internal naming and sometimes internal services fronted publicly.
|
||||
|
||||
### Passive DNS
|
||||
|
||||
- Forward-resolve every name (A/AAAA/CNAME); keep CNAME chains — they reveal third-party providers and CDNs.
|
||||
- **Reverse DNS (PTR)** on discovered IPs surfaces co-located hostnames.
|
||||
- **Historical/passive DNS** (SecurityTrails, VirusTotal, `chaos`, passivedns providers) recovers names that no longer resolve but may still front live infra.
|
||||
|
||||
### ASN & IP Ranges
|
||||
|
||||
- Map a known IP to its ASN and netblock: `whois -h whois.cymru.com " -v <IP>"` or a BGP/ASN lookup.
|
||||
- If the org runs its own ASN, enumerate all announced prefixes and treat them as candidate assets.
|
||||
- For cloud-hosted targets the IP belongs to the provider, not the org — pivot via cert/vhost instead of netblock.
|
||||
|
||||
## Recommended Tooling
|
||||
|
||||
Prefer the projectdiscovery suite (already available in the sandbox and pipeline-friendly with JSON output):
|
||||
|
||||
- **`subfinder`** — passive subdomain aggregation across many sources incl. CT: `subfinder -d example.com -all -recursive -silent -oJ -o subs.jsonl`
|
||||
- **`tlsx`** — TLS/cert data at scale; grab SANs and issuer/org to pivot: `tlsx -l hosts.txt -san -cn -tls-version -json -o tls.jsonl`
|
||||
- **`uncover`** — query Shodan/Censys/Fofa/Quake/crt.sh engines from one CLI: `uncover -q 'ssl:"Example Inc"' -e shodan,censys,fofa -json`
|
||||
- **`asnmap`** — org/domain/ASN → CIDR ranges: `asnmap -d example.com -json` / `asnmap -org "Example Inc"`
|
||||
- **`mapcidr`** — expand/aggregate CIDRs into host lists for probing: `mapcidr -cidr 192.0.2.0/24 -o hosts.txt`
|
||||
- **`dnsx`** — fast resolution, PTR, and wildcard filtering: `dnsx -l names.txt -a -aaaa -cname -ptr -resp -json -o dns.jsonl`
|
||||
- **`httpx`** — live probing + cert grab in one pass (see methodology).
|
||||
- **`naabu`** — port sweep for non-HTTP services: `naabu -list hosts.txt -top-ports 100 -verify -silent`
|
||||
|
||||
Also useful: **`amass`** (`amass intel`/`enum` for ASN, cert, and passive sources), **`cero`** (bulk SAN extraction from IPs/ranges), and direct **crt.sh** JSON queries when no keys are configured. Cross-source results — CT + passive DNS + `subfinder` together beat any single source.
|
||||
|
||||
## Key Techniques
|
||||
|
||||
### Iterative Seed Expansion
|
||||
|
||||
Every new name, PTR result, CNAME target, and cert SAN becomes a fresh seed. Loop CT → SAN extraction → passive DNS → ASN/range expansion until the asset set stops growing.
|
||||
|
||||
### Cert-Fingerprint Pivoting
|
||||
|
||||
Search Censys/Shodan (or `uncover`) by a cert's `fingerprint_sha256` to find every other host presenting the same certificate — the strongest cross-asset link for tying acquisitions and shadow infra to the target.
|
||||
|
||||
### Naming-Convention Inference
|
||||
|
||||
Wildcard SANs and observed hostnames expose the org's naming scheme; generate targeted candidates from it (`<service>.<env>.example.com`) rather than blind brute force.
|
||||
|
||||
### IP-First Discovery
|
||||
|
||||
For ASN-owned ranges, sweep IPs directly with `naabu`/`httpx` and read served certs (`tlsx`) to find services that have no DNS name at all.
|
||||
|
||||
## Advanced Techniques
|
||||
|
||||
- **Active SAN harvesting** across whole ranges with `tlsx`/`cero` recovers internal hostnames never logged to public CT.
|
||||
- **Favicon and response hashing** (`httpx -favicon`, hash pivots in Shodan) clusters instances of the same app across unrelated hostnames.
|
||||
- **Vhost differentials**: probe a single IP with multiple `Host:` values to unmask co-located apps behind one address.
|
||||
- **Historical CT/DNS diffing** highlights recently issued certs and newly appearing hosts — high-signal for fresh or misconfigured deployments.
|
||||
|
||||
## Consolidation & Probing
|
||||
|
||||
1. **Dedupe** names and IPs into one inventory; record source(s) per asset for confidence.
|
||||
2. **Live probe** with `httpx`, capturing status/title/tech/server and cert SANs in one pass — each grabbed SAN feeds back as a new seed:
|
||||
`httpx -l hosts.txt -sc -title -server -td -tls-grab -json -o assets.jsonl`
|
||||
3. **Classify** assets by function from title/tech/path signals: app, API, marketing, auth, CI/CD, observability, storage, admin, VCS, mail. Cluster by role, not by a specific product.
|
||||
4. **Port sweep** interesting hosts with `naabu` for non-HTTP services (DBs, caches, brokers, mgmt ports).
|
||||
5. **Prioritize** by exposure and value, then hand each finding to the right specialist skill:
|
||||
- Exposed dashboards / debug / observability / metadata leaks → `information_disclosure`
|
||||
- Login/admin panels with default or weak creds → `weak_password_detection`
|
||||
- Dangling DNS / unclaimed provider resources → `subdomain_takeover`
|
||||
- Cloud consoles/metadata surfaces → `aws` / `gcp` / `kubernetes`
|
||||
|
||||
## Testing Methodology
|
||||
|
||||
1. **Seed** - domains, org/legal names, known IPs, email domains, code-host org
|
||||
2. **Certificate transparency** - pull all logged certs per seed domain and org name (crt.sh, `uncover`)
|
||||
3. **SAN/CN extraction** - parse every Subject CN and SAN with `tlsx`; each new name is a new seed
|
||||
4. **Passive DNS** - resolve forward and reverse with `dnsx`; harvest historical records
|
||||
5. **ASN/IP mapping** - `asnmap` → `mapcidr` to expand owned ranges, then sweep for live hosts
|
||||
6. **Active TLS pivot** - `tlsx`/`cero` on live IPs/ports to grab SANs missing from public CT
|
||||
7. **Consolidate & probe** - dedupe, `httpx` probe, classify, and route to specialists
|
||||
|
||||
## Validation
|
||||
|
||||
1. Confirm each discovered asset actually resolves and serves content (live `httpx` result, not just a passive hit)
|
||||
2. Attribute assets to the target via matching cert org, shared cert fingerprint, or DNS under a seed domain
|
||||
3. Deduplicate vhost aliases and CDN edges down to distinct origins so the surface is not inflated
|
||||
4. Record provenance (which source produced each asset) for reproducibility
|
||||
|
||||
## False Positives
|
||||
|
||||
- CDN/edge hostnames and provider default names that are not org-owned
|
||||
- Shared-hosting neighbors on the same IP (vhost co-tenancy, not the target's asset)
|
||||
- Stale historical DNS entries pointing at reassigned infrastructure
|
||||
- Wildcard-cert-implied hostnames that never actually resolve or serve content
|
||||
|
||||
## Impact
|
||||
|
||||
- Expanded attack surface: forgotten, staging, and internal-named hosts brute force misses
|
||||
- Discovery of misconfigured or unauthenticated services fronted by leaked internal hostnames
|
||||
- Attribution of shadow infra, acquisitions, and sibling domains to the target
|
||||
- A prioritized, classified inventory that feeds every downstream specialist skill
|
||||
|
||||
## Pro Tips
|
||||
|
||||
1. Loop the pipeline — every SAN, PTR, and CNAME target is a new seed until the set converges.
|
||||
2. crt.sh is the cheapest high-yield source (no key); Censys/Shodan via `uncover` add cert-fingerprint and vhost pivoting when keys exist.
|
||||
3. Always cert-grab live hosts with `tlsx` — active SANs catch internal hostnames never sent to public CT.
|
||||
4. Internal-looking SANs (`*.internal`, `*.svc.cluster.local`, staging names) are the highest-signal leads.
|
||||
5. Wildcard SANs reveal naming conventions — seed targeted guesses instead of blind brute force.
|
||||
6. Cluster by function, not product name, so the workflow generalizes to any exposed service.
|
||||
7. Keep JSON output throughout so stages chain cleanly (`subfinder` → `dnsx` → `httpx` → `naabu`).
|
||||
|
||||
## Summary
|
||||
|
||||
Broad passive discovery — CT + TLS SAN pivoting + passive DNS + ASN/IP mapping, looped until convergence — finds the assets brute force misses, especially internal-named and forgotten services leaked through certificates. Build the inventory with the projectdiscovery suite, probe and classify it generically, then route each interesting asset to the specialist skill for its class.
|
||||
@@ -3,7 +3,9 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
import urllib.request
|
||||
@@ -26,6 +28,9 @@ if TYPE_CHECKING:
|
||||
from caido_sdk_client import Client as CaidoClient
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
RequestPart = Literal["request", "response"]
|
||||
SortBy = Literal[
|
||||
"timestamp",
|
||||
@@ -45,6 +50,18 @@ _SITEMAP_PAGE_SIZE = 30
|
||||
_DEFAULT_CAIDO_URL = "http://127.0.0.1:48080"
|
||||
_CLIENT_CACHE: dict[str, Client] = {}
|
||||
_CLIENT_LOCK = asyncio.Lock()
|
||||
|
||||
# Substrings that mean the shared client's transport has died or is being used
|
||||
# concurrently — recoverable by rebuilding the client and retrying once.
|
||||
_CONNECTION_ERROR_MARKERS = (
|
||||
"transport is already connected",
|
||||
"connector is closed",
|
||||
"server disconnected",
|
||||
"session is closed",
|
||||
"cannot write to closing transport",
|
||||
"connection reset",
|
||||
"connection closed",
|
||||
)
|
||||
_REQ_FIELD_MAP: dict[SortBy, tuple[str, str]] = {
|
||||
"timestamp": ("req", "created_at"),
|
||||
"host": ("req", "host"),
|
||||
@@ -91,6 +108,22 @@ async def _new_client() -> Client:
|
||||
return client
|
||||
|
||||
|
||||
async def _safe_aclose(client: Client | None) -> None:
|
||||
"""Close a (possibly dead) client without letting teardown errors escape."""
|
||||
if client is None:
|
||||
return
|
||||
with contextlib.suppress(Exception):
|
||||
await client.aclose()
|
||||
|
||||
|
||||
def _is_connection_error(exc: BaseException) -> bool:
|
||||
message = str(exc).lower()
|
||||
if any(marker in message for marker in _CONNECTION_ERROR_MARKERS):
|
||||
return True
|
||||
cause = exc.__cause__ or exc.__context__
|
||||
return cause is not None and cause is not exc and _is_connection_error(cause)
|
||||
|
||||
|
||||
async def get_client() -> Client:
|
||||
"""Return the shared Caido client, creating it under a lock if needed.
|
||||
|
||||
@@ -106,19 +139,73 @@ async def get_client() -> Client:
|
||||
return client
|
||||
|
||||
|
||||
async def call_with_client[T](fn: Callable[[Client], Awaitable[T]]) -> T:
|
||||
"""Run ``fn`` against the shared client, serialized through ``_CLIENT_LOCK``.
|
||||
async def call_with_client[T](
|
||||
fn: Callable[[Client], Awaitable[T]], *, idempotent: bool = True
|
||||
) -> T:
|
||||
"""Run ``fn`` against the shared client, serialized and reconnect-safe.
|
||||
|
||||
The Caido GraphQL transport is not safe for concurrent use: two in-flight
|
||||
requests race and raise "Transport is already connected". Serializing every
|
||||
proxy call through the lock prevents that.
|
||||
requests race and raise "Transport is already connected". All proxy calls
|
||||
are therefore serialized through ``_CLIENT_LOCK``. If the cached client's
|
||||
transport has since died ("Connector is closed" / "Server disconnected"),
|
||||
the stale client is closed and rebuilt so subsequent calls stop failing
|
||||
against a dead client.
|
||||
|
||||
``fn`` is only re-run automatically when ``idempotent`` is true. For
|
||||
mutations (replay, scope create/update/delete) a connection error may
|
||||
arrive *after* Caido applied the change, so we heal the client for future
|
||||
calls but re-raise instead of risking a double-apply.
|
||||
"""
|
||||
async with _CLIENT_LOCK:
|
||||
client = _CLIENT_CACHE.get("default")
|
||||
if client is None:
|
||||
client = await _new_client()
|
||||
_CLIENT_CACHE["default"] = client
|
||||
return await fn(client)
|
||||
try:
|
||||
return await fn(client)
|
||||
except Exception as exc:
|
||||
if not _is_connection_error(exc):
|
||||
raise
|
||||
new_client = await _new_client()
|
||||
_CLIENT_CACHE["default"] = new_client
|
||||
await _safe_aclose(client)
|
||||
if not idempotent:
|
||||
raise
|
||||
return await fn(new_client)
|
||||
|
||||
|
||||
class SharedCaidoClient:
|
||||
"""Serialized, reconnect-safe wrapper around one host-side Caido client.
|
||||
|
||||
Every agent in a scan shares a single instance (propagated through the
|
||||
shallow-copied run context). ``call`` serializes access — the SDK transport
|
||||
is not concurrency-safe — and, when the transport dies, rebuilds the client
|
||||
via ``reconnect`` (which preserves the Caido project) and closes the dead
|
||||
one, so a transient Caido restart no longer disables proxy tools for the
|
||||
rest of the scan.
|
||||
"""
|
||||
|
||||
def __init__(self, client: Client, reconnect: Callable[[], Awaitable[Client]]) -> None:
|
||||
self._client = client
|
||||
self._reconnect = reconnect
|
||||
self._lock = asyncio.Lock()
|
||||
|
||||
async def call[T](self, fn: Callable[[Client], Awaitable[T]], *, idempotent: bool = True) -> T:
|
||||
async with self._lock:
|
||||
try:
|
||||
return await fn(self._client)
|
||||
except Exception as exc:
|
||||
if not _is_connection_error(exc):
|
||||
raise
|
||||
dead, self._client = self._client, await self._reconnect()
|
||||
await _safe_aclose(dead)
|
||||
if not idempotent:
|
||||
raise
|
||||
return await fn(self._client)
|
||||
|
||||
async def aclose(self) -> None:
|
||||
async with self._lock:
|
||||
await _safe_aclose(self._client)
|
||||
|
||||
|
||||
async def close_client() -> None:
|
||||
@@ -459,7 +546,9 @@ async def repeat_request(
|
||||
)
|
||||
return await replay_send_raw(client, raw=raw, connection=connection)
|
||||
|
||||
return await call_with_client(_run)
|
||||
# A replay mutates server state; don't auto-retry if the transport dies
|
||||
# mid-send (the request may already have been sent).
|
||||
return await call_with_client(_run, idempotent=False)
|
||||
|
||||
|
||||
async def scope_rules(
|
||||
@@ -480,7 +569,8 @@ async def scope_rules(
|
||||
scope_name=scope_name,
|
||||
)
|
||||
|
||||
return await call_with_client(_run)
|
||||
# get/list are read-only and safe to retry; create/update/delete mutate.
|
||||
return await call_with_client(_run, idempotent=action in {"get", "list"})
|
||||
|
||||
|
||||
async def _scope_rules_with_client(
|
||||
@@ -729,9 +819,11 @@ async def view_sitemap_entry(entry_id: str) -> dict[str, Any]:
|
||||
__all__ = [
|
||||
"RequestPart",
|
||||
"ScopeAction",
|
||||
"SharedCaidoClient",
|
||||
"SitemapDepth",
|
||||
"SortBy",
|
||||
"SortOrder",
|
||||
"call_with_client",
|
||||
"close_client",
|
||||
"get_client",
|
||||
"list_requests",
|
||||
|
||||
+77
-48
@@ -2,7 +2,6 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import dataclasses
|
||||
import json
|
||||
import logging
|
||||
@@ -14,14 +13,13 @@ from typing import TYPE_CHECKING, Any, Literal
|
||||
from agents import RunContextWrapper, function_tool
|
||||
|
||||
from strix.tools.proxy import caido_api
|
||||
from strix.tools.proxy.caido_api import SharedCaidoClient
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Awaitable, Callable
|
||||
|
||||
from caido_sdk_client import Client
|
||||
|
||||
from strix.tools.proxy.caido_api import (
|
||||
@@ -31,7 +29,7 @@ if TYPE_CHECKING:
|
||||
SortOrder,
|
||||
)
|
||||
else:
|
||||
from strix.tools.proxy.caido_api import ( # noqa: TC001
|
||||
from strix.tools.proxy.caido_api import (
|
||||
RequestPart,
|
||||
SitemapDepth,
|
||||
SortBy,
|
||||
@@ -41,21 +39,19 @@ else:
|
||||
|
||||
ScopeAction = Literal["get", "list", "create", "update", "delete"]
|
||||
|
||||
# All agents in a scan share one host-side Caido client whose GraphQL transport
|
||||
# is not concurrency-safe (parallel calls raise "Transport is already
|
||||
# connected"). Serialize every host-side proxy call through this lock.
|
||||
_CAIDO_CALL_LOCK = asyncio.Lock()
|
||||
|
||||
def _ctx_proxy(ctx: RunContextWrapper) -> SharedCaidoClient | None:
|
||||
"""Return the scan-wide serialized, reconnect-safe Caido client holder.
|
||||
|
||||
def _ctx_client(ctx: RunContextWrapper) -> Client | None:
|
||||
All agents in a scan share one :class:`SharedCaidoClient` whose GraphQL
|
||||
transport is not concurrency-safe (parallel calls raise "Transport is
|
||||
already connected"). ``SharedCaidoClient.call`` serializes access and
|
||||
rebuilds the transport if it dies mid-scan. Returns ``None`` when no holder
|
||||
is present (e.g. standalone tool invocation outside a scan run).
|
||||
"""
|
||||
inner = ctx.context if isinstance(ctx.context, dict) else {}
|
||||
return inner.get("caido_client")
|
||||
|
||||
|
||||
async def _call[T](client: Client, fn: Callable[[Client], Awaitable[T]]) -> T:
|
||||
"""Run ``fn`` against the shared client, serialized under ``_CAIDO_CALL_LOCK``."""
|
||||
async with _CAIDO_CALL_LOCK:
|
||||
return await fn(client)
|
||||
proxy = inner.get("caido_client")
|
||||
return proxy if isinstance(proxy, SharedCaidoClient) else None
|
||||
|
||||
|
||||
def _to_tool_json(value: Any) -> Any:
|
||||
@@ -97,6 +93,39 @@ def _err(name: str, exc: Exception) -> str:
|
||||
)
|
||||
|
||||
|
||||
_HTTPQL_HINT = (
|
||||
"HTTPQL syntax: quote string values and leave integers unquoted; combine "
|
||||
"terms with AND / OR (there is no NOT). Numeric fields (resp.code, req.port, "
|
||||
"id, roundtrip) use eq/ne/gt/gte/lt/lte; text/byte fields (req.host, req.path, "
|
||||
"req.method, req.raw, resp.raw) use cont/ncont/eq/ne/like/nlike/regex/nregex. "
|
||||
"Example: 'resp.code.gte:200 AND resp.code.lt:300 AND req.host.cont:\"api\"'."
|
||||
)
|
||||
|
||||
|
||||
def _is_httpql_error(exc: Exception) -> bool:
|
||||
message = str(exc).lower()
|
||||
return "httpql" in message or ("filter" in message and "pars" in message)
|
||||
|
||||
|
||||
def _httpql_error(exc: Exception, httpql_filter: str | None) -> str:
|
||||
"""Return an actionable error for a rejected HTTPQL filter.
|
||||
|
||||
Preserves Caido's exact parser message and echoes the offending query so
|
||||
the agent can self-correct instead of retrying the same broken filter.
|
||||
"""
|
||||
logger.info("list_requests rejected HTTPQL filter %r: %s", httpql_filter, exc)
|
||||
return json.dumps(
|
||||
{
|
||||
"success": False,
|
||||
"error": f"Invalid HTTPQL filter: {exc}",
|
||||
"httpql_filter": httpql_filter,
|
||||
"hint": _HTTPQL_HINT,
|
||||
},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
|
||||
|
||||
@function_tool(timeout=120)
|
||||
async def list_requests(
|
||||
ctx: RunContextWrapper,
|
||||
@@ -155,13 +184,12 @@ async def list_requests(
|
||||
sort_order: ``asc`` or ``desc``.
|
||||
scope_id: Restrict to a Caido scope (managed via ``scope_rules``).
|
||||
"""
|
||||
client = _ctx_client(ctx)
|
||||
if client is None:
|
||||
proxy = _ctx_proxy(ctx)
|
||||
if proxy is None:
|
||||
return _no_client()
|
||||
|
||||
try:
|
||||
connection = await _call(
|
||||
client,
|
||||
connection = await proxy.call(
|
||||
lambda client: caido_api.list_requests_with_client(
|
||||
client,
|
||||
httpql_filter=httpql_filter,
|
||||
@@ -170,7 +198,7 @@ async def list_requests(
|
||||
sort_by=sort_by,
|
||||
sort_order=sort_order,
|
||||
scope_id=scope_id,
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
entries = []
|
||||
@@ -224,6 +252,8 @@ async def list_requests(
|
||||
default=str,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
if httpql_filter and _is_httpql_error(exc):
|
||||
return _httpql_error(exc, httpql_filter)
|
||||
return _err("list_requests", exc)
|
||||
|
||||
|
||||
@@ -261,14 +291,13 @@ async def view_request(
|
||||
page: 1-indexed page number (only when no ``search_pattern``).
|
||||
page_size: Lines per page.
|
||||
"""
|
||||
client = _ctx_client(ctx)
|
||||
if client is None:
|
||||
proxy = _ctx_proxy(ctx)
|
||||
if proxy is None:
|
||||
return _no_client()
|
||||
|
||||
try:
|
||||
result = await _call(
|
||||
client,
|
||||
lambda client: caido_api.get_request_with_client(client, request_id, part=part),
|
||||
result = await proxy.call(
|
||||
lambda client: caido_api.get_request_with_client(client, request_id, part=part)
|
||||
)
|
||||
if result is None:
|
||||
return json.dumps(
|
||||
@@ -379,8 +408,8 @@ async def repeat_request(
|
||||
- ``body`` — replace the body string entirely.
|
||||
- ``cookies`` — dict of cookies to add/update.
|
||||
"""
|
||||
client = _ctx_client(ctx)
|
||||
if client is None:
|
||||
proxy = _ctx_proxy(ctx)
|
||||
if proxy is None:
|
||||
return _no_client()
|
||||
mods = modifications or {}
|
||||
|
||||
@@ -402,7 +431,9 @@ async def repeat_request(
|
||||
return await caido_api.replay_send_raw(client, raw=raw, connection=connection)
|
||||
|
||||
try:
|
||||
replay = await _call(client, _do)
|
||||
# A replay mutates target state, so don't auto-retry on a mid-send
|
||||
# transport failure (the request may already have been sent).
|
||||
replay = await proxy.call(_do, idempotent=False)
|
||||
if replay is None:
|
||||
return json.dumps(
|
||||
{"success": False, "error": f"Request {request_id} not found"},
|
||||
@@ -461,19 +492,18 @@ async def list_sitemap(
|
||||
(recursive subtree). Only meaningful with ``parent_id``.
|
||||
page: 1-indexed page (30 entries per page).
|
||||
"""
|
||||
client = _ctx_client(ctx)
|
||||
if client is None:
|
||||
proxy = _ctx_proxy(ctx)
|
||||
if proxy is None:
|
||||
return _no_client()
|
||||
try:
|
||||
payload = await _call(
|
||||
client,
|
||||
payload = await proxy.call(
|
||||
lambda client: caido_api.list_sitemap_with_client(
|
||||
client,
|
||||
scope_id=scope_id,
|
||||
parent_id=parent_id,
|
||||
depth=depth,
|
||||
page=page,
|
||||
),
|
||||
)
|
||||
)
|
||||
return json.dumps(payload, ensure_ascii=False, default=str)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
@@ -495,13 +525,12 @@ async def view_sitemap_entry(
|
||||
Args:
|
||||
entry_id: ID from ``list_sitemap`` (or any nested entry).
|
||||
"""
|
||||
client = _ctx_client(ctx)
|
||||
if client is None:
|
||||
proxy = _ctx_proxy(ctx)
|
||||
if proxy is None:
|
||||
return _no_client()
|
||||
try:
|
||||
payload = await _call(
|
||||
client,
|
||||
lambda client: caido_api.view_sitemap_entry_with_client(client, entry_id),
|
||||
payload = await proxy.call(
|
||||
lambda client: caido_api.view_sitemap_entry_with_client(client, entry_id)
|
||||
)
|
||||
return json.dumps(payload, ensure_ascii=False, default=str)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
@@ -554,13 +583,13 @@ async def scope_rules(
|
||||
scope_id: Required for ``get`` / ``update`` / ``delete``.
|
||||
scope_name: Required for ``create`` / ``update``.
|
||||
"""
|
||||
client = _ctx_client(ctx)
|
||||
if client is None:
|
||||
proxy = _ctx_proxy(ctx)
|
||||
if proxy is None:
|
||||
return _no_client()
|
||||
|
||||
try:
|
||||
if action == "list":
|
||||
scopes = await _call(client, caido_api.scope_list)
|
||||
scopes = await proxy.call(caido_api.scope_list)
|
||||
return json.dumps(
|
||||
{"success": True, "scopes": [_to_tool_json(s) for s in scopes]},
|
||||
ensure_ascii=False,
|
||||
@@ -573,7 +602,7 @@ async def scope_rules(
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
scope = await _call(client, lambda client: caido_api.scope_get(client, scope_id))
|
||||
scope = await proxy.call(lambda client: caido_api.scope_get(client, scope_id))
|
||||
return json.dumps(
|
||||
{"success": True, "scope": _to_tool_json(scope)},
|
||||
ensure_ascii=False,
|
||||
@@ -586,11 +615,11 @@ async def scope_rules(
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
scope = await _call(
|
||||
client,
|
||||
scope = await proxy.call(
|
||||
lambda client: caido_api.scope_create(
|
||||
client, name=scope_name, allowlist=allowlist, denylist=denylist
|
||||
),
|
||||
idempotent=False,
|
||||
)
|
||||
return json.dumps(
|
||||
{"success": True, "scope": _to_tool_json(scope)},
|
||||
@@ -607,11 +636,11 @@ async def scope_rules(
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
scope = await _call(
|
||||
client,
|
||||
scope = await proxy.call(
|
||||
lambda client: caido_api.scope_update(
|
||||
client, scope_id, name=scope_name, allowlist=allowlist, denylist=denylist
|
||||
),
|
||||
idempotent=False,
|
||||
)
|
||||
return json.dumps(
|
||||
{"success": True, "scope": _to_tool_json(scope)},
|
||||
@@ -624,7 +653,7 @@ async def scope_rules(
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
await _call(client, lambda client: caido_api.scope_delete(client, scope_id))
|
||||
await proxy.call(lambda client: caido_api.scope_delete(client, scope_id), idempotent=False)
|
||||
return json.dumps(
|
||||
{
|
||||
"success": True,
|
||||
|
||||
@@ -6,7 +6,6 @@ from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import litellm
|
||||
import pytest
|
||||
|
||||
from strix.config.models import _configure_litellm_compatibility
|
||||
from strix.report.state import litellm_cost_callback
|
||||
@@ -43,111 +42,3 @@ def test_cost_callback_reads_usage_cost_from_mapping_response() -> None:
|
||||
litellm_cost_callback({}, response)
|
||||
|
||||
report_state.record_observed_llm_cost.assert_called_once_with(0.125)
|
||||
|
||||
|
||||
def test_cost_callback_reads_byok_upstream_inference_cost() -> None:
|
||||
report_state = MagicMock()
|
||||
response = SimpleNamespace(
|
||||
usage=SimpleNamespace(
|
||||
cost=0,
|
||||
is_byok=True,
|
||||
cost_details=SimpleNamespace(upstream_inference_cost=6.75e-06),
|
||||
),
|
||||
_hidden_params={},
|
||||
)
|
||||
|
||||
with patch("strix.report.state.get_global_report_state", return_value=report_state):
|
||||
litellm_cost_callback({"response_cost": None}, response)
|
||||
|
||||
report_state.record_observed_llm_cost.assert_called_once_with(6.75e-06)
|
||||
|
||||
|
||||
def test_cost_callback_sums_usage_cost_and_upstream_inference_cost() -> None:
|
||||
report_state = MagicMock()
|
||||
response = {
|
||||
"usage": {
|
||||
"cost": 0.01,
|
||||
"is_byok": True,
|
||||
"cost_details": {"upstream_inference_cost": 0.2},
|
||||
}
|
||||
}
|
||||
|
||||
with patch("strix.report.state.get_global_report_state", return_value=report_state):
|
||||
litellm_cost_callback({}, response)
|
||||
|
||||
report_state.record_observed_llm_cost.assert_called_once_with(pytest.approx(0.21))
|
||||
|
||||
|
||||
def test_cost_callback_ignores_upstream_cost_for_non_byok_responses() -> None:
|
||||
report_state = MagicMock()
|
||||
response = {
|
||||
"usage": {
|
||||
"cost": 0.05,
|
||||
"is_byok": False,
|
||||
"cost_details": {"upstream_inference_cost": 0.04},
|
||||
}
|
||||
}
|
||||
|
||||
with patch("strix.report.state.get_global_report_state", return_value=report_state):
|
||||
litellm_cost_callback({}, response)
|
||||
|
||||
report_state.record_observed_llm_cost.assert_called_once_with(0.05)
|
||||
|
||||
|
||||
def test_cost_callback_estimates_cost_with_provider_prefixed_model() -> None:
|
||||
report_state = MagicMock()
|
||||
response = {"usage": {"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15}}
|
||||
kwargs = {
|
||||
"response_cost": None,
|
||||
"model": "anthropic/claude-sonnet-4.5",
|
||||
"litellm_params": {"custom_llm_provider": "openrouter"},
|
||||
}
|
||||
|
||||
def fake_completion_cost(**kwargs: object) -> float:
|
||||
if kwargs["model"] == "openrouter/anthropic/claude-sonnet-4.5":
|
||||
return 0.5
|
||||
raise ValueError(kwargs["model"])
|
||||
|
||||
with (
|
||||
patch("strix.report.state.get_global_report_state", return_value=report_state),
|
||||
patch("litellm.completion_cost", side_effect=fake_completion_cost),
|
||||
):
|
||||
litellm_cost_callback(kwargs, response)
|
||||
|
||||
report_state.record_observed_llm_cost.assert_called_once_with(0.5)
|
||||
|
||||
|
||||
def test_cost_callback_estimates_cost_with_bare_model_fallback() -> None:
|
||||
report_state = MagicMock()
|
||||
response = {"usage": {"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15}}
|
||||
kwargs = {
|
||||
"response_cost": None,
|
||||
"model": "openai/gpt-4o-mini",
|
||||
"litellm_params": {"custom_llm_provider": "openrouter"},
|
||||
}
|
||||
|
||||
def fake_completion_cost(**kwargs: object) -> float:
|
||||
if kwargs["model"] == "gpt-4o-mini":
|
||||
return 0.025
|
||||
raise ValueError(kwargs["model"])
|
||||
|
||||
with (
|
||||
patch("strix.report.state.get_global_report_state", return_value=report_state),
|
||||
patch("litellm.completion_cost", side_effect=fake_completion_cost),
|
||||
):
|
||||
litellm_cost_callback(kwargs, response)
|
||||
|
||||
report_state.record_observed_llm_cost.assert_called_once_with(0.025)
|
||||
|
||||
|
||||
def test_cost_callback_records_nothing_when_no_cost_available() -> None:
|
||||
report_state = MagicMock()
|
||||
response = {"usage": {"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15}}
|
||||
|
||||
with (
|
||||
patch("strix.report.state.get_global_report_state", return_value=report_state),
|
||||
patch("litellm.completion_cost", side_effect=ValueError("unknown model")),
|
||||
):
|
||||
litellm_cost_callback({"response_cost": None, "model": "x/y"}, response)
|
||||
|
||||
report_state.record_observed_llm_cost.assert_not_called()
|
||||
|
||||
@@ -155,33 +155,3 @@ def test_make_model_settings_forces_required_for_anyllm_routed_openai_model() ->
|
||||
)
|
||||
|
||||
assert settings.tool_choice == "required"
|
||||
|
||||
|
||||
def test_make_model_settings_sets_request_timeout() -> None:
|
||||
settings = make_model_settings(
|
||||
"none",
|
||||
model_name="gpt-4o",
|
||||
request_timeout=300.0,
|
||||
)
|
||||
|
||||
assert settings.extra_args is not None
|
||||
assert settings.extra_args["timeout"] == 300.0
|
||||
|
||||
|
||||
def test_make_model_settings_omits_timeout_when_unset() -> None:
|
||||
settings = make_model_settings("none", model_name="gpt-4o")
|
||||
|
||||
assert settings.extra_args is None
|
||||
|
||||
|
||||
def test_make_model_settings_timeout_survives_reasoning_resolve() -> None:
|
||||
# Reasoning is resolved via ModelSettings.resolve(); the timeout in extra_args
|
||||
# must not be dropped when a reasoning override is merged in.
|
||||
settings = make_model_settings(
|
||||
"high",
|
||||
model_name="openai/o3",
|
||||
request_timeout=120.0,
|
||||
)
|
||||
|
||||
assert settings.extra_args is not None
|
||||
assert settings.extra_args["timeout"] == 120.0
|
||||
|
||||
@@ -1,77 +0,0 @@
|
||||
"""Tests for the model retry policy used by every agent model call.
|
||||
|
||||
The SDK's built-in ``http_status`` policy only retries errors that carry a known
|
||||
HTTP status code. Quota/billing (and other provider-side) failures often surface
|
||||
*inside* a streamed response as a bare error with no status code, so Strix adds a
|
||||
statusless retry policy to ``DEFAULT_MODEL_RETRY`` to keep them recoverable — the
|
||||
behavior the pre-SDK engine had.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
from agents.retry import ModelRetryNormalizedError, RetryPolicyContext
|
||||
|
||||
from strix.config.models import DEFAULT_MODEL_RETRY, _retry_statusless_provider_errors
|
||||
|
||||
|
||||
def _context(normalized: ModelRetryNormalizedError) -> RetryPolicyContext:
|
||||
return RetryPolicyContext(
|
||||
error=RuntimeError("boom"),
|
||||
attempt=1,
|
||||
max_retries=5,
|
||||
stream=True,
|
||||
normalized=normalized,
|
||||
provider_advice=None,
|
||||
)
|
||||
|
||||
|
||||
def _retries(normalized: ModelRetryNormalizedError) -> bool:
|
||||
"""Evaluate the composed DEFAULT_MODEL_RETRY policy for a normalized error."""
|
||||
policy = DEFAULT_MODEL_RETRY.policy
|
||||
assert policy is not None
|
||||
decision = asyncio.run(policy(_context(normalized)))
|
||||
return bool(getattr(decision, "retry", decision))
|
||||
|
||||
|
||||
def test_statusless_error_is_retried() -> None:
|
||||
# A mid-stream quota/billing error arrives with no HTTP status code.
|
||||
assert _retries(ModelRetryNormalizedError(status_code=None)) is True
|
||||
|
||||
|
||||
def test_statusless_abort_is_not_retried() -> None:
|
||||
# A user/client cancellation must never be retried.
|
||||
assert _retries(ModelRetryNormalizedError(status_code=None, is_abort=True)) is False
|
||||
|
||||
|
||||
def test_client_error_is_not_retried() -> None:
|
||||
# A definitive 4xx client error (bad request/auth) is not recoverable.
|
||||
assert _retries(ModelRetryNormalizedError(status_code=400)) is False
|
||||
|
||||
|
||||
def test_rate_limit_and_server_errors_are_retried() -> None:
|
||||
for status in (429, 500, 502, 503, 504):
|
||||
assert _retries(ModelRetryNormalizedError(status_code=status)) is True
|
||||
|
||||
|
||||
def test_timeout_error_is_retried() -> None:
|
||||
# A stalled model stream trips the per-request read/inactivity timeout, which
|
||||
# the SDK normalizes as a timeout. DEFAULT_MODEL_RETRY must retry it so a hung
|
||||
# turn recovers instead of silently wedging the agent.
|
||||
assert _retries(ModelRetryNormalizedError(is_timeout=True)) is True
|
||||
assert _retries(ModelRetryNormalizedError(is_network_error=True)) is True
|
||||
|
||||
|
||||
def test_policy_helper_matches_statusless_only() -> None:
|
||||
assert _retry_statusless_provider_errors(_context(ModelRetryNormalizedError())) is True
|
||||
assert (
|
||||
_retry_statusless_provider_errors(_context(ModelRetryNormalizedError(status_code=400)))
|
||||
is False
|
||||
)
|
||||
assert (
|
||||
_retry_statusless_provider_errors(
|
||||
_context(ModelRetryNormalizedError(status_code=None, is_abort=True))
|
||||
)
|
||||
is False
|
||||
)
|
||||
+1
-23
@@ -3,13 +3,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from agents.model_settings import ModelSettings
|
||||
|
||||
from strix.config.models import (
|
||||
RECOMMENDED_MODEL_NAMES,
|
||||
is_recommended_or_frontier_model,
|
||||
request_timeout_extra_args,
|
||||
)
|
||||
from strix.config.models import RECOMMENDED_MODEL_NAMES, is_recommended_or_frontier_model
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model_name", RECOMMENDED_MODEL_NAMES)
|
||||
@@ -17,23 +12,6 @@ def test_recommended_models_are_accepted(model_name: str) -> None:
|
||||
assert is_recommended_or_frontier_model(model_name)
|
||||
|
||||
|
||||
def test_request_timeout_extra_args_positive() -> None:
|
||||
assert request_timeout_extra_args(300) == {"timeout": 300}
|
||||
assert request_timeout_extra_args(10) == {"timeout": 10}
|
||||
|
||||
|
||||
def test_request_timeout_extra_args_survives_model_settings_json_dump() -> None:
|
||||
"""The Chat Completions and LiteLLM paths pydantic-serialize ModelSettings for
|
||||
their tracing span; a non-JSON-serializable timeout fails every turn there."""
|
||||
settings = ModelSettings(extra_args=request_timeout_extra_args(300))
|
||||
assert settings.to_json_dict()["extra_args"] == {"timeout": 300}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("value", [None, 0, -1])
|
||||
def test_request_timeout_extra_args_disabled(value: float | None) -> None:
|
||||
assert request_timeout_extra_args(value) is None
|
||||
|
||||
|
||||
def test_recommended_models_are_matched_case_insensitively() -> None:
|
||||
assert is_recommended_or_frontier_model("Vertex_AI/Gemini-3-Pro-Preview")
|
||||
|
||||
|
||||
+182
-17
@@ -1,19 +1,20 @@
|
||||
"""Tests for the shared Caido client lifecycle and proxy call serialization.
|
||||
"""Tests for the shared Caido client lifecycle and proxy error handling.
|
||||
|
||||
Covers the caching + serialization guarantees of ``caido_api.call_with_client``
|
||||
(the sandbox-imported path) and ``proxy.tools._call`` (the host-side path). The
|
||||
Caido GraphQL transport is not concurrency-safe, so both paths must run one
|
||||
call at a time against the shared client.
|
||||
Covers the concurrency/reconnect guarantees of ``caido_api.call_with_client``
|
||||
(the sandbox-imported path) and ``caido_api.SharedCaidoClient`` (the host-side
|
||||
holder), plus the actionable HTTPQL errors in ``proxy.tools``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
from typing import TYPE_CHECKING, Any, cast
|
||||
|
||||
import pytest
|
||||
|
||||
from strix.tools.proxy import caido_api, tools
|
||||
from strix.tools.proxy.caido_api import SharedCaidoClient
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -90,15 +91,83 @@ async def test_failed_init_does_not_poison_cache(monkeypatch: pytest.MonkeyPatch
|
||||
assert "default" not in caido_api._CLIENT_CACHE
|
||||
|
||||
|
||||
async def test_call_with_client_propagates_errors() -> None:
|
||||
async def test_call_with_client_reconnects_and_closes_dead_transport(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
dead = _FakeClient("dead")
|
||||
fresh = _FakeClient("fresh")
|
||||
caido_api._CLIENT_CACHE["default"] = cast("Any", dead)
|
||||
|
||||
new_calls = {"n": 0}
|
||||
|
||||
async def _new() -> Any:
|
||||
new_calls["n"] += 1
|
||||
return fresh
|
||||
|
||||
monkeypatch.setattr(caido_api, "_new_client", _new)
|
||||
|
||||
attempts: list[Any] = []
|
||||
|
||||
async def fn(client: Any) -> str:
|
||||
attempts.append(client)
|
||||
if len(attempts) == 1:
|
||||
raise RuntimeError("Transport is already connected")
|
||||
return "ok"
|
||||
|
||||
assert await caido_api.call_with_client(fn) == "ok"
|
||||
assert attempts == [dead, fresh]
|
||||
assert new_calls["n"] == 1
|
||||
assert caido_api._CLIENT_CACHE["default"] is fresh
|
||||
assert dead.closed is True # stale transport is not leaked
|
||||
|
||||
|
||||
async def test_call_with_client_non_idempotent_rebuilds_but_reraises(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
dead = _FakeClient("dead")
|
||||
fresh = _FakeClient("fresh")
|
||||
caido_api._CLIENT_CACHE["default"] = cast("Any", dead)
|
||||
|
||||
async def _new() -> Any:
|
||||
return fresh
|
||||
|
||||
monkeypatch.setattr(caido_api, "_new_client", _new)
|
||||
|
||||
calls = {"n": 0}
|
||||
|
||||
async def fn(_client: Any) -> str:
|
||||
calls["n"] += 1
|
||||
raise RuntimeError("Server disconnected")
|
||||
|
||||
# A mutation must not be auto-retried (it may already have applied), but the
|
||||
# dead client is still healed so later calls succeed.
|
||||
with pytest.raises(RuntimeError, match="Server disconnected"):
|
||||
await caido_api.call_with_client(fn, idempotent=False)
|
||||
assert calls["n"] == 1
|
||||
assert caido_api._CLIENT_CACHE["default"] is fresh
|
||||
assert dead.closed is True
|
||||
|
||||
|
||||
async def test_call_with_client_does_not_retry_application_errors(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
cached = _FakeClient("cached")
|
||||
caido_api._CLIENT_CACHE["default"] = cast("Any", cached)
|
||||
|
||||
async def _new() -> Any:
|
||||
raise AssertionError("deterministic errors must not trigger a reconnect")
|
||||
|
||||
monkeypatch.setattr(caido_api, "_new_client", _new)
|
||||
|
||||
calls = {"n": 0}
|
||||
|
||||
async def fn(_client: Any) -> str:
|
||||
calls["n"] += 1
|
||||
raise ValueError("Invalid HTTPQL filter")
|
||||
|
||||
with pytest.raises(ValueError, match="Invalid HTTPQL"):
|
||||
await caido_api.call_with_client(fn)
|
||||
assert calls["n"] == 1
|
||||
assert caido_api._CLIENT_CACHE["default"] is cached
|
||||
|
||||
|
||||
@@ -108,7 +177,7 @@ async def test_call_with_client_serializes_concurrent_calls(
|
||||
caido_api._CLIENT_CACHE["default"] = cast("Any", _FakeClient("shared"))
|
||||
|
||||
async def _new() -> Any:
|
||||
raise AssertionError("no new client expected")
|
||||
raise AssertionError("no reconnect expected")
|
||||
|
||||
monkeypatch.setattr(caido_api, "_new_client", _new)
|
||||
|
||||
@@ -125,8 +194,61 @@ async def test_call_with_client_serializes_concurrent_calls(
|
||||
assert state["max"] == 1
|
||||
|
||||
|
||||
async def test_host_call_serializes_concurrent_calls() -> None:
|
||||
client = _FakeClient("host")
|
||||
async def test_shared_client_reconnects_and_closes_dead_transport() -> None:
|
||||
dead = _FakeClient("dead")
|
||||
fresh = _FakeClient("fresh")
|
||||
|
||||
async def _reconnect() -> Any:
|
||||
return fresh
|
||||
|
||||
holder = SharedCaidoClient(cast("Any", dead), _reconnect)
|
||||
|
||||
attempts: list[Any] = []
|
||||
|
||||
async def fn(client: Any) -> str:
|
||||
attempts.append(client)
|
||||
if len(attempts) == 1:
|
||||
raise RuntimeError("Connector is closed")
|
||||
return "ok"
|
||||
|
||||
assert await holder.call(fn) == "ok"
|
||||
assert attempts == [dead, fresh]
|
||||
assert dead.closed is True
|
||||
|
||||
|
||||
async def test_shared_client_non_idempotent_rebuilds_but_reraises() -> None:
|
||||
dead = _FakeClient("dead")
|
||||
fresh = _FakeClient("fresh")
|
||||
|
||||
async def _reconnect() -> Any:
|
||||
return fresh
|
||||
|
||||
holder = SharedCaidoClient(cast("Any", dead), _reconnect)
|
||||
|
||||
calls = {"n": 0}
|
||||
|
||||
async def fn(_client: Any) -> str:
|
||||
calls["n"] += 1
|
||||
raise RuntimeError("Server disconnected")
|
||||
|
||||
with pytest.raises(RuntimeError, match="Server disconnected"):
|
||||
await holder.call(fn, idempotent=False)
|
||||
assert calls["n"] == 1
|
||||
assert dead.closed is True
|
||||
# The healthy client remains for the next call.
|
||||
assert await holder.call(lambda _c: _ok()) == "ok"
|
||||
|
||||
|
||||
async def _ok() -> str:
|
||||
return "ok"
|
||||
|
||||
|
||||
async def test_shared_client_serializes_concurrent_calls() -> None:
|
||||
async def _reconnect() -> Any:
|
||||
raise AssertionError("no reconnect expected")
|
||||
|
||||
holder = SharedCaidoClient(cast("Any", _FakeClient("shared")), _reconnect)
|
||||
|
||||
state = {"active": 0, "max": 0}
|
||||
|
||||
async def fn(_client: Any) -> str:
|
||||
@@ -136,21 +258,64 @@ async def test_host_call_serializes_concurrent_calls() -> None:
|
||||
state["active"] -= 1
|
||||
return "ok"
|
||||
|
||||
await asyncio.gather(*(tools._call(cast("Any", client), fn) for _ in range(6)))
|
||||
await asyncio.gather(*(holder.call(fn) for _ in range(6)))
|
||||
assert state["max"] == 1
|
||||
|
||||
|
||||
async def test_shared_client_passes_through_application_errors() -> None:
|
||||
async def _reconnect() -> Any:
|
||||
raise AssertionError("deterministic errors must not trigger a reconnect")
|
||||
|
||||
holder = SharedCaidoClient(cast("Any", _FakeClient("c")), _reconnect)
|
||||
|
||||
async def fn(_client: Any) -> str:
|
||||
raise ValueError("Invalid HTTPQL filter")
|
||||
|
||||
with pytest.raises(ValueError, match="Invalid HTTPQL"):
|
||||
await holder.call(fn)
|
||||
|
||||
|
||||
def test_is_connection_error_matches_markers_and_causes() -> None:
|
||||
assert caido_api._is_connection_error(RuntimeError("Transport is already connected"))
|
||||
assert caido_api._is_connection_error(RuntimeError("Connector is closed"))
|
||||
assert caido_api._is_connection_error(RuntimeError("Server disconnected"))
|
||||
assert not caido_api._is_connection_error(ValueError("Invalid HTTPQL filter"))
|
||||
|
||||
nested = RuntimeError("wrapper")
|
||||
nested.__cause__ = RuntimeError("connection reset by peer")
|
||||
assert caido_api._is_connection_error(nested)
|
||||
|
||||
|
||||
class _Ctx:
|
||||
def __init__(self, context: Any) -> None:
|
||||
self.context = context
|
||||
|
||||
|
||||
def test_ctx_client_returns_client_when_present() -> None:
|
||||
client = _FakeClient("host")
|
||||
got = tools._ctx_client(cast("Any", _Ctx({"caido_client": client})))
|
||||
assert got is client
|
||||
def test_ctx_proxy_returns_holder_when_present() -> None:
|
||||
async def _reconnect() -> Any:
|
||||
raise AssertionError("unused")
|
||||
|
||||
holder = SharedCaidoClient(cast("Any", _FakeClient("c")), _reconnect)
|
||||
got = tools._ctx_proxy(cast("Any", _Ctx({"caido_client": holder})))
|
||||
assert got is holder
|
||||
|
||||
|
||||
def test_ctx_client_returns_none_without_client() -> None:
|
||||
assert tools._ctx_client(cast("Any", _Ctx({}))) is None
|
||||
assert tools._ctx_client(cast("Any", _Ctx(None))) is None
|
||||
def test_ctx_proxy_returns_none_without_holder() -> None:
|
||||
assert tools._ctx_proxy(cast("Any", _Ctx({}))) is None
|
||||
assert tools._ctx_proxy(cast("Any", _Ctx(None))) is None
|
||||
assert tools._ctx_proxy(cast("Any", _Ctx({"caido_client": object()}))) is None
|
||||
|
||||
|
||||
def test_is_httpql_error_detection() -> None:
|
||||
assert tools._is_httpql_error(RuntimeError("HTTPQL parse error at column 4"))
|
||||
assert tools._is_httpql_error(RuntimeError("failed to parse filter"))
|
||||
assert not tools._is_httpql_error(RuntimeError("Transport is already connected"))
|
||||
|
||||
|
||||
def test_httpql_error_preserves_message_and_query() -> None:
|
||||
exc = RuntimeError("HTTPQL parse error: unexpected token at column 12")
|
||||
payload = json.loads(tools._httpql_error(exc, 'resp.code.eq:"200"'))
|
||||
assert payload["success"] is False
|
||||
assert "unexpected token at column 12" in payload["error"]
|
||||
assert payload["httpql_filter"] == 'resp.code.eq:"200"'
|
||||
assert "AND / OR" in payload["hint"]
|
||||
|
||||
@@ -37,7 +37,6 @@ async def test_persistent_rate_limit_stops_gracefully(
|
||||
model="openai/gpt-4o",
|
||||
reasoning_effort="high",
|
||||
force_required_tool_choice=False,
|
||||
timeout=300,
|
||||
),
|
||||
runtime=types.SimpleNamespace(max_context_images=3),
|
||||
)
|
||||
|
||||
@@ -45,7 +45,6 @@ def _patch_engine_scaffold(
|
||||
model="openai/gpt-4o",
|
||||
reasoning_effort="high",
|
||||
force_required_tool_choice=False,
|
||||
timeout=300,
|
||||
)
|
||||
)
|
||||
monkeypatch.setattr(runner, "load_settings", lambda: settings)
|
||||
@@ -125,7 +124,8 @@ async def test_root_prompt_options_flow_into_root_agent(
|
||||
assert "https://example.com" in instructions_override
|
||||
assert "CUSTOM SCAN PROMPT" in instructions_override
|
||||
assert (
|
||||
"cannot expand, replace, or weaken authorized target constraints" in instructions_override
|
||||
"cannot expand, replace, or weaken authorized target constraints"
|
||||
in instructions_override
|
||||
)
|
||||
assert kwargs["system_prompt_context"] == {
|
||||
**scope_context,
|
||||
|
||||
Reference in New Issue
Block a user