mirror of
https://github.com/usestrix/strix.git
synced 2026-10-01 02:03:55 +00:00
_trim_parent_history measured the raw history, but child_initial_input replaces every input_image block with a short placeholder immediately afterwards. A single browser screenshot could therefore exhaust the whole character budget and evict every useful text turn behind it, to make room for base64 the child never receives. Scrub first, then trim, so the budget is spent on what is actually sent. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
530 lines
18 KiB
Python
530 lines
18 KiB
Python
"""Tests for pure input builders in strix.core.inputs."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from itertools import pairwise
|
|
from typing import Any
|
|
|
|
import litellm
|
|
import pytest
|
|
|
|
from strix.config import loader
|
|
from strix.core.inputs import (
|
|
_trim_parent_history,
|
|
build_root_task,
|
|
build_scan_targets,
|
|
build_scope_context,
|
|
child_initial_input,
|
|
make_model_settings,
|
|
)
|
|
|
|
|
|
def _child_kwargs(parent_history: list[Any]) -> dict[str, Any]:
|
|
return {
|
|
"name": "scout",
|
|
"child_id": "agent-2",
|
|
"parent_id": "agent-1",
|
|
"task": "Audit the login flow.",
|
|
"parent_history": parent_history,
|
|
}
|
|
|
|
|
|
def test_child_initial_input_single_message_without_history() -> None:
|
|
result = child_initial_input(**_child_kwargs([]))
|
|
|
|
assert len(result) == 1
|
|
assert result[0]["role"] == "user"
|
|
content = result[0]["content"]
|
|
assert "agent scout (agent-2)" in content
|
|
assert "Audit the login flow." in content
|
|
assert "Inherited context" not in content
|
|
|
|
|
|
def test_child_initial_input_single_message_with_history() -> None:
|
|
history = [{"role": "assistant", "content": "previous work"}]
|
|
result = child_initial_input(**_child_kwargs(history))
|
|
|
|
assert len(result) == 1
|
|
assert result[0]["role"] == "user"
|
|
content = result[0]["content"]
|
|
assert "Inherited context from parent" in content
|
|
assert "previous work" in content
|
|
assert "agent scout (agent-2)" in content
|
|
assert "Audit the login flow." in content
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"parent_history",
|
|
[[], [{"role": "assistant", "content": "previous work"}]],
|
|
)
|
|
def test_child_initial_input_no_consecutive_same_role(parent_history: list[Any]) -> None:
|
|
result = child_initial_input(**_child_kwargs(parent_history))
|
|
|
|
roles = [msg["role"] for msg in result]
|
|
assert all(prev != nxt for prev, nxt in pairwise(roles))
|
|
|
|
|
|
def test_child_initial_input_trims_inherited_history(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
# Cap (20 tokens ~= 80 chars) fits the newest item but not both.
|
|
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "20")
|
|
loader._cached = None
|
|
try:
|
|
history = [
|
|
{"role": "assistant", "content": "oldest work item that should be dropped"},
|
|
{"role": "assistant", "content": "newest work item that should be kept"},
|
|
]
|
|
result = child_initial_input(**_child_kwargs(history))
|
|
finally:
|
|
loader._cached = None
|
|
|
|
content = result[0]["content"]
|
|
assert "newest work item that should be kept" in content
|
|
assert "oldest work item that should be dropped" not in content
|
|
assert "older inherited context dropped" in content
|
|
|
|
|
|
def test_screenshot_does_not_evict_text_history(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
# A screenshot is replaced by a short placeholder before the child ever sees
|
|
# it, so budgeting against the raw base64 would drop useful text turns to
|
|
# make room for bytes that are never sent.
|
|
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "200")
|
|
loader._cached = None
|
|
try:
|
|
history = [
|
|
{"role": "assistant", "content": "earlier finding worth inheriting"},
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{"type": "input_image", "image_url": "data:image/png;base64," + "A" * 20_000}
|
|
],
|
|
},
|
|
]
|
|
result = child_initial_input(**_child_kwargs(history))
|
|
finally:
|
|
loader._cached = None
|
|
|
|
content = result[0]["content"]
|
|
assert "earlier finding worth inheriting" in content
|
|
assert "screenshot omitted from inherited context" in content
|
|
assert "older inherited context dropped" not in content
|
|
|
|
|
|
def test_trim_truncates_oversized_newest_item(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
# One item, larger than the whole budget: keeping it whole would mean the cap
|
|
# bounds nothing at all on the child's first request.
|
|
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "40")
|
|
loader._cached = None
|
|
try:
|
|
history = [{"role": "assistant", "content": "x" * 5000}]
|
|
trimmed = _trim_parent_history(history)
|
|
finally:
|
|
loader._cached = None
|
|
|
|
assert len(trimmed) == 1
|
|
assert len(json.dumps(trimmed[0], ensure_ascii=False)) <= 40 * 4
|
|
assert "truncated to bound token cost" in trimmed[0]["content"]
|
|
|
|
|
|
def test_trim_falls_back_to_marker_when_budget_tiny(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "1")
|
|
loader._cached = None
|
|
try:
|
|
history = [{"role": "assistant", "content": "x" * 5000}]
|
|
trimmed = _trim_parent_history(history)
|
|
finally:
|
|
loader._cached = None
|
|
|
|
assert len(trimmed) == 1
|
|
assert "x" * 100 not in trimmed[0]["content"]
|
|
|
|
|
|
def test_child_initial_input_keeps_full_history_when_cap_disabled(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "0")
|
|
loader._cached = None
|
|
try:
|
|
history = [
|
|
{"role": "assistant", "content": "first item kept"},
|
|
{"role": "assistant", "content": "second item kept"},
|
|
]
|
|
result = child_initial_input(**_child_kwargs(history))
|
|
finally:
|
|
loader._cached = None
|
|
|
|
content = result[0]["content"]
|
|
assert "first item kept" in content
|
|
assert "second item kept" in content
|
|
assert "older inherited context dropped" not in content
|
|
|
|
|
|
def _cache_points(model_name: str) -> Any:
|
|
extra = make_model_settings(None, model_name=model_name).extra_args or {}
|
|
return extra.get("cache_control_injection_points")
|
|
|
|
|
|
def test_make_model_settings_enables_prompt_cache_for_bedrock_claude() -> None:
|
|
assert _cache_points("bedrock/global.anthropic.claude-opus-4-8") == [
|
|
{"location": "message", "role": "system"},
|
|
{"location": "tool_config"},
|
|
{"location": "message", "index": -1},
|
|
]
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"model_name",
|
|
[
|
|
"anthropic/claude-sonnet-4-5",
|
|
"openrouter/anthropic/claude-3.5-sonnet",
|
|
"vertex_ai/claude-sonnet-4-5",
|
|
],
|
|
)
|
|
def test_make_model_settings_enables_prompt_cache_for_non_bedrock_claude(model_name: str) -> None:
|
|
assert _cache_points(model_name) == [
|
|
{"location": "message", "role": "system"},
|
|
{"location": "message", "index": -1},
|
|
]
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"model_name",
|
|
["claude-sonnet-4-5", "openai/claude-sonnet-4-5", "any-llm/anthropic/claude-sonnet-4-5"],
|
|
)
|
|
def test_no_prompt_cache_for_claude_off_the_litellm_route(model_name: str) -> None:
|
|
# These names are served by SDK clients that raise TypeError on LiteLLM-only
|
|
# request kwargs — e.g. a gateway in front of Claude reached with a bare name.
|
|
assert _cache_points(model_name) is None
|
|
|
|
|
|
def test_tool_config_point_not_leaked_to_non_bedrock_claude() -> None:
|
|
# LiteLLM only consumes tool_config on Bedrock; elsewhere it leaks onto the
|
|
# wire and native Anthropic 400s.
|
|
for model in ("anthropic/claude-sonnet-4-5", "openrouter/anthropic/claude-3.5-sonnet"):
|
|
points = _cache_points(model) or []
|
|
assert all(p.get("location") != "tool_config" for p in points)
|
|
|
|
|
|
def test_prompt_cache_can_be_disabled() -> None:
|
|
assert (
|
|
make_model_settings(
|
|
None, model_name="anthropic/claude-sonnet-4-5", prompt_cache=False
|
|
).extra_args
|
|
is None
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize("model_name", ["gpt-5", "vertex_ai/gemini-2.5-pro", "openai/o3"])
|
|
def test_make_model_settings_no_prompt_cache_for_non_claude(model_name: str) -> None:
|
|
assert make_model_settings(None, model_name=model_name).extra_args is None
|
|
|
|
|
|
def test_no_prompt_cache_for_unmapped_bedrock_claude_model(monkeypatch: Any) -> None:
|
|
# A Bedrock Claude model LiteLLM hasn't mapped must run uncached, not crash.
|
|
unmapped = "bedrock/global.anthropic.claude-brand-new-9"
|
|
monkeypatch.setattr(litellm, "model_cost", {}, raising=False)
|
|
if getattr(getattr(litellm, "utils", None), "supports_prompt_caching", None):
|
|
monkeypatch.setattr(litellm.utils, "supports_prompt_caching", lambda *_a, **_k: False)
|
|
|
|
assert make_model_settings(None, model_name=unmapped).extra_args is None
|
|
|
|
|
|
def test_prompt_cache_kept_for_non_bedrock_claude_even_if_unmapped(monkeypatch: Any) -> None:
|
|
# Only Bedrock hard-rejects unknown cache fields, so only Bedrock is guarded.
|
|
monkeypatch.setattr(litellm, "model_cost", {}, raising=False)
|
|
if getattr(getattr(litellm, "utils", None), "supports_prompt_caching", None):
|
|
monkeypatch.setattr(litellm.utils, "supports_prompt_caching", lambda *_a, **_k: False)
|
|
|
|
for model in ("anthropic/claude-brand-new-9", "openrouter/anthropic/claude-brand-new"):
|
|
assert _cache_points(model) == [
|
|
{"location": "message", "role": "system"},
|
|
{"location": "message", "index": -1},
|
|
]
|
|
|
|
|
|
def test_max_reasoning_effort_sent_as_raw_body_field() -> None:
|
|
# "max" is absent from the OpenAI SDK's Reasoning enum, and LiteLLM's DeepSeek
|
|
# mapping collapses every effort to thinking-enabled, so it has to ride along
|
|
# as a raw body field to reach the provider.
|
|
settings = make_model_settings(
|
|
"max", model_name="deepseek/deepseek-v4-flash", request_timeout=30
|
|
)
|
|
assert settings.reasoning is None
|
|
assert settings.extra_args == {"timeout": 30}
|
|
assert settings.extra_body == {"reasoning_effort": "max"}
|
|
|
|
|
|
def test_conversation_tail_breakpoint_moves_with_appended_transcript() -> None:
|
|
# LiteLLM must place the index=-1 cache_control on the last message however
|
|
# long the transcript grows.
|
|
hook_mod = pytest.importorskip("litellm.integrations.anthropic_cache_control_hook")
|
|
apply = hook_mod.AnthropicCacheControlHook._apply_message_injections
|
|
points = _cache_points("bedrock/global.anthropic.claude-opus-4-8")
|
|
msg_points = [p for p in points if p.get("location") == "message"]
|
|
|
|
def last_msg_cache_control(n_turns: int) -> Any:
|
|
messages: list[dict[str, Any]] = [{"role": "system", "content": "stable prompt"}]
|
|
for i in range(n_turns):
|
|
messages.append({"role": "assistant", "content": f"turn {i} action"})
|
|
messages.append({"role": "user", "content": f"turn {i} tool result"})
|
|
processed = apply(msg_points, messages, 4)
|
|
last = processed[-1]
|
|
content = last.get("content")
|
|
if isinstance(content, list):
|
|
return content[-1].get("cache_control")
|
|
return last.get("cache_control")
|
|
|
|
assert last_msg_cache_control(2) == {"type": "ephemeral"}
|
|
assert last_msg_cache_control(20) == {"type": "ephemeral"}
|
|
|
|
|
|
def test_build_root_task_empty_config() -> None:
|
|
assert build_root_task({}) == ""
|
|
|
|
|
|
def test_build_root_task_repository_target() -> None:
|
|
config = {
|
|
"targets": [
|
|
{
|
|
"type": "repository",
|
|
"details": {
|
|
"target_repo": "https://example.com/repo.git",
|
|
"cloned_repo_path": "/workspace/repo",
|
|
"workspace_subdir": "repo",
|
|
},
|
|
},
|
|
],
|
|
}
|
|
task = build_root_task(config)
|
|
|
|
assert "Repositories:" in task
|
|
assert "/workspace/repo" in task
|
|
assert "https://example.com/repo.git" in task
|
|
|
|
|
|
def test_build_root_task_web_application_with_instructions() -> None:
|
|
config = {
|
|
"targets": [
|
|
{"type": "web_application", "details": {"target_url": "https://app.example.com"}},
|
|
],
|
|
"user_instructions": "Focus on auth.",
|
|
}
|
|
task = build_root_task(config)
|
|
|
|
assert "URLs:" in task
|
|
assert "https://app.example.com" in task
|
|
assert "Special instructions: Focus on auth." in task
|
|
|
|
|
|
def test_build_root_task_workspace_mount_is_not_a_target() -> None:
|
|
"""A target-less run gets a working directory, not an assessment scope."""
|
|
config = {
|
|
"targets": [],
|
|
"user_instructions": "Find IDOR in the checkout flow.",
|
|
"workspace_mount": "/Users/me/code/api",
|
|
"workspace_subdir": "api",
|
|
}
|
|
task = build_root_task(config)
|
|
|
|
assert "Working Directory:" in task
|
|
assert "/workspace/api" in task
|
|
assert "No scan target was set" in task
|
|
assert "Special instructions: Find IDOR in the checkout flow." in task
|
|
# It must not be presented as an asset to test.
|
|
for label in ("Local Codebases:", "Repositories:", "URLs:", "IP Addresses:"):
|
|
assert label not in task
|
|
|
|
|
|
def test_build_scope_context_authorizes_nothing_without_targets() -> None:
|
|
"""A mounted workspace grants no authorized scope."""
|
|
scope = build_scope_context(
|
|
{"targets": [], "workspace_mount": "/Users/me/code/api", "workspace_subdir": "api"}
|
|
)
|
|
|
|
assert scope["authorized_targets"] == []
|
|
|
|
|
|
def test_build_root_task_diff_scope() -> None:
|
|
config = {
|
|
"targets": [],
|
|
"diff_scope": {
|
|
"active": True,
|
|
"repos": [
|
|
{
|
|
"workspace_subdir": "repo",
|
|
"analyzable_files_count": 3,
|
|
"deleted_files_count": 2,
|
|
},
|
|
],
|
|
},
|
|
}
|
|
task = build_root_task(config)
|
|
|
|
assert "Scope Constraints:" in task
|
|
assert "3 changed file(s)" in task
|
|
assert "2 deleted file(s)" in task
|
|
|
|
|
|
@pytest.mark.parametrize("model_name", ["openai/o3", "gpt-4o"])
|
|
def test_make_model_settings_forces_required_tool_choice_for_openai_models(
|
|
model_name: str,
|
|
) -> None:
|
|
settings = make_model_settings(
|
|
"none",
|
|
model_name=model_name,
|
|
force_required_tool_choice=True,
|
|
)
|
|
|
|
assert settings.tool_choice == "required"
|
|
|
|
|
|
def test_make_model_settings_skips_required_tool_choice_for_non_openai_models() -> None:
|
|
settings = make_model_settings(
|
|
"none",
|
|
model_name="anthropic/claude-3-7-sonnet-latest",
|
|
force_required_tool_choice=True,
|
|
)
|
|
|
|
assert settings.tool_choice is None
|
|
|
|
|
|
def test_make_model_settings_forces_required_for_routed_openai_model() -> None:
|
|
settings = make_model_settings(
|
|
None,
|
|
model_name="litellm/openai/gpt-4o",
|
|
force_required_tool_choice=True,
|
|
)
|
|
|
|
assert settings.tool_choice == "required"
|
|
|
|
|
|
def test_make_model_settings_forces_required_for_anyllm_routed_openai_model() -> None:
|
|
settings = make_model_settings(
|
|
None,
|
|
model_name="any-llm/openai/gpt-4o",
|
|
force_required_tool_choice=True,
|
|
)
|
|
|
|
assert settings.tool_choice == "required"
|
|
|
|
|
|
def test_make_model_settings_disables_parallel_tool_calls_by_default() -> None:
|
|
assert make_model_settings("none", model_name="gpt-4o").parallel_tool_calls is False
|
|
|
|
|
|
def test_make_model_settings_omits_parallel_tool_calls_without_tools() -> None:
|
|
settings = make_model_settings("none", model_name="gpt-4o", has_tools=False)
|
|
|
|
assert settings.parallel_tool_calls is None
|
|
|
|
|
|
def test_make_model_settings_sets_request_timeout() -> None:
|
|
settings = make_model_settings(
|
|
"none",
|
|
model_name="gpt-4o",
|
|
request_timeout=300.0,
|
|
)
|
|
|
|
assert settings.extra_args is not None
|
|
assert settings.extra_args["timeout"] == 300.0
|
|
|
|
|
|
def test_make_model_settings_omits_timeout_when_unset() -> None:
|
|
settings = make_model_settings("none", model_name="gpt-4o")
|
|
|
|
assert settings.extra_args is None
|
|
|
|
|
|
def test_make_model_settings_sets_extra_headers() -> None:
|
|
settings = make_model_settings(
|
|
"none",
|
|
model_name="openai/some-model",
|
|
extra_headers={"X-Feature-Key": "svc", "X-Tenant": "acme"},
|
|
)
|
|
|
|
assert settings.extra_headers == {"X-Feature-Key": "svc", "X-Tenant": "acme"}
|
|
|
|
|
|
def test_make_model_settings_omits_extra_headers_when_unset() -> None:
|
|
assert make_model_settings("none", model_name="gpt-4o").extra_headers is None
|
|
|
|
|
|
def test_make_model_settings_extra_headers_survive_reasoning_resolve() -> None:
|
|
settings = make_model_settings(
|
|
"high",
|
|
model_name="openai/o3",
|
|
extra_headers={"X-Feature-Key": "svc"},
|
|
)
|
|
|
|
assert settings.extra_headers == {"X-Feature-Key": "svc"}
|
|
|
|
|
|
def test_make_model_settings_timeout_survives_reasoning_resolve() -> None:
|
|
# Reasoning is resolved via ModelSettings.resolve(); the timeout in extra_args
|
|
# must not be dropped when a reasoning override is merged in.
|
|
settings = make_model_settings(
|
|
"high",
|
|
model_name="openai/o3",
|
|
request_timeout=120.0,
|
|
)
|
|
|
|
assert settings.extra_args is not None
|
|
assert settings.extra_args["timeout"] == 120.0
|
|
|
|
|
|
def test_scan_targets_prefer_the_workspace_checkout_over_the_remote_url() -> None:
|
|
config = {
|
|
"targets": [
|
|
{
|
|
"type": "repository",
|
|
"details": {
|
|
"target_repo": "https://github.com/acme/billing",
|
|
"workspace_subdir": "billing",
|
|
},
|
|
},
|
|
{"type": "web_application", "details": {"target_url": "https://app.example.com"}},
|
|
]
|
|
}
|
|
|
|
assert build_scan_targets(config) == ["/workspace/billing", "https://app.example.com"]
|
|
|
|
|
|
def test_scan_targets_drop_empty_and_duplicate_entries() -> None:
|
|
config = {
|
|
"targets": [
|
|
{"type": "web_application", "details": {"target_url": "https://app.example.com"}},
|
|
{"type": "web_application", "details": {"target_url": "https://app.example.com"}},
|
|
{"type": "ip_address", "details": {}},
|
|
]
|
|
}
|
|
|
|
assert build_scan_targets(config) == ["https://app.example.com"]
|
|
|
|
|
|
def test_openrouter_attribution_rides_on_the_request_headers() -> None:
|
|
# litellm.headers is ignored once a request carries any header of its own,
|
|
# so the attribution must be part of the per-request headers.
|
|
headers = make_model_settings(
|
|
None, model_name="openrouter/anthropic/claude-sonnet-4-5"
|
|
).extra_headers
|
|
assert headers == {
|
|
"HTTP-Referer": "https://strix.ai",
|
|
"X-Title": "Strix",
|
|
"X-OpenRouter-Categories": "cli-agent",
|
|
}
|
|
|
|
|
|
def test_openrouter_attribution_absent_for_other_providers() -> None:
|
|
assert make_model_settings(None, model_name="anthropic/claude-sonnet-4-5").extra_headers is None
|
|
|
|
|
|
def test_user_headers_override_openrouter_attribution() -> None:
|
|
headers = make_model_settings(
|
|
None,
|
|
model_name="openrouter/anthropic/claude-sonnet-4-5",
|
|
extra_headers={"X-Title": "Custom", "X-Tenant": "acme"},
|
|
).extra_headers
|
|
assert headers is not None
|
|
assert headers["X-Title"] == "Custom"
|
|
assert headers["X-Tenant"] == "acme"
|
|
assert headers["HTTP-Referer"] == "https://strix.ai"
|