From 954bc0d52773d165eb3c48b4888aa62c15297094 Mon Sep 17 00:00:00 2001 From: ian-at-strix Date: Tue, 29 Sep 2026 17:35:22 -0400 Subject: [PATCH] perf(prompt): load requested skills after a cache point (#1382) Siblings differ only in the skills they were spawned with, but those came first in , so their prompts diverged at 39%. Shared skills and the catalog now come first, and the requested skills follow a cache point, so siblings share 93%. The extra system message takes a fourth Claude breakpoint, so the Bedrock tool_config one goes: the first system breakpoint already covers the tools. Co-authored-by: Claude Opus 5.5 --- strix/agents/prompt.py | 15 ++++++++++++++- strix/agents/prompts/system_prompt.jinja | 17 ++++++++++++++--- strix/core/inputs.py | 22 ++++++++++++---------- tests/test_inputs.py | 1 - tests/test_runner_root_prompt.py | 11 +++++++++++ 5 files changed, 51 insertions(+), 15 deletions(-) diff --git a/strix/agents/prompt.py b/strix/agents/prompt.py index 5cfb8711b..cf53ca671 100644 --- a/strix/agents/prompt.py +++ b/strix/agents/prompt.py @@ -102,6 +102,16 @@ def render_system_prompt( ), ) + shared = { + name.split("/")[-1] + for name in _resolve_skills( + requested=None, + scan_mode=scan_mode, + is_whitebox=is_whitebox, + is_root=is_root, + is_diff_scoped=is_diff_scoped, + ) + } skills_to_load = _resolve_skills( requested=skills, scan_mode=scan_mode, @@ -112,8 +122,11 @@ def render_system_prompt( skill_content = load_skills(skills_to_load) env.globals["get_skill"] = lambda name: skill_content.get(name, "") + # Skills every agent of this kind loads come first, so siblings share them + # as a cached prefix; the ones the caller asked for vary and go after. rendered = env.get_template("system_prompt.jinja").render( - loaded_skill_names=list(skill_content.keys()), + shared_skill_names=[name for name in skill_content if name in shared], + requested_skill_names=[name for name in skill_content if name not in shared], available_skills=get_available_skills(), interactive=interactive, is_root=is_root, diff --git a/strix/agents/prompts/system_prompt.jinja b/strix/agents/prompts/system_prompt.jinja index b42cbc983..4af8ff4ff 100644 --- a/strix/agents/prompts/system_prompt.jinja +++ b/strix/agents/prompts/system_prompt.jinja @@ -493,9 +493,9 @@ Directories: Default user: pentester (sudo available) -{% if loaded_skill_names %} +{% if shared_skill_names %} -{% for skill_name in loaded_skill_names %} +{% for skill_name in shared_skill_names %} <{{ skill_name }}> {{ get_skill(skill_name) }} @@ -505,7 +505,7 @@ Default user: pentester (sudo available) {% if available_skills %} -On-demand specialist skills. Spawn a specialist via `create_agent(skills=[...])`, or pull guidance inline for yourself via `load_skill(skills=[...])`. Anything wrapped in `` above is already loaded for you. +On-demand specialist skills. Spawn a specialist via `create_agent(skills=[...])`, or pull guidance inline for yourself via `load_skill(skills=[...])`. Anything wrapped in `` is already loaded for you. {% for category, skills in available_skills | dictsort -%} {% for skill in skills -%} @@ -515,6 +515,17 @@ On-demand specialist skills. Spawn a specialist via `create_agent(skills=[...])` {% endif %} +{% if requested_skill_names %} + + +{% for skill_name in requested_skill_names %} +<{{ skill_name }}> +{{ get_skill(skill_name) }} + +{% endfor %} + +{% endif %} + {% if include_scope %} {% include "scope.jinja" %} {% endif %} diff --git a/strix/core/inputs.py b/strix/core/inputs.py index 3dd0d701d..c05373865 100644 --- a/strix/core/inputs.py +++ b/strix/core/inputs.py @@ -312,11 +312,12 @@ def _reasoning_settings(effort: ReasoningEffort) -> ModelSettings: def _prompt_cache_extra_args(model_name: str) -> dict[str, Any] | None: """LiteLLM ``cache_control_injection_points`` for Claude prompt caching. - System prompt + rolling last-message breakpoint everywhere; ``tool_config`` - only on Bedrock Converse (the only route whose LiteLLM transform consumes - it — elsewhere it leaks onto the wire and native Anthropic 400s). Unmapped - Bedrock models get no points at all: Bedrock rejects the passed-through - field outright. + A breakpoint on each system message, plus a rolling last-message one. The + system prompt is split into up to three messages, which with the last + message uses all four breakpoints Claude allows. There is none on + ``tool_config``: the tools come before the system prompt, so its first + breakpoint caches them too. Unmapped Bedrock models get no points at all: + Bedrock rejects the passed-through field outright. The field is LiteLLM's own, consumed by its transform, so it only goes to routes LiteLLM serves. A bare ``claude-...`` name is served by the SDK's @@ -328,11 +329,12 @@ def _prompt_cache_extra_args(model_name: str) -> dict[str, Any] | None: if is_bedrock_route(model_name) and not bedrock_route_supports_prompt_caching(model_name): return None - points: list[dict[str, Any]] = [{"location": "message", "role": "system"}] - if is_bedrock_route(model_name): - points.append({"location": "tool_config"}) - points.append({"location": "message", "index": -1}) - return {"cache_control_injection_points": points} + return { + "cache_control_injection_points": [ + {"location": "message", "role": "system"}, + {"location": "message", "index": -1}, + ] + } def child_initial_input( diff --git a/tests/test_inputs.py b/tests/test_inputs.py index 5a483edf2..febe8600a 100644 --- a/tests/test_inputs.py +++ b/tests/test_inputs.py @@ -70,7 +70,6 @@ def _cache_points(model_name: str) -> Any: def test_make_model_settings_enables_prompt_cache_for_bedrock_claude() -> None: assert _cache_points("bedrock/global.anthropic.claude-opus-4-8") == [ {"location": "message", "role": "system"}, - {"location": "tool_config"}, {"location": "message", "index": -1}, ] diff --git a/tests/test_runner_root_prompt.py b/tests/test_runner_root_prompt.py index 393da669f..31d153a71 100644 --- a/tests/test_runner_root_prompt.py +++ b/tests/test_runner_root_prompt.py @@ -6,6 +6,7 @@ flow through to the root agent's ``build_strix_agent`` call. from __future__ import annotations +import os import types from typing import Any @@ -279,6 +280,16 @@ def test_scope_is_rendered_once_at_the_end_of_the_prompt() -> None: assert prompt.index("") < prompt.index("SYSTEM-VERIFIED SCOPE") +def test_requested_skills_follow_the_shared_prefix() -> None: + xss = render_system_prompt(skills=["xss"], include_scope=False) + sqli = render_system_prompt(skills=["sql_injection"], include_scope=False) + + shared = os.path.commonprefix([xss, sqli]) + assert "" in shared + assert shared.count("") == 1 + assert "" in xss.split("")[1] + + def test_scope_is_sent_as_its_own_system_message_on_cache_point_routes() -> None: settings = make_model_settings(None, model_name="anthropic/claude-sonnet-5-5") prompt = render_system_prompt(