perf(prompt): load requested skills after a cache point (#1382)

Siblings differ only in the skills they were spawned with, but those came
first in <specialized_knowledge>, so their prompts diverged at 39%. Shared
skills and the catalog now come first, and the requested skills follow a
cache point, so siblings share 93%.

The extra system message takes a fourth Claude breakpoint, so the Bedrock
tool_config one goes: the first system breakpoint already covers the tools.

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
ian-at-strix 2026-09-29 17:35:22 -04:00 • committed by GitHub
parent e66c56c473
commit 954bc0d527
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 51 additions and 15 deletions

View file

@ -102,6 +102,16 @@ def render_system_prompt(
),
)
shared = {
name.split("/")[-1]
for name in _resolve_skills(
requested=None,
scan_mode=scan_mode,
is_whitebox=is_whitebox,
is_root=is_root,
is_diff_scoped=is_diff_scoped,
)
}
skills_to_load = _resolve_skills(
requested=skills,
scan_mode=scan_mode,
@ -112,8 +122,11 @@ def render_system_prompt(
skill_content = load_skills(skills_to_load)
env.globals["get_skill"] = lambda name: skill_content.get(name, "")
# Skills every agent of this kind loads come first, so siblings share them
# as a cached prefix; the ones the caller asked for vary and go after.
rendered = env.get_template("system_prompt.jinja").render(
loaded_skill_names=list(skill_content.keys()),
shared_skill_names=[name for name in skill_content if name in shared],
requested_skill_names=[name for name in skill_content if name not in shared],
available_skills=get_available_skills(),
interactive=interactive,
is_root=is_root,

View file

@ -493,9 +493,9 @@ Directories:
Default user: pentester (sudo available)
</environment>
{% if loaded_skill_names %}
{% if shared_skill_names %}
<specialized_knowledge>
{% for skill_name in loaded_skill_names %}
{% for skill_name in shared_skill_names %}
<{{ skill_name }}>
{{ get_skill(skill_name) }}
</{{ skill_name }}>
@ -505,7 +505,7 @@ Default user: pentester (sudo available)
{% if available_skills %}
<available_skills>
On-demand specialist skills. Spawn a specialist via `create_agent(skills=[...])`, or pull guidance inline for yourself via `load_skill(skills=[...])`. Anything wrapped in `<specialized_knowledge>` above is already loaded for you.
On-demand specialist skills. Spawn a specialist via `create_agent(skills=[...])`, or pull guidance inline for yourself via `load_skill(skills=[...])`. Anything wrapped in `<specialized_knowledge>` is already loaded for you.
{% for category, skills in available_skills | dictsort -%}
{% for skill in skills -%}
@ -515,6 +515,17 @@ On-demand specialist skills. Spawn a specialist via `create_agent(skills=[...])`
</available_skills>
{% endif %}
{% if requested_skill_names %}
<cache_point>
<specialized_knowledge>
{% for skill_name in requested_skill_names %}
<{{ skill_name }}>
{{ get_skill(skill_name) }}
</{{ skill_name }}>
{% endfor %}
</specialized_knowledge>
{% endif %}
{% if include_scope %}
{% include "scope.jinja" %}
{% endif %}

View file

@ -312,11 +312,12 @@ def _reasoning_settings(effort: ReasoningEffort) -> ModelSettings:
def _prompt_cache_extra_args(model_name: str) -> dict[str, Any] | None:
"""LiteLLM ``cache_control_injection_points`` for Claude prompt caching.
System prompt + rolling last-message breakpoint everywhere; ``tool_config``
only on Bedrock Converse (the only route whose LiteLLM transform consumes
it — elsewhere it leaks onto the wire and native Anthropic 400s). Unmapped
Bedrock models get no points at all: Bedrock rejects the passed-through
field outright.
A breakpoint on each system message, plus a rolling last-message one. The
system prompt is split into up to three messages, which with the last
message uses all four breakpoints Claude allows. There is none on
``tool_config``: the tools come before the system prompt, so its first
breakpoint caches them too. Unmapped Bedrock models get no points at all:
Bedrock rejects the passed-through field outright.
The field is LiteLLM's own, consumed by its transform, so it only goes to
routes LiteLLM serves. A bare ``claude-...`` name is served by the SDK's
@ -328,11 +329,12 @@ def _prompt_cache_extra_args(model_name: str) -> dict[str, Any] | None:
if is_bedrock_route(model_name) and not bedrock_route_supports_prompt_caching(model_name):
return None
points: list[dict[str, Any]] = [{"location": "message", "role": "system"}]
if is_bedrock_route(model_name):
points.append({"location": "tool_config"})
points.append({"location": "message", "index": -1})
return {"cache_control_injection_points": points}
return {
"cache_control_injection_points": [
{"location": "message", "role": "system"},
{"location": "message", "index": -1},
]
}
def child_initial_input(

View file

@ -70,7 +70,6 @@ def _cache_points(model_name: str) -> Any:
def test_make_model_settings_enables_prompt_cache_for_bedrock_claude() -> None:
assert _cache_points("bedrock/global.anthropic.claude-opus-4-8") == [
{"location": "message", "role": "system"},
{"location": "tool_config"},
{"location": "message", "index": -1},
]

View file

@ -6,6 +6,7 @@ flow through to the root agent's ``build_strix_agent`` call.
from __future__ import annotations
import os
import types
from typing import Any
@ -279,6 +280,16 @@ def test_scope_is_rendered_once_at_the_end_of_the_prompt() -> None:
assert prompt.index("</available_skills>") < prompt.index("SYSTEM-VERIFIED SCOPE")
def test_requested_skills_follow_the_shared_prefix() -> None:
xss = render_system_prompt(skills=["xss"], include_scope=False)
sqli = render_system_prompt(skills=["sql_injection"], include_scope=False)
shared = os.path.commonprefix([xss, sqli])
assert "</available_skills>" in shared
assert shared.count("<cache_point>") == 1
assert "<xss>" in xss.split("<cache_point>")[1]
def test_scope_is_sent_as_its_own_system_message_on_cache_point_routes() -> None:
settings = make_model_settings(None, model_name="anthropic/claude-sonnet-5-5")
prompt = render_system_prompt(