mirror of
https://github.com/usestrix/strix.git
synced 2026-10-02 02:13:43 +00:00
perf(prompt): load requested skills after a cache point (#1382)
Siblings differ only in the skills they were spawned with, but those came first in <specialized_knowledge>, so their prompts diverged at 39%. Shared skills and the catalog now come first, and the requested skills follow a cache point, so siblings share 93%. The extra system message takes a fourth Claude breakpoint, so the Bedrock tool_config one goes: the first system breakpoint already covers the tools. Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
e66c56c473
commit
954bc0d527
5 changed files with 51 additions and 15 deletions
|
|
@ -102,6 +102,16 @@ def render_system_prompt(
|
|||
),
|
||||
)
|
||||
|
||||
shared = {
|
||||
name.split("/")[-1]
|
||||
for name in _resolve_skills(
|
||||
requested=None,
|
||||
scan_mode=scan_mode,
|
||||
is_whitebox=is_whitebox,
|
||||
is_root=is_root,
|
||||
is_diff_scoped=is_diff_scoped,
|
||||
)
|
||||
}
|
||||
skills_to_load = _resolve_skills(
|
||||
requested=skills,
|
||||
scan_mode=scan_mode,
|
||||
|
|
@ -112,8 +122,11 @@ def render_system_prompt(
|
|||
skill_content = load_skills(skills_to_load)
|
||||
env.globals["get_skill"] = lambda name: skill_content.get(name, "")
|
||||
|
||||
# Skills every agent of this kind loads come first, so siblings share them
|
||||
# as a cached prefix; the ones the caller asked for vary and go after.
|
||||
rendered = env.get_template("system_prompt.jinja").render(
|
||||
loaded_skill_names=list(skill_content.keys()),
|
||||
shared_skill_names=[name for name in skill_content if name in shared],
|
||||
requested_skill_names=[name for name in skill_content if name not in shared],
|
||||
available_skills=get_available_skills(),
|
||||
interactive=interactive,
|
||||
is_root=is_root,
|
||||
|
|
|
|||
|
|
@ -493,9 +493,9 @@ Directories:
|
|||
Default user: pentester (sudo available)
|
||||
</environment>
|
||||
|
||||
{% if loaded_skill_names %}
|
||||
{% if shared_skill_names %}
|
||||
<specialized_knowledge>
|
||||
{% for skill_name in loaded_skill_names %}
|
||||
{% for skill_name in shared_skill_names %}
|
||||
<{{ skill_name }}>
|
||||
{{ get_skill(skill_name) }}
|
||||
</{{ skill_name }}>
|
||||
|
|
@ -505,7 +505,7 @@ Default user: pentester (sudo available)
|
|||
|
||||
{% if available_skills %}
|
||||
<available_skills>
|
||||
On-demand specialist skills. Spawn a specialist via `create_agent(skills=[...])`, or pull guidance inline for yourself via `load_skill(skills=[...])`. Anything wrapped in `<specialized_knowledge>` above is already loaded for you.
|
||||
On-demand specialist skills. Spawn a specialist via `create_agent(skills=[...])`, or pull guidance inline for yourself via `load_skill(skills=[...])`. Anything wrapped in `<specialized_knowledge>` is already loaded for you.
|
||||
|
||||
{% for category, skills in available_skills | dictsort -%}
|
||||
{% for skill in skills -%}
|
||||
|
|
@ -515,6 +515,17 @@ On-demand specialist skills. Spawn a specialist via `create_agent(skills=[...])`
|
|||
</available_skills>
|
||||
{% endif %}
|
||||
|
||||
{% if requested_skill_names %}
|
||||
<cache_point>
|
||||
<specialized_knowledge>
|
||||
{% for skill_name in requested_skill_names %}
|
||||
<{{ skill_name }}>
|
||||
{{ get_skill(skill_name) }}
|
||||
</{{ skill_name }}>
|
||||
{% endfor %}
|
||||
</specialized_knowledge>
|
||||
{% endif %}
|
||||
|
||||
{% if include_scope %}
|
||||
{% include "scope.jinja" %}
|
||||
{% endif %}
|
||||
|
|
|
|||
|
|
@ -312,11 +312,12 @@ def _reasoning_settings(effort: ReasoningEffort) -> ModelSettings:
|
|||
def _prompt_cache_extra_args(model_name: str) -> dict[str, Any] | None:
|
||||
"""LiteLLM ``cache_control_injection_points`` for Claude prompt caching.
|
||||
|
||||
System prompt + rolling last-message breakpoint everywhere; ``tool_config``
|
||||
only on Bedrock Converse (the only route whose LiteLLM transform consumes
|
||||
it — elsewhere it leaks onto the wire and native Anthropic 400s). Unmapped
|
||||
Bedrock models get no points at all: Bedrock rejects the passed-through
|
||||
field outright.
|
||||
A breakpoint on each system message, plus a rolling last-message one. The
|
||||
system prompt is split into up to three messages, which with the last
|
||||
message uses all four breakpoints Claude allows. There is none on
|
||||
``tool_config``: the tools come before the system prompt, so its first
|
||||
breakpoint caches them too. Unmapped Bedrock models get no points at all:
|
||||
Bedrock rejects the passed-through field outright.
|
||||
|
||||
The field is LiteLLM's own, consumed by its transform, so it only goes to
|
||||
routes LiteLLM serves. A bare ``claude-...`` name is served by the SDK's
|
||||
|
|
@ -328,11 +329,12 @@ def _prompt_cache_extra_args(model_name: str) -> dict[str, Any] | None:
|
|||
if is_bedrock_route(model_name) and not bedrock_route_supports_prompt_caching(model_name):
|
||||
return None
|
||||
|
||||
points: list[dict[str, Any]] = [{"location": "message", "role": "system"}]
|
||||
if is_bedrock_route(model_name):
|
||||
points.append({"location": "tool_config"})
|
||||
points.append({"location": "message", "index": -1})
|
||||
return {"cache_control_injection_points": points}
|
||||
return {
|
||||
"cache_control_injection_points": [
|
||||
{"location": "message", "role": "system"},
|
||||
{"location": "message", "index": -1},
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def child_initial_input(
|
||||
|
|
|
|||
|
|
@ -70,7 +70,6 @@ def _cache_points(model_name: str) -> Any:
|
|||
def test_make_model_settings_enables_prompt_cache_for_bedrock_claude() -> None:
|
||||
assert _cache_points("bedrock/global.anthropic.claude-opus-4-8") == [
|
||||
{"location": "message", "role": "system"},
|
||||
{"location": "tool_config"},
|
||||
{"location": "message", "index": -1},
|
||||
]
|
||||
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ flow through to the root agent's ``build_strix_agent`` call.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import types
|
||||
from typing import Any
|
||||
|
||||
|
|
@ -279,6 +280,16 @@ def test_scope_is_rendered_once_at_the_end_of_the_prompt() -> None:
|
|||
assert prompt.index("</available_skills>") < prompt.index("SYSTEM-VERIFIED SCOPE")
|
||||
|
||||
|
||||
def test_requested_skills_follow_the_shared_prefix() -> None:
|
||||
xss = render_system_prompt(skills=["xss"], include_scope=False)
|
||||
sqli = render_system_prompt(skills=["sql_injection"], include_scope=False)
|
||||
|
||||
shared = os.path.commonprefix([xss, sqli])
|
||||
assert "</available_skills>" in shared
|
||||
assert shared.count("<cache_point>") == 1
|
||||
assert "<xss>" in xss.split("<cache_point>")[1]
|
||||
|
||||
|
||||
def test_scope_is_sent_as_its_own_system_message_on_cache_point_routes() -> None:
|
||||
settings = make_model_settings(None, model_name="anthropic/claude-sonnet-5-5")
|
||||
prompt = render_system_prompt(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue