feat(context_management): configurable summary max_tokens; surface ignored knobs

- compact_20260112: read summary max_tokens from general_settings
  (context_management_summary_max_tokens) so operators can fit the
  chosen summary model's output budget; falls back to the compiled
  default for missing or invalid values.

- clear_tool_uses_20250919: log unsupported knobs at warning level
  (was debug, which silently dropped misconfiguration) and surface
  them as warnings on the AppliedEdit so clients see what was ignored.
This commit is contained in:
mateo-berri 2026-05-28 07:44:06 +00:00
parent 745fd65bf5
commit 32d9937bab
No known key found for this signature in database
5 changed files with 124 additions and 11 deletions

View file

@ -14,8 +14,10 @@ COMPACT_MIN_TRIGGER_TOKENS = 50_000
# Default ``max_tokens`` for the summary call. Required by providers like
# Anthropic that reject requests without it; safely accepted by providers that
# don't strictly require it. Chosen to comfortably fit a long structured
# summary.
# summary. Operators can override via
# ``general_settings.context_management_summary_max_tokens``.
COMPACT_SUMMARY_MAX_TOKENS = 4096
COMPACT_SUMMARY_MAX_TOKENS_SETTING_KEY = "context_management_summary_max_tokens"
COMPACT_SUMMARY_MODEL_SETTING_KEY = "context_management_summary_model"
COMPACT_SUMMARY_SYSTEM_PREFIX = "Previous conversation summary: "

View file

@ -148,14 +148,18 @@ def apply_clear_tool_uses_20250919(
edit_spec: Dict[str, Any],
) -> Tuple[List[Dict[str, Any]], Optional[AppliedEdit]]:
"""Apply clear_tool_uses; return (messages, AppliedEdit or None)."""
for ignored_knob in ("clear_at_least", "exclude_tools", "clear_tool_inputs"):
if ignored_knob in edit_spec:
verbose_logger.debug(
"context_management polyfill: ignoring '%s' on %s "
"(supported only on Anthropic-family forwarding path in v0)",
ignored_knob,
CLEAR_TOOL_USES_EDIT_TYPE,
)
ignored_knobs = [
knob
for knob in ("clear_at_least", "exclude_tools", "clear_tool_inputs")
if knob in edit_spec
]
for ignored_knob in ignored_knobs:
verbose_logger.warning(
"context_management polyfill: ignoring '%s' on %s "
"(supported only on Anthropic-family forwarding path in v0)",
ignored_knob,
CLEAR_TOOL_USES_EDIT_TYPE,
)
trigger = edit_spec.get("trigger") or {
"type": "input_tokens",
@ -201,4 +205,6 @@ def apply_clear_tool_uses_20250919(
"cleared_tool_uses": cleared_count,
"cleared_input_tokens": cleared_input_tokens,
}
if ignored_knobs:
applied["warnings"] = [f"{knob}_ignored" for knob in ignored_knobs]
return edited, applied

View file

@ -30,6 +30,7 @@ from ..constants import (
COMPACT_MIN_TRIGGER_TOKENS,
COMPACT_NO_TOOL_CALLS_SUFFIX,
COMPACT_SUMMARY_MAX_TOKENS,
COMPACT_SUMMARY_MAX_TOKENS_SETTING_KEY,
COMPACT_SUMMARY_MODEL_SETTING_KEY,
COMPACT_SUMMARY_SYSTEM_PREFIX,
)
@ -65,6 +66,23 @@ def _read_summary_model_setting() -> Optional[str]:
return value if isinstance(value, str) and value else None
def _read_summary_max_tokens_setting() -> int:
"""Look up the configured summary ``max_tokens`` from proxy general_settings.
Falls back to :data:`COMPACT_SUMMARY_MAX_TOKENS` when the setting is
missing or invalid (non-positive int, wrong type). Operators tune this
when the default doesn't fit their chosen summary model's output budget.
"""
try:
from litellm.proxy.proxy_server import general_settings
except Exception:
return COMPACT_SUMMARY_MAX_TOKENS
value = general_settings.get(COMPACT_SUMMARY_MAX_TOKENS_SETTING_KEY)
if isinstance(value, int) and value > 0:
return value
return COMPACT_SUMMARY_MAX_TOKENS
def _check_summary_model_access(
user_api_key_auth: Any,
summary_model: str,
@ -526,6 +544,7 @@ async def _call_summary_model(
metadata: Dict[str, Any],
llm_router: Any,
allowed_model_region: Optional[str] = None,
max_tokens: int = COMPACT_SUMMARY_MAX_TOKENS,
) -> Any:
"""Invoke the configured summary model.
@ -536,7 +555,10 @@ async def _call_summary_model(
# ``max_tokens`` is required by providers like Anthropic and silently
# accepted by providers that don't strictly require it (OpenAI etc.).
# Setting a sensible default here means the feature works regardless of
# which model an admin configures as ``context_management_summary_model``.
# which model an admin configures as ``context_management_summary_model``;
# operators can override via ``context_management_summary_max_tokens`` in
# ``general_settings`` when the default doesn't fit the chosen model's
# output budget.
# The propagated proxy auth/spend-attribution fields (``user_api_key`` etc.)
# must travel as ``litellm_metadata`` — that is the parameter the proxy's
# post-call spend hooks read for budget attribution. The provider-level
@ -550,7 +572,7 @@ async def _call_summary_model(
call_kwargs: Dict[str, Any] = {
"model": summary_model,
"messages": summary_messages,
"max_tokens": COMPACT_SUMMARY_MAX_TOKENS,
"max_tokens": max_tokens,
"litellm_metadata": metadata,
}
if allowed_model_region is not None:
@ -771,6 +793,7 @@ async def apply_compact_20260112( # noqa: PLR0915
metadata=propagated_metadata,
llm_router=llm_router,
allowed_model_region=allowed_model_region,
max_tokens=_read_summary_max_tokens_setting(),
)
except Exception as e:
verbose_logger.warning("compact_20260112: summary call failed: %s", e)

View file

@ -224,6 +224,32 @@ def test_ignored_knobs_do_not_alter_behavior():
# Despite clear_tool_inputs=True, inputs are NOT cleared (knob ignored).
assert applied is not None
assert applied["cleared_tool_uses"] == 2
# Ignored knobs surface as warnings on the AppliedEdit so operators can
# see what was dropped (the v0 polyfill silently dropping them at debug
# log level made misconfiguration invisible from the response).
assert set(applied.get("warnings", [])) == {
"clear_at_least_ignored",
"exclude_tools_ignored",
"clear_tool_inputs_ignored",
}
def test_no_ignored_knobs_omits_warnings_field():
"""When the caller doesn't pass any unsupported knobs, no ``warnings`` are added."""
messages = _make_history(n_pairs=3)
_, applied = apply_clear_tool_uses_20250919(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec={
"type": "clear_tool_uses_20250919",
"trigger": {"type": "tool_uses", "value": 0},
"keep": {"type": "tool_uses", "value": 1},
},
)
assert applied is not None
assert "warnings" not in applied
def test_tool_result_list_content_shape_preserved():

View file

@ -1049,6 +1049,62 @@ async def test_summary_call_sends_default_max_tokens():
assert captured_kwargs.get("max_tokens") == COMPACT_SUMMARY_MAX_TOKENS
async def test_summary_call_honors_max_tokens_override():
"""Operators can override the default summary ``max_tokens`` via
``general_settings.context_management_summary_max_tokens``."""
from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import (
_read_summary_max_tokens_setting,
)
captured_kwargs: dict = {}
class _FakeRouter:
async def acompletion(self, **kwargs):
captured_kwargs.update(kwargs)
return _make_mock_response("<summary>x</summary>")
with patch(
"litellm.proxy.proxy_server.general_settings",
{"context_management_summary_max_tokens": 8192},
):
assert _read_summary_max_tokens_setting() == 8192
from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import (
_call_summary_model,
)
await _call_summary_model(
summary_model="claude-haiku-4-5",
summary_messages=[{"role": "user", "content": "hi"}],
metadata={},
llm_router=_FakeRouter(),
max_tokens=_read_summary_max_tokens_setting(),
)
assert captured_kwargs.get("max_tokens") == 8192
def test_summary_max_tokens_setting_falls_back_for_invalid_values():
"""Invalid override values (non-int, non-positive, missing) fall back to
the compiled default so a typo in ``general_settings`` doesn't break the
summary call."""
from litellm.llms.anthropic.experimental_pass_through.context_management.constants import (
COMPACT_SUMMARY_MAX_TOKENS,
)
from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import (
_read_summary_max_tokens_setting,
)
for bad in ("4096", 0, -1, None, {"value": 1024}):
with patch(
"litellm.proxy.proxy_server.general_settings",
{"context_management_summary_max_tokens": bad},
):
assert (
_read_summary_max_tokens_setting() == COMPACT_SUMMARY_MAX_TOKENS
), f"expected default for invalid override {bad!r}"
# ---------------------------------------------------------------------------
# Editor: summary model key/team access gate
# ---------------------------------------------------------------------------