fix(compact_20260112): include system prompt tokens in threshold check

The threshold check in Phase B previously counted only message tokens and
the compaction-block content, omitting the system prompt entirely. When
the system carried a prior compaction summary (via _augment_system_with_summary)
or was otherwise large, the threshold could fire later than intended,
allowing the conversation to exceed the model's context window before
compaction activated.

_count_effective_tokens now also counts the (augmented) system prompt
text. The caller passes compaction_block=None when augmented_system
already includes the prior summary, to avoid double-counting.

Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
Cursor Agent 2026-05-28 05:00:29 +00:00
parent bdddff0c36
commit bbc69ef9e7
No known key found for this signature in database

View file

@ -220,11 +220,15 @@ def _count_effective_tokens(
effective_messages: List[Dict[str, Any]],
compaction_block: Optional[Dict[str, Any]],
tools: Optional[List[Dict[str, Any]]],
system: Optional[Union[str, List[Dict[str, Any]]]] = None,
) -> int:
"""Token-count the conversation as it will appear downstream.
The compaction block (if any) becomes a system prefix on the downstream
call, so its content still counts even though it isn't in ``messages``.
The system prompt (which may already include a prior compaction summary
prepended via ``_augment_system_with_summary``) is also counted so the
threshold check matches the downstream ``input_tokens`` metric.
"""
# Local import to avoid pulling the adapter at module load time.
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
@ -274,9 +278,30 @@ def _count_effective_tokens(
content = compaction_block.get("content") or ""
if content:
total += litellm.token_counter(model=model, text=content)
system_text = _system_to_text(system)
if system_text:
total += litellm.token_counter(model=model, text=system_text)
return total
def _system_to_text(
system: Optional[Union[str, List[Dict[str, Any]]]],
) -> str:
"""Flatten an Anthropic-style ``system`` value into a single string for
token counting. Returns ``""`` when ``system`` carries no text."""
if system is None:
return ""
if isinstance(system, str):
return system
parts: List[str] = []
for block in system:
if isinstance(block, dict) and block.get("type") == "text":
text = block.get("text")
if isinstance(text, str) and text:
parts.append(text)
return "\n".join(parts)
def _select_last_user_question(
messages: List[Dict[str, Any]],
) -> List[Dict[str, Any]]:
@ -591,8 +616,12 @@ async def apply_compact_20260112(
current_tokens = _count_effective_tokens(
model=model,
effective_messages=effective_messages,
compaction_block=prior_compaction_block,
# ``augmented_system`` already carries the prior compaction summary
# (prepended via ``_augment_system_with_summary``); pass ``None``
# here so we don't double-count the summary text.
compaction_block=None,
tools=tools,
system=augmented_system,
)
except Exception as e:
verbose_logger.warning(