mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(compact_20260112): include system prompt tokens in threshold check
The threshold check in Phase B previously counted only message tokens and the compaction-block content, omitting the system prompt entirely. When the system carried a prior compaction summary (via _augment_system_with_summary) or was otherwise large, the threshold could fire later than intended, allowing the conversation to exceed the model's context window before compaction activated. _count_effective_tokens now also counts the (augmented) system prompt text. The caller passes compaction_block=None when augmented_system already includes the prior summary, to avoid double-counting. Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
parent
bdddff0c36
commit
bbc69ef9e7
1 changed files with 30 additions and 1 deletions
|
|
@ -220,11 +220,15 @@ def _count_effective_tokens(
|
|||
effective_messages: List[Dict[str, Any]],
|
||||
compaction_block: Optional[Dict[str, Any]],
|
||||
tools: Optional[List[Dict[str, Any]]],
|
||||
system: Optional[Union[str, List[Dict[str, Any]]]] = None,
|
||||
) -> int:
|
||||
"""Token-count the conversation as it will appear downstream.
|
||||
|
||||
The compaction block (if any) becomes a system prefix on the downstream
|
||||
call, so its content still counts even though it isn't in ``messages``.
|
||||
The system prompt (which may already include a prior compaction summary
|
||||
prepended via ``_augment_system_with_summary``) is also counted so the
|
||||
threshold check matches the downstream ``input_tokens`` metric.
|
||||
"""
|
||||
# Local import to avoid pulling the adapter at module load time.
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
|
||||
|
|
@ -274,9 +278,30 @@ def _count_effective_tokens(
|
|||
content = compaction_block.get("content") or ""
|
||||
if content:
|
||||
total += litellm.token_counter(model=model, text=content)
|
||||
system_text = _system_to_text(system)
|
||||
if system_text:
|
||||
total += litellm.token_counter(model=model, text=system_text)
|
||||
return total
|
||||
|
||||
|
||||
def _system_to_text(
|
||||
system: Optional[Union[str, List[Dict[str, Any]]]],
|
||||
) -> str:
|
||||
"""Flatten an Anthropic-style ``system`` value into a single string for
|
||||
token counting. Returns ``""`` when ``system`` carries no text."""
|
||||
if system is None:
|
||||
return ""
|
||||
if isinstance(system, str):
|
||||
return system
|
||||
parts: List[str] = []
|
||||
for block in system:
|
||||
if isinstance(block, dict) and block.get("type") == "text":
|
||||
text = block.get("text")
|
||||
if isinstance(text, str) and text:
|
||||
parts.append(text)
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
def _select_last_user_question(
|
||||
messages: List[Dict[str, Any]],
|
||||
) -> List[Dict[str, Any]]:
|
||||
|
|
@ -591,8 +616,12 @@ async def apply_compact_20260112(
|
|||
current_tokens = _count_effective_tokens(
|
||||
model=model,
|
||||
effective_messages=effective_messages,
|
||||
compaction_block=prior_compaction_block,
|
||||
# ``augmented_system`` already carries the prior compaction summary
|
||||
# (prepended via ``_augment_system_with_summary``); pass ``None``
|
||||
# here so we don't double-count the summary text.
|
||||
compaction_block=None,
|
||||
tools=tools,
|
||||
system=augmented_system,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue