mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
Merge pull request #34589 from BerriAI/litellm_lit4798_glm_stop_thinking
fix(anthropic-adapter): translate stop_sequences and disabled thinking for non-Claude targets
This commit is contained in:
commit
32a4377acd
3 changed files with 135 additions and 39 deletions
|
|
@ -331,6 +331,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
"thinking",
|
||||
"output_format",
|
||||
"output_config",
|
||||
"stop_sequences",
|
||||
]
|
||||
|
||||
def _is_web_search_tool(self, tool: Dict[str, Any]) -> bool:
|
||||
|
|
@ -615,7 +616,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
thinking_type = thinking.get("type", "disabled")
|
||||
|
||||
if thinking_type == "disabled":
|
||||
return None
|
||||
return "none"
|
||||
elif thinking_type == "enabled":
|
||||
return reasoning_effort_from_thinking_budget(thinking.get("budget_tokens", 0))
|
||||
elif thinking_type == "adaptive":
|
||||
|
|
@ -683,25 +684,37 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
thinking
|
||||
)
|
||||
if reasoning_effort:
|
||||
summary = thinking.get("summary") if isinstance(thinking, dict) else None
|
||||
auto_summary = is_reasoning_auto_summary_enabled()
|
||||
if summary:
|
||||
return {
|
||||
"reasoning_effort": {
|
||||
"effort": reasoning_effort,
|
||||
"summary": summary,
|
||||
}
|
||||
}
|
||||
elif auto_summary:
|
||||
return {
|
||||
"reasoning_effort": {
|
||||
"effort": reasoning_effort,
|
||||
"summary": "detailed",
|
||||
}
|
||||
}
|
||||
return {"reasoning_effort": reasoning_effort}
|
||||
return {
|
||||
"reasoning_effort": LiteLLMAnthropicMessagesAdapter._apply_reasoning_summary_wrapping(
|
||||
reasoning_effort, thinking
|
||||
)
|
||||
}
|
||||
return {}
|
||||
|
||||
@staticmethod
|
||||
def _apply_reasoning_summary_wrapping(
|
||||
reasoning_effort: str,
|
||||
thinking: Dict[str, Any],
|
||||
) -> Any:
|
||||
"""
|
||||
Apply the reasoning_effort/summary wrapping rules shared by every
|
||||
thinking->reasoning_effort translation path.
|
||||
|
||||
Disabled thinking always stays a plain string - there's no reasoning
|
||||
trace to summarize, and non-Claude providers (e.g. Fireworks) expect
|
||||
reasoning_effort as a plain string, not a summary dict.
|
||||
"""
|
||||
thinking_type = thinking.get("type") if isinstance(thinking, dict) else None
|
||||
if thinking_type == "disabled":
|
||||
return reasoning_effort
|
||||
|
||||
summary = thinking.get("summary") if isinstance(thinking, dict) else None
|
||||
if summary:
|
||||
return {"effort": reasoning_effort, "summary": summary}
|
||||
if is_reasoning_auto_summary_enabled():
|
||||
return {"effort": reasoning_effort, "summary": "detailed"}
|
||||
return reasoning_effort
|
||||
|
||||
def translate_anthropic_tool_choice_to_openai(
|
||||
self, tool_choice: AnthropicMessagesToolChoice
|
||||
) -> ChatCompletionToolChoiceValues:
|
||||
|
|
@ -919,6 +932,18 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
tool_choice=cast(AnthropicMessagesToolChoice, tool_choice)
|
||||
)
|
||||
|
||||
def _translate_stop_sequences_to_openai(
|
||||
self,
|
||||
anthropic_message_request: AnthropicMessagesRequest,
|
||||
new_kwargs: ChatCompletionRequest,
|
||||
) -> None:
|
||||
if "stop_sequences" not in anthropic_message_request:
|
||||
return
|
||||
stop_sequences = anthropic_message_request["stop_sequences"]
|
||||
if not stop_sequences:
|
||||
return
|
||||
new_kwargs["stop"] = stop_sequences
|
||||
|
||||
def _translate_tools_to_openai(
|
||||
self,
|
||||
anthropic_message_request: AnthropicMessagesRequest,
|
||||
|
|
@ -976,32 +1001,17 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
if not reasoning_effort:
|
||||
return
|
||||
|
||||
thinking_type = thinking.get("type") if isinstance(thinking, dict) else None
|
||||
|
||||
# For adaptive thinking, override with output_config.effort if available
|
||||
if isinstance(thinking, dict) and thinking.get("type") == "adaptive":
|
||||
if thinking_type == "adaptive":
|
||||
output_config = anthropic_message_request.get("output_config")
|
||||
if isinstance(output_config, dict) and output_config.get("effort"):
|
||||
reasoning_effort = output_config["effort"]
|
||||
|
||||
summary = thinking.get("summary") if isinstance(thinking, dict) else None
|
||||
auto_summary = is_reasoning_auto_summary_enabled()
|
||||
if summary:
|
||||
new_kwargs["reasoning_effort"] = cast(
|
||||
Any,
|
||||
{
|
||||
"effort": reasoning_effort,
|
||||
"summary": summary,
|
||||
},
|
||||
)
|
||||
elif auto_summary:
|
||||
new_kwargs["reasoning_effort"] = cast(
|
||||
Any,
|
||||
{
|
||||
"effort": reasoning_effort,
|
||||
"summary": "detailed",
|
||||
},
|
||||
)
|
||||
else:
|
||||
new_kwargs["reasoning_effort"] = reasoning_effort
|
||||
new_kwargs["reasoning_effort"] = self._apply_reasoning_summary_wrapping(
|
||||
reasoning_effort, cast(Dict[str, Any], thinking)
|
||||
)
|
||||
|
||||
def _translate_output_format_to_openai(
|
||||
self,
|
||||
|
|
@ -1098,6 +1108,11 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
anthropic_message_request=anthropic_message_request,
|
||||
new_kwargs=new_kwargs,
|
||||
)
|
||||
## CONVERT STOP_SEQUENCES
|
||||
self._translate_stop_sequences_to_openai(
|
||||
anthropic_message_request=anthropic_message_request,
|
||||
new_kwargs=new_kwargs,
|
||||
)
|
||||
## CONVERT OUTPUT_FORMAT to RESPONSE_FORMAT
|
||||
self._translate_output_format_to_openai(
|
||||
anthropic_message_request=anthropic_message_request,
|
||||
|
|
|
|||
|
|
@ -1582,6 +1582,67 @@ def test_thinking_still_translated_to_reasoning_effort_for_non_claude_model():
|
|||
assert new_kwargs["reasoning_effort"] == "low"
|
||||
|
||||
|
||||
def test_thinking_disabled_translated_to_reasoning_effort_none_for_non_claude_model():
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
thinking = {"type": "disabled"}
|
||||
|
||||
new_kwargs = {"model": CACHE_CONTROL_NON_ANTHROPIC_MODEL}
|
||||
adapter._translate_thinking_to_openai(cast(Any, {"thinking": thinking}), cast(Any, new_kwargs))
|
||||
|
||||
assert "thinking" not in new_kwargs
|
||||
assert new_kwargs["reasoning_effort"] == "none"
|
||||
|
||||
|
||||
def test_thinking_disabled_stays_plain_string_when_auto_summary_enabled():
|
||||
import litellm
|
||||
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
thinking = {"type": "disabled"}
|
||||
|
||||
original = litellm.reasoning_auto_summary
|
||||
try:
|
||||
litellm.reasoning_auto_summary = True
|
||||
new_kwargs = {"model": CACHE_CONTROL_NON_ANTHROPIC_MODEL}
|
||||
adapter._translate_thinking_to_openai(cast(Any, {"thinking": thinking}), cast(Any, new_kwargs))
|
||||
finally:
|
||||
litellm.reasoning_auto_summary = original
|
||||
|
||||
assert new_kwargs["reasoning_effort"] == "none"
|
||||
|
||||
|
||||
def test_stop_sequences_translated_to_stop_for_non_claude_model():
|
||||
from litellm.types.llms.anthropic import AnthropicMessagesRequest
|
||||
|
||||
anthropic_request = AnthropicMessagesRequest(
|
||||
model=CACHE_CONTROL_NON_ANTHROPIC_MODEL,
|
||||
max_tokens=1024,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
stop_sequences=["</block>"],
|
||||
)
|
||||
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
openai_request, _ = adapter.translate_anthropic_to_openai(anthropic_message_request=anthropic_request)
|
||||
|
||||
assert openai_request["stop"] == ["</block>"]
|
||||
assert "stop_sequences" not in openai_request
|
||||
|
||||
|
||||
def test_empty_stop_sequences_does_not_set_stop():
|
||||
from litellm.types.llms.anthropic import AnthropicMessagesRequest
|
||||
|
||||
anthropic_request = AnthropicMessagesRequest(
|
||||
model=CACHE_CONTROL_NON_ANTHROPIC_MODEL,
|
||||
max_tokens=1024,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
stop_sequences=[],
|
||||
)
|
||||
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
openai_request, _ = adapter.translate_anthropic_to_openai(anthropic_message_request=anthropic_request)
|
||||
|
||||
assert "stop" not in openai_request
|
||||
|
||||
|
||||
def test_cache_control_preserved_in_image_content_for_claude():
|
||||
"""Cache control should be preserved in image content for Claude models."""
|
||||
anthropic_messages = [
|
||||
|
|
|
|||
|
|
@ -651,6 +651,26 @@ class TestThinkingSummaryPreservation:
|
|||
"reasoning_effort": {"effort": "high", "summary": "concise"}
|
||||
}
|
||||
|
||||
def test_translate_thinking_for_model_disabled_stays_plain_string_when_auto_summary_enabled(self):
|
||||
"""Disabled thinking must stay a plain string even when reasoning_auto_summary is on."""
|
||||
import litellm
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
|
||||
LiteLLMAnthropicMessagesAdapter,
|
||||
)
|
||||
|
||||
original = litellm.reasoning_auto_summary
|
||||
try:
|
||||
litellm.reasoning_auto_summary = True
|
||||
thinking = {"type": "disabled"}
|
||||
result = LiteLLMAnthropicMessagesAdapter.translate_thinking_for_model(
|
||||
thinking=thinking,
|
||||
model="openai/gpt-5.2",
|
||||
)
|
||||
finally:
|
||||
litellm.reasoning_auto_summary = original
|
||||
|
||||
assert result == {"reasoning_effort": "none"}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Parity tests: redundant empty-text-block sanitization scan removal.
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue