fix(anthropic-adapter): translate stop_sequences and disabled thinking for non-Claude targets

Claude Code's auto-mode classifier sends stop_sequences and thinking:
{type: disabled} on /v1/messages. The Anthropic adapter passed
stop_sequences through unchanged instead of mapping it to OpenAI's stop,
which Fireworks' OpenAI-compatible endpoint rejects with HTTP 400. It also
dropped disabled thinking instead of mapping it to reasoning_effort: none,
so the model spent its output budget on reasoning it was told to skip.

Resolves LIT-4798
This commit is contained in:
Tin Chi Lo 2026-07-24 18:07:00 -07:00
parent fc5ab31fba
commit b3e27a0bc3
2 changed files with 47 additions and 1 deletions

View file

@ -331,6 +331,7 @@ class LiteLLMAnthropicMessagesAdapter:
"thinking",
"output_format",
"output_config",
"stop_sequences",
]
def _is_web_search_tool(self, tool: Dict[str, Any]) -> bool:
@ -615,7 +616,7 @@ class LiteLLMAnthropicMessagesAdapter:
thinking_type = thinking.get("type", "disabled")
if thinking_type == "disabled":
return None
return "none"
elif thinking_type == "enabled":
return reasoning_effort_from_thinking_budget(thinking.get("budget_tokens", 0))
elif thinking_type == "adaptive":
@ -919,6 +920,18 @@ class LiteLLMAnthropicMessagesAdapter:
tool_choice=cast(AnthropicMessagesToolChoice, tool_choice)
)
def _translate_stop_sequences_to_openai(
self,
anthropic_message_request: AnthropicMessagesRequest,
new_kwargs: ChatCompletionRequest,
) -> None:
if "stop_sequences" not in anthropic_message_request:
return
stop_sequences = anthropic_message_request["stop_sequences"]
if not stop_sequences:
return
new_kwargs["stop"] = stop_sequences
def _translate_tools_to_openai(
self,
anthropic_message_request: AnthropicMessagesRequest,
@ -1098,6 +1111,11 @@ class LiteLLMAnthropicMessagesAdapter:
anthropic_message_request=anthropic_message_request,
new_kwargs=new_kwargs,
)
## CONVERT STOP_SEQUENCES
self._translate_stop_sequences_to_openai(
anthropic_message_request=anthropic_message_request,
new_kwargs=new_kwargs,
)
## CONVERT OUTPUT_FORMAT to RESPONSE_FORMAT
self._translate_output_format_to_openai(
anthropic_message_request=anthropic_message_request,

View file

@ -1582,6 +1582,34 @@ def test_thinking_still_translated_to_reasoning_effort_for_non_claude_model():
assert new_kwargs["reasoning_effort"] == "low"
def test_thinking_disabled_translated_to_reasoning_effort_none_for_non_claude_model():
adapter = LiteLLMAnthropicMessagesAdapter()
thinking = {"type": "disabled"}
new_kwargs = {"model": CACHE_CONTROL_NON_ANTHROPIC_MODEL}
adapter._translate_thinking_to_openai(cast(Any, {"thinking": thinking}), cast(Any, new_kwargs))
assert "thinking" not in new_kwargs
assert new_kwargs["reasoning_effort"] == "none"
def test_stop_sequences_translated_to_stop_for_non_claude_model():
from litellm.types.llms.anthropic import AnthropicMessagesRequest
anthropic_request = AnthropicMessagesRequest(
model=CACHE_CONTROL_NON_ANTHROPIC_MODEL,
max_tokens=1024,
messages=[{"role": "user", "content": "hi"}],
stop_sequences=["</block>"],
)
adapter = LiteLLMAnthropicMessagesAdapter()
openai_request, _ = adapter.translate_anthropic_to_openai(anthropic_message_request=anthropic_request)
assert openai_request["stop"] == ["</block>"]
assert "stop_sequences" not in openai_request
def test_cache_control_preserved_in_image_content_for_claude():
"""Cache control should be preserved in image content for Claude models."""
anthropic_messages = [