mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
fix(bedrock): forward output_config.effort for adaptive-thinking Claude
Claude Sonnet 4.6 / Opus 4.6 / Opus 4.7 expose Anthropic's effort parameter through 'output_config.effort' (Messages API and Bedrock Invoke) and through 'additionalModelRequestFields.thinking.effort' (Bedrock Converse). LiteLLM was stripping output_config on every Bedrock route, so Sonnet 4.6 callers saw the effort never reach the wire body. This fix: - Bedrock Invoke /v1/messages: preserve output_config for adaptive-thinking models (Claude 4.6/4.7) and for Opus 4.5 with the effort-2025-11-24 beta header (auto-attached via the existing is_effort_used path). - Bedrock Invoke /completion: same — stop stripping output_config for the supported model set. - Bedrock Converse: fold output_config.effort into thinking.effort and switch reasoning_effort to populate thinking.effort on adaptive-thinking models. output_config itself is still removed before the request signs, matching Bedrock's wire shape. - Map effort-2025-11-24 to itself for 'bedrock' in the beta-headers JSON so the auto-attach for Opus 4.5 isn't dropped on the way out. - Add adaptive-thinking and Opus-4.5-with-beta tests on all three Bedrock routes; haiku/sonnet-4 stays in the strip path. Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
8300657af9
commit
f52cbae60c
8 changed files with 354 additions and 8 deletions
|
|
@ -103,7 +103,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": null,
|
||||
"effort-2025-11-24": null,
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": null,
|
||||
"files-api-2025-04-14": null,
|
||||
"fine-grained-tool-streaming-2025-05-14": null,
|
||||
|
|
|
|||
|
|
@ -32,6 +32,7 @@ from litellm.litellm_core_utils.prompt_templates.factory import (
|
|||
_bedrock_tools_pt,
|
||||
)
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException
|
||||
from litellm.types.llms.bedrock import *
|
||||
from litellm.types.llms.openai import (
|
||||
|
|
@ -452,6 +453,70 @@ class AmazonConverseConfig(BaseConfig):
|
|||
optional_params["thinking"] = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort=reasoning_effort, model=model
|
||||
)
|
||||
# Claude 4.6/4.7 adaptive thinking takes ``effort`` inside the
|
||||
# ``thinking`` block on Bedrock Converse (Anthropic's
|
||||
# ``output_config.effort`` is API-only). Surface the mapped
|
||||
# effort here so it survives into ``additionalModelRequestFields``.
|
||||
if (
|
||||
AnthropicModelInfo._is_adaptive_thinking_model(model)
|
||||
and isinstance(optional_params.get("thinking"), dict)
|
||||
and optional_params["thinking"].get("type") == "adaptive"
|
||||
and "effort" not in optional_params["thinking"]
|
||||
):
|
||||
effort_map = {
|
||||
"low": "low",
|
||||
"minimal": "low",
|
||||
"medium": "medium",
|
||||
"high": "high",
|
||||
"xhigh": "xhigh",
|
||||
"max": "max",
|
||||
}
|
||||
optional_params["thinking"]["effort"] = effort_map.get(
|
||||
reasoning_effort, reasoning_effort
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _fold_output_config_effort_into_thinking(
|
||||
inference_params: dict, model: str
|
||||
) -> None:
|
||||
"""
|
||||
Fold ``output_config.effort`` into ``thinking.effort`` for Bedrock
|
||||
Converse on adaptive-thinking models (Claude 4.6 / 4.7).
|
||||
|
||||
Bedrock Converse exposes Anthropic's ``effort`` parameter inside
|
||||
``additionalModelRequestFields.thinking`` (per the AWS adaptive
|
||||
thinking docs), not as a top-level ``output_config`` block. Callers
|
||||
commonly send the Anthropic Messages API shape — ``output_config:
|
||||
{effort: ...}`` — when chaining a Messages-style request through
|
||||
Converse, which leaves the value on the floor.
|
||||
|
||||
This helper preserves caller intent: if the model accepts adaptive
|
||||
thinking and the caller did not already set ``thinking.effort``, we
|
||||
attach the effort there. ``output_config`` is still stripped from
|
||||
``inference_params`` separately so it never reaches the wire body.
|
||||
"""
|
||||
if not AnthropicModelInfo._is_adaptive_thinking_model(model):
|
||||
return
|
||||
output_config = inference_params.get("output_config")
|
||||
if not isinstance(output_config, dict):
|
||||
return
|
||||
effort = output_config.get("effort")
|
||||
if not (effort and isinstance(effort, str)):
|
||||
return
|
||||
thinking = inference_params.get("thinking")
|
||||
if isinstance(thinking, dict):
|
||||
# Don't override an explicit thinking.effort set by the caller.
|
||||
if "effort" not in thinking:
|
||||
thinking["effort"] = effort
|
||||
# Make sure the type is adaptive — only adaptive thinking accepts
|
||||
# effort. If the caller passed ``enabled``, leave it alone; the
|
||||
# downstream `_translate_legacy_thinking_for_adaptive_model` flow
|
||||
# owns that translation.
|
||||
else:
|
||||
inference_params["thinking"] = {
|
||||
"type": "adaptive",
|
||||
"effort": effort,
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _clamp_thinking_budget_tokens(optional_params: dict) -> None:
|
||||
|
|
@ -1192,6 +1257,17 @@ class AmazonConverseConfig(BaseConfig):
|
|||
+ supported_config_params
|
||||
)
|
||||
inference_params.pop("json_mode", None) # used for handling json_schema
|
||||
|
||||
# Bedrock Converse exposes the Anthropic ``effort`` parameter through
|
||||
# ``additionalModelRequestFields.thinking.effort`` (per AWS adaptive
|
||||
# thinking docs). If the caller supplied ``output_config.effort`` —
|
||||
# the Anthropic Messages API shape — fold it into ``thinking`` for
|
||||
# Claude 4.6/4.7 adaptive-thinking models so the value reaches the
|
||||
# wire body. For all other models the field is unsupported and gets
|
||||
# stripped below.
|
||||
self._fold_output_config_effort_into_thinking(
|
||||
inference_params=inference_params, model=model
|
||||
)
|
||||
# Anthropic-only key. Bedrock expects `outputConfig` (camelCase) and
|
||||
# will reject `output_config` if it leaks through pass-through routes.
|
||||
inference_params.pop("output_config", None)
|
||||
|
|
@ -1204,9 +1280,6 @@ class AmazonConverseConfig(BaseConfig):
|
|||
output_config: Optional[OutputConfigBlock] = inference_params.pop(
|
||||
"outputConfig", None
|
||||
)
|
||||
inference_params.pop(
|
||||
"output_config", None
|
||||
) # Bedrock Converse doesn't support it
|
||||
|
||||
# keep supported params in 'inference_params', and set all model-specific params in 'additional_request_params'
|
||||
additional_request_params = {
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import (
|
|||
convert_url_to_base64,
|
||||
)
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation import (
|
||||
AmazonInvokeConfig,
|
||||
)
|
||||
|
|
@ -31,6 +32,21 @@ else:
|
|||
LiteLLMLoggingObj = Any
|
||||
|
||||
|
||||
def _supports_effort_on_bedrock_invoke(model: str) -> bool:
|
||||
"""
|
||||
Bedrock Invoke (legacy /completion path) accepts ``output_config.effort``
|
||||
on Claude 4.6/4.7 adaptive-thinking models (no beta header) and on Claude
|
||||
Opus 4.5 with the ``effort-2025-11-24`` beta header. All other Claude
|
||||
models reject the field.
|
||||
"""
|
||||
if AnthropicModelInfo._is_adaptive_thinking_model(model):
|
||||
return True
|
||||
model_lower = model.lower()
|
||||
return any(
|
||||
p in model_lower for p in ("opus-4.5", "opus_4.5", "opus-4-5", "opus_4_5")
|
||||
)
|
||||
|
||||
|
||||
class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
|
||||
"""
|
||||
Reference:
|
||||
|
|
@ -169,7 +185,12 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
|
|||
anthropic_request.pop("model", None)
|
||||
anthropic_request.pop("stream", None)
|
||||
anthropic_request.pop("output_format", None)
|
||||
anthropic_request.pop("output_config", None)
|
||||
# Bedrock accepts ``output_config`` (carrying ``effort``) on Claude 4.6/4.7
|
||||
# adaptive-thinking models natively, and on Claude Opus 4.5 with the
|
||||
# ``effort-2025-11-24`` beta header. Older Claude models reject the
|
||||
# field — strip it for them so the request signs cleanly.
|
||||
if not _supports_effort_on_bedrock_invoke(model):
|
||||
anthropic_request.pop("output_config", None)
|
||||
if "anthropic_version" not in anthropic_request:
|
||||
anthropic_request["anthropic_version"] = self.anthropic_version
|
||||
|
||||
|
|
|
|||
|
|
@ -290,6 +290,19 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
)
|
||||
return True
|
||||
|
||||
def _supports_effort_on_bedrock(self, model: str) -> bool:
|
||||
"""
|
||||
Whether Bedrock accepts ``output_config.effort`` for this model.
|
||||
|
||||
Adaptive-thinking models (Claude 4.6 / 4.7) take ``effort`` natively
|
||||
with no beta header. Claude Opus 4.5 takes ``effort`` only when the
|
||||
``effort-2025-11-24`` beta header is attached. Older Claude models
|
||||
reject the field outright.
|
||||
"""
|
||||
return AnthropicModelInfo._is_adaptive_thinking_model(
|
||||
model
|
||||
) or self._is_claude_opus_4_5(model)
|
||||
|
||||
def _is_claude_opus_4_5(self, model: str) -> bool:
|
||||
"""
|
||||
Check if the model is Claude Opus 4.5.
|
||||
|
|
@ -511,6 +524,16 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
anthropic_messages_request=anthropic_messages_request,
|
||||
)
|
||||
|
||||
# 5b. ``output_config`` (carries the Anthropic ``effort`` parameter)
|
||||
# is accepted by Bedrock on:
|
||||
# - Claude 4.6/4.7 adaptive-thinking models (no beta header needed)
|
||||
# - Claude Opus 4.5 with the ``effort-2025-11-24`` beta header
|
||||
# Other Claude models on Bedrock reject the field, so strip it for
|
||||
# them. The allowlist (step 7 below) only preserves keys that survive
|
||||
# this filter.
|
||||
if not self._supports_effort_on_bedrock(model):
|
||||
anthropic_messages_request.pop("output_config", None)
|
||||
|
||||
# 5a. Remove `custom` field from tools (Bedrock doesn't support it)
|
||||
# Claude Code sends `custom: {defer_loading: true}` on tool definitions,
|
||||
# which causes Bedrock to reject the request with "Extra inputs are not permitted"
|
||||
|
|
|
|||
|
|
@ -1041,3 +1041,12 @@ class BedrockInvokeAnthropicMessagesRequest(TypedDict, total=False):
|
|||
# `metadata` is part of the common Anthropic Messages API shape.
|
||||
thinking: dict
|
||||
metadata: dict
|
||||
|
||||
# `output_config` carries the Anthropic ``effort`` parameter. Bedrock
|
||||
# accepts it on the Invoke Messages API for Claude 4.6+ adaptive thinking
|
||||
# (no beta header) and for Opus 4.5 with the ``effort-2025-11-24`` beta.
|
||||
# Older Claude models on Bedrock reject this field, so the runtime strips
|
||||
# it for them — see ``AmazonAnthropicClaudeMessagesConfig`` for that
|
||||
# model-aware filtering. Including it here lets the allowlist preserve it
|
||||
# for the supported-model path.
|
||||
output_config: dict
|
||||
|
|
|
|||
|
|
@ -440,6 +440,51 @@ def test_output_config_removed_from_bedrock_chat_invoke_request():
|
|||
assert result["max_tokens"] == 100
|
||||
|
||||
|
||||
def test_output_config_preserved_for_adaptive_thinking_on_bedrock_invoke():
|
||||
"""
|
||||
Bedrock Invoke (legacy /completion path) must forward ``output_config``
|
||||
for Claude 4.6/4.7 adaptive-thinking models — the field reaches the wire
|
||||
body. This mirrors the /v1/messages behavior for the same models.
|
||||
"""
|
||||
config = AmazonAnthropicClaudeConfig()
|
||||
messages = [{"role": "user", "content": "test"}]
|
||||
optional_params = {
|
||||
"max_tokens": 100,
|
||||
"output_config": {"effort": "high"},
|
||||
}
|
||||
|
||||
result = config.transform_request(
|
||||
model="anthropic.claude-sonnet-4-6-v1:0",
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert result.get("output_config") == {"effort": "high"}
|
||||
|
||||
|
||||
def test_output_config_preserved_for_opus_4_5_on_bedrock_invoke():
|
||||
"""
|
||||
Bedrock Invoke /completion forwards ``output_config`` on Claude Opus 4.5
|
||||
(gated behind the ``effort-2025-11-24`` beta header on Bedrock).
|
||||
"""
|
||||
config = AmazonAnthropicClaudeConfig()
|
||||
messages = [{"role": "user", "content": "test"}]
|
||||
optional_params = {
|
||||
"max_tokens": 100,
|
||||
"output_config": {"effort": "low"},
|
||||
}
|
||||
|
||||
result = config.transform_request(
|
||||
model="anthropic.claude-opus-4-5-20251101-v1:0",
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert result.get("output_config") == {"effort": "low"}
|
||||
|
||||
|
||||
def test_output_format_removed_from_bedrock_invoke_request():
|
||||
"""
|
||||
Test that output_format parameter is removed from Bedrock Invoke requests.
|
||||
|
|
|
|||
|
|
@ -3284,6 +3284,98 @@ def test_transform_request_with_output_config():
|
|||
)
|
||||
|
||||
|
||||
def test_converse_folds_output_config_effort_into_thinking_for_46():
|
||||
"""
|
||||
Bedrock Converse exposes the Anthropic ``effort`` parameter through
|
||||
``additionalModelRequestFields.thinking.effort`` (per AWS adaptive thinking
|
||||
docs), not as a top-level ``output_config`` block. The transform must
|
||||
fold ``output_config.effort`` into ``thinking`` for adaptive-thinking
|
||||
models so the value survives onto the wire body.
|
||||
|
||||
Customer regression: Sonnet 4.6 ``output_config.effort`` was being dropped
|
||||
on Bedrock Converse and never reached the request payload.
|
||||
"""
|
||||
config = AmazonConverseConfig()
|
||||
messages = [{"role": "user", "content": "hello"}]
|
||||
|
||||
result = config._transform_request(
|
||||
model="us.anthropic.claude-sonnet-4-6-v1:0",
|
||||
messages=messages,
|
||||
optional_params={
|
||||
"maxTokens": 64,
|
||||
"output_config": {"effort": "low"},
|
||||
},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert "outputConfig" not in result
|
||||
additional_fields = result.get("additionalModelRequestFields", {})
|
||||
assert "output_config" not in additional_fields
|
||||
thinking = additional_fields.get("thinking")
|
||||
assert isinstance(thinking, dict)
|
||||
assert thinking.get("type") == "adaptive"
|
||||
assert thinking.get("effort") == "low"
|
||||
|
||||
|
||||
def test_converse_folds_output_config_effort_preserves_existing_thinking():
|
||||
"""
|
||||
When the caller already supplied a ``thinking`` block, fold ``effort``
|
||||
into it without overriding any explicit ``thinking.effort``.
|
||||
"""
|
||||
config = AmazonConverseConfig()
|
||||
messages = [{"role": "user", "content": "hello"}]
|
||||
|
||||
# Caller already set thinking with no effort — fold into it.
|
||||
result = config._transform_request(
|
||||
model="us.anthropic.claude-opus-4-7-v1",
|
||||
messages=messages,
|
||||
optional_params={
|
||||
"maxTokens": 64,
|
||||
"thinking": {"type": "adaptive"},
|
||||
"output_config": {"effort": "high"},
|
||||
},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
thinking = result["additionalModelRequestFields"].get("thinking")
|
||||
assert thinking == {"type": "adaptive", "effort": "high"}
|
||||
|
||||
# Caller set thinking.effort explicitly — must not be overridden.
|
||||
result = config._transform_request(
|
||||
model="us.anthropic.claude-opus-4-7-v1",
|
||||
messages=messages,
|
||||
optional_params={
|
||||
"maxTokens": 64,
|
||||
"thinking": {"type": "adaptive", "effort": "max"},
|
||||
"output_config": {"effort": "low"},
|
||||
},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
thinking = result["additionalModelRequestFields"].get("thinking")
|
||||
assert thinking.get("effort") == "max"
|
||||
|
||||
|
||||
def test_converse_reasoning_effort_sets_thinking_effort_for_46():
|
||||
"""
|
||||
On adaptive-thinking models, OpenAI ``reasoning_effort`` should produce
|
||||
``thinking.type=adaptive`` plus ``thinking.effort`` so Converse forwards
|
||||
the value through ``additionalModelRequestFields``.
|
||||
"""
|
||||
config = AmazonConverseConfig()
|
||||
optional_params: dict = {}
|
||||
config._handle_reasoning_effort_parameter(
|
||||
model="us.anthropic.claude-sonnet-4-6-v1:0",
|
||||
reasoning_effort="medium",
|
||||
optional_params=optional_params,
|
||||
)
|
||||
thinking = optional_params.get("thinking")
|
||||
assert isinstance(thinking, dict)
|
||||
assert thinking.get("type") == "adaptive"
|
||||
assert thinking.get("effort") == "medium"
|
||||
|
||||
|
||||
def test_transform_request_strips_anthropic_output_config():
|
||||
"""
|
||||
output_config is Anthropic-specific and must never be forwarded to Bedrock.
|
||||
|
|
|
|||
|
|
@ -625,6 +625,79 @@ def test_bedrock_messages_strips_output_config():
|
|||
assert result.get("max_tokens") == 4096
|
||||
|
||||
|
||||
def test_bedrock_messages_preserves_output_config_for_adaptive_thinking_models():
|
||||
"""
|
||||
Bedrock Invoke accepts ``output_config.effort`` natively on Claude 4.6/4.7
|
||||
adaptive-thinking models (no beta header required). The transform must
|
||||
forward the field instead of stripping it.
|
||||
|
||||
Customer regression: Sonnet 4.6 ``output_config.effort`` was being dropped
|
||||
on Bedrock Invoke and never appeared in the wire body.
|
||||
"""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
|
||||
|
||||
for model in (
|
||||
"anthropic.claude-sonnet-4-6-v1:0",
|
||||
"global.anthropic.claude-sonnet-4-6-v1:0",
|
||||
"anthropic.claude-opus-4-6-v1",
|
||||
"us.anthropic.claude-opus-4-7-v1",
|
||||
):
|
||||
optional_params = {
|
||||
"max_tokens": 4096,
|
||||
"output_config": {"effort": "low"},
|
||||
}
|
||||
result = cfg.transform_anthropic_messages_request(
|
||||
model=model,
|
||||
messages=messages,
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
assert result.get("output_config") == {
|
||||
"effort": "low"
|
||||
}, f"output_config should be forwarded for {model}, got {result!r}"
|
||||
# Adaptive-thinking models don't need the effort beta header.
|
||||
betas = result.get("anthropic_beta") or []
|
||||
assert "effort-2025-11-24" not in betas
|
||||
|
||||
|
||||
def test_bedrock_messages_preserves_output_config_for_opus_4_5_with_beta(
|
||||
monkeypatch,
|
||||
):
|
||||
"""
|
||||
Bedrock Invoke accepts ``output_config.effort`` on Claude Opus 4.5 only
|
||||
when the ``effort-2025-11-24`` beta header is attached. The transform must
|
||||
forward the field and auto-attach that beta.
|
||||
"""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
import litellm.anthropic_beta_headers_manager as beta_mgr
|
||||
|
||||
# Force local beta-header config — the remote fetch may lag the in-tree
|
||||
# mapping update that allows ``effort-2025-11-24`` for ``bedrock``.
|
||||
monkeypatch.setenv("LITELLM_LOCAL_ANTHROPIC_BETA_HEADERS", "True")
|
||||
monkeypatch.setattr(beta_mgr, "_BETA_HEADERS_CONFIG", None)
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
|
||||
optional_params = {
|
||||
"max_tokens": 4096,
|
||||
"output_config": {"effort": "medium"},
|
||||
}
|
||||
result = cfg.transform_anthropic_messages_request(
|
||||
model="anthropic.claude-opus-4-5-20251101-v1:0",
|
||||
messages=messages,
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
assert result.get("output_config") == {"effort": "medium"}
|
||||
betas = result.get("anthropic_beta") or []
|
||||
assert "effort-2025-11-24" in betas
|
||||
|
||||
|
||||
def test_bedrock_messages_strips_output_config_with_output_format():
|
||||
"""
|
||||
When both output_config and output_format are present, both should be
|
||||
|
|
@ -882,7 +955,10 @@ async def test_promote_message_start_cache_when_message_stop_omits_cache_fields(
|
|||
"delta": {"stop_reason": "end_turn", "stop_sequence": None},
|
||||
"usage": {"input_tokens": 10, "output_tokens": 181},
|
||||
}
|
||||
yield {"type": "message_stop", "usage": {"input_tokens": 10, "output_tokens": 181}}
|
||||
yield {
|
||||
"type": "message_stop",
|
||||
"usage": {"input_tokens": 10, "output_tokens": 181},
|
||||
}
|
||||
|
||||
merged: list[dict] = []
|
||||
async for chunk in cfg._promote_message_stop_usage(_stream()):
|
||||
|
|
@ -936,7 +1012,11 @@ async def test_unified_bedrock_messages_cache_on_start_only_never_negative_cost(
|
|||
},
|
||||
},
|
||||
}
|
||||
yield {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}
|
||||
yield {
|
||||
"type": "content_block_start",
|
||||
"index": 0,
|
||||
"content_block": {"type": "text", "text": ""},
|
||||
}
|
||||
yield {
|
||||
"type": "content_block_delta",
|
||||
"index": 0,
|
||||
|
|
@ -948,7 +1028,10 @@ async def test_unified_bedrock_messages_cache_on_start_only_never_negative_cost(
|
|||
"delta": {"stop_reason": "end_turn", "stop_sequence": None},
|
||||
"usage": {"output_tokens": 181, "input_tokens": 10},
|
||||
}
|
||||
yield {"type": "message_stop", "usage": {"input_tokens": 10, "output_tokens": 181}}
|
||||
yield {
|
||||
"type": "message_stop",
|
||||
"usage": {"input_tokens": 10, "output_tokens": 181},
|
||||
}
|
||||
|
||||
logging_obj = LiteLLMLoggingObj(
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue