π
LiteLLM
@@ -253,3 +264,6 @@ html_form = f"""
"""
+
+
+html_form = build_ui_login_form(show_deprecation_banner=True)
diff --git a/litellm/proxy/hooks/dynamic_rate_limiter_v3.py b/litellm/proxy/hooks/dynamic_rate_limiter_v3.py
index 53419ef6ad7..755f5fdc201 100644
--- a/litellm/proxy/hooks/dynamic_rate_limiter_v3.py
+++ b/litellm/proxy/hooks/dynamic_rate_limiter_v3.py
@@ -58,6 +58,35 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
def update_variables(self, llm_router: Router):
self.llm_router = llm_router
+ def _get_saturation_check_cache_ttl(self) -> int:
+ """Get the configurable TTL for local cache when reading saturation values."""
+ return litellm.priority_reservation_settings.saturation_check_cache_ttl
+
+ async def _get_saturation_value_from_cache(
+ self,
+ counter_key: str,
+ ) -> Optional[str]:
+ """
+ Get saturation value with configurable local cache TTL.
+
+ Uses DualCache with configurable TTL for local cache storage.
+ TTL is configurable via litellm.priority_reservation_settings.saturation_check_cache_ttl
+
+ Args:
+ counter_key: The cache key for the saturation counter
+
+ Returns:
+ Counter value as string, or None if not found
+ """
+ local_cache_ttl = self._get_saturation_check_cache_ttl()
+
+ return await self.internal_usage_cache.async_get_cache(
+ key=counter_key,
+ litellm_parent_otel_span=None,
+ local_only=False,
+ ttl=local_cache_ttl,
+ )
+
def _get_priority_weight(
self, priority: Optional[str], model_info: Optional[ModelGroupInfo] = None
) -> float:
@@ -195,7 +224,7 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
try:
max_saturation = 0.0
- # Query RPM saturation
+ # Query RPM saturation - always read from Redis for multi-node consistency
if model_group_info.rpm is not None and model_group_info.rpm > 0:
# Use v3 limiter's key format: {key:value}:rate_limit_type
counter_key = self.v3_limiter.create_rate_limit_keys(
@@ -204,11 +233,9 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
rate_limit_type="requests",
)
- # Query cache for current counter value
- counter_value = await self.internal_usage_cache.async_get_cache(
- key=counter_key,
- litellm_parent_otel_span=None,
- local_only=False, # Check Redis too
+ # Query Redis directly for current counter value (skip local cache for consistency)
+ counter_value = await self._get_saturation_value_from_cache(
+ counter_key=counter_key
)
if counter_value is not None:
@@ -229,10 +256,8 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
rate_limit_type="tokens",
)
- counter_value = await self.internal_usage_cache.async_get_cache(
- key=counter_key,
- litellm_parent_otel_span=None,
- local_only=False,
+ counter_value = await self._get_saturation_value_from_cache(
+ counter_key=counter_key
)
if counter_value is not None:
diff --git a/litellm/proxy/management_endpoints/mcp_management_endpoints.py b/litellm/proxy/management_endpoints/mcp_management_endpoints.py
index 5acfbf2cc79..f0eddcc8683 100644
--- a/litellm/proxy/management_endpoints/mcp_management_endpoints.py
+++ b/litellm/proxy/management_endpoints/mcp_management_endpoints.py
@@ -63,6 +63,9 @@ if MCP_AVAILABLE:
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
global_mcp_server_manager,
)
+ from litellm.proxy._experimental.mcp_server.ui_session_utils import (
+ build_effective_auth_contexts,
+ )
from litellm.proxy.common_utils.http_parsing_utils import _read_request_body
from litellm.proxy._types import (
LiteLLM_MCPServerTable,
@@ -422,13 +425,18 @@ if MCP_AVAILABLE:
```
"""
- # Use server manager to get all servers with health and team data
- mcp_servers = (
- await global_mcp_server_manager.get_all_mcp_servers_with_health_and_teams(
- user_api_key_auth=user_api_key_dict
+ auth_contexts = await build_effective_auth_contexts(user_api_key_dict)
+
+ aggregated_servers: Dict[str, LiteLLM_MCPServerTable] = {}
+ for auth_context in auth_contexts:
+ servers = await global_mcp_server_manager.get_all_mcp_servers_with_health_and_teams(
+ user_api_key_auth=auth_context
)
- )
- redacted_mcp_servers = _redact_mcp_credentials_list(mcp_servers)
+ for server in servers:
+ if server.server_id not in aggregated_servers:
+ aggregated_servers[server.server_id] = server
+
+ redacted_mcp_servers = _redact_mcp_credentials_list(aggregated_servers.values())
# augment the mcp servers with public status
if litellm.public_mcp_servers is not None:
diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py
index 9b6e22b8196..b990f4ca6e9 100644
--- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py
+++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py
@@ -96,9 +96,20 @@ class AnthropicPassthroughLoggingHandler:
handles streaming and non-streaming responses
"""
try:
+ # Get custom_llm_provider from logging object if available (e.g., azure_ai for Azure Anthropic)
+ custom_llm_provider = logging_obj.model_call_details.get(
+ "custom_llm_provider"
+ )
+
+ # Prepend custom_llm_provider to model if not already present
+ model_for_cost = model
+ if custom_llm_provider and not model.startswith(f"{custom_llm_provider}/"):
+ model_for_cost = f"{custom_llm_provider}/{model}"
+
response_cost = litellm.completion_cost(
completion_response=litellm_model_response,
- model=model,
+ model=model_for_cost,
+ custom_llm_provider=custom_llm_provider,
)
kwargs["response_cost"] = response_cost
@@ -157,19 +168,14 @@ class AnthropicPassthroughLoggingHandler:
"""
model = request_body.get("model", "")
- # Dheck if it's available in the logging object
+ # Check if it's available in the logging object
if (
not model
and hasattr(litellm_logging_obj, "model_call_details")
and litellm_logging_obj.model_call_details.get("model")
):
model = cast(str, litellm_logging_obj.model_call_details.get("model"))
- custom_llm_provider = litellm_logging_obj.model_call_details.get(
- "custom_llm_provider"
- )
- if custom_llm_provider and not model.startswith(custom_llm_provider):
- model = f"{custom_llm_provider}/{model}"
complete_streaming_response = (
AnthropicPassthroughLoggingHandler._build_complete_streaming_response(
all_chunks=all_chunks,
diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py
index 5d097ac4408..6aef4ab493b 100644
--- a/litellm/proxy/proxy_server.py
+++ b/litellm/proxy/proxy_server.py
@@ -236,7 +236,7 @@ from litellm.proxy.common_utils.encrypt_decrypt_utils import (
decrypt_value_helper,
encrypt_value_helper,
)
-from litellm.proxy.common_utils.html_forms.ui_login import html_form
+from litellm.proxy.common_utils.html_forms.ui_login import build_ui_login_form
from litellm.proxy.common_utils.http_parsing_utils import (
_read_request_body,
check_file_size_under_limit,
@@ -8306,11 +8306,15 @@ async def fallback_login(request: Request):
# Use UI Credentials set in .env
from fastapi.responses import HTMLResponse
- return HTMLResponse(content=html_form, status_code=200)
+ return HTMLResponse(
+ content=build_ui_login_form(show_deprecation_banner=False), status_code=200
+ )
else:
from fastapi.responses import HTMLResponse
- return HTMLResponse(content=html_form, status_code=200)
+ return HTMLResponse(
+ content=build_ui_login_form(show_deprecation_banner=False), status_code=200
+ )
@router.post(
diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json
index ddd41ca0b1d..629760a7dd2 100644
--- a/litellm/proxy/public_endpoints/provider_create_fields.json
+++ b/litellm/proxy/public_endpoints/provider_create_fields.json
@@ -2446,6 +2446,24 @@
],
"default_model_placeholder": "gpt-3.5-turbo"
},
+ {
+ "provider": "SAP",
+ "provider_display_name": "SAP Generative AI Hub",
+ "litellm_provider": "sap",
+ "credential_fields": [
+ {
+ "key": "api_key",
+ "label": "SAP AI Core Service Key (JSON)",
+ "placeholder": null,
+ "tooltip": "Paste your SAP AI Core service key JSON. Contains clientid, clientsecret, and service URLs.",
+ "required": true,
+ "field_type": "textarea",
+ "options": null,
+ "default_value": null
+ }
+ ],
+ "default_model_placeholder": "sap/gpt-4"
+ },
{
"provider": "Snowflake",
"provider_display_name": "Snowflake",
diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py
index ec9daebbf70..09f3af20e7c 100644
--- a/litellm/proxy/utils.py
+++ b/litellm/proxy/utils.py
@@ -995,7 +995,7 @@ class ProxyLogging:
):
result = await self._process_guardrail_callback(
callback=_callback,
- data=data,
+ data=data, # type: ignore
user_api_key_dict=user_api_key_dict,
call_type=call_type,
)
diff --git a/litellm/proxy/vertex_ai_endpoints/langfuse_endpoints.py b/litellm/proxy/vertex_ai_endpoints/langfuse_endpoints.py
index 684e2ad0617..ce27c830f6f 100644
--- a/litellm/proxy/vertex_ai_endpoints/langfuse_endpoints.py
+++ b/litellm/proxy/vertex_ai_endpoints/langfuse_endpoints.py
@@ -128,12 +128,12 @@ async def langfuse_proxy_route(
endpoint=endpoint,
target=str(updated_url),
custom_headers={"Authorization": langfuse_combined_key},
+ query_params=dict(request.query_params), # type: ignore
) # dynamically construct pass-through endpoint based on incoming path
received_value = await endpoint_func(
request,
fastapi_response,
user_api_key_dict,
- query_params=dict(request.query_params), # type: ignore
)
return received_value
diff --git a/litellm/router_utils/common_utils.py b/litellm/router_utils/common_utils.py
index 15725e30d04..10acc343abd 100644
--- a/litellm/router_utils/common_utils.py
+++ b/litellm/router_utils/common_utils.py
@@ -110,7 +110,7 @@ def filter_web_search_deployments(
return healthy_deployments
is_web_search_request = False
- tools = request_kwargs.get("tools", [])
+ tools = request_kwargs.get("tools") or []
for tool in tools:
# These are the two websearch tools for OpenAI / Azure.
if tool.get("type") == "web_search" or tool.get("type") == "web_search_preview":
diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py
index df6580f9b3a..23dd661e9ad 100644
--- a/litellm/types/llms/anthropic.py
+++ b/litellm/types/llms/anthropic.py
@@ -1,7 +1,7 @@
from enum import Enum
from typing import Any, Dict, Iterable, List, Optional, Union
-from pydantic import BaseModel
+from pydantic import BaseModel, ConfigDict
from typing_extensions import Literal, Required, TypedDict
from .openai import (
@@ -535,8 +535,7 @@ class AnthropicResponseContentBlockToolUse(BaseModel):
input: dict
provider_specific_fields: Optional[Dict[str, Any]] = None
- class Config:
- extra = "allow" # Allow provider_specific_fields
+ model_config = ConfigDict(extra="allow") # Allow provider_specific_fields
class AnthropicResponseContentBlockThinking(BaseModel):
diff --git a/litellm/types/rag.py b/litellm/types/rag.py
index 7a964931af1..dd724ca217a 100644
--- a/litellm/types/rag.py
+++ b/litellm/types/rag.py
@@ -4,7 +4,7 @@ Type definitions for RAG (Retrieval Augmented Generation) Ingest API.
from typing import Any, Dict, List, Literal, Optional, Union
-from pydantic import BaseModel
+from pydantic import BaseModel, ConfigDict
from typing_extensions import TypedDict
@@ -185,6 +185,5 @@ class RAGIngestRequest(BaseModel):
file_id: Optional[str] = None # Existing file ID
ingest_options: Dict[str, Any] # RAGIngestOptions as dict for flexibility
- class Config:
- extra = "allow" # Allow additional fields
+ model_config = ConfigDict(extra="allow") # Allow additional fields
diff --git a/litellm/types/utils.py b/litellm/types/utils.py
index 5821ae3d233..2ccc14a2719 100644
--- a/litellm/types/utils.py
+++ b/litellm/types/utils.py
@@ -2982,6 +2982,7 @@ class LlmProviders(str, Enum):
LANGFUSE = "langfuse"
HUMANLOOP = "humanloop"
TOPAZ = "topaz"
+ SAP_GENERATIVE_AI_HUB = "sap"
ASSEMBLYAI = "assemblyai"
GITHUB_COPILOT = "github_copilot"
SNOWFLAKE = "snowflake"
@@ -2989,6 +2990,7 @@ class LlmProviders(str, Enum):
LLAMA = "meta_llama"
NSCALE = "nscale"
PG_VECTOR = "pg_vector"
+ HELICONE = "helicone"
HYPERBOLIC = "hyperbolic"
RECRAFT = "recraft"
FAL_AI = "fal_ai"
@@ -3308,4 +3310,9 @@ class PriorityReservationSettings(BaseModel):
description="Saturation threshold (0.0-1.0) at which strict priority enforcement begins. Below this threshold, generous mode allows priority borrowing. Above this threshold, strict mode enforces normalized priority limits.",
)
+ saturation_check_cache_ttl: int = Field(
+ default=60,
+ description="TTL in seconds for local cache when reading saturation check values from Redis.",
+ )
+
model_config = ConfigDict(protected_namespaces=())
diff --git a/litellm/utils.py b/litellm/utils.py
index d58eb28a061..9279703af1a 100644
--- a/litellm/utils.py
+++ b/litellm/utils.py
@@ -2886,6 +2886,21 @@ def get_optional_params_embeddings( # noqa: PLR0915
model=model,
drop_params=drop_params if drop_params is not None else False,
)
+ final_params = {**optional_params, **kwargs}
+ return final_params
+ elif custom_llm_provider == "sap":
+ supported_params = get_supported_openai_params(
+ model=model,
+ custom_llm_provider="sap",
+ request_type="embeddings",
+ )
+ _check_valid_arg(supported_params=supported_params)
+ optional_params = litellm.GenAIHubEmbeddingConfig().map_openai_params(
+ non_default_params=non_default_params,
+ optional_params={},
+ model=model,
+ drop_params=drop_params if drop_params is not None else False
+ )
elif custom_llm_provider == "infinity":
supported_params = get_supported_openai_params(
model=model,
@@ -2899,6 +2914,10 @@ def get_optional_params_embeddings( # noqa: PLR0915
model=model,
drop_params=drop_params if drop_params is not None else False,
)
+
+ final_params = {**optional_params, **kwargs}
+ return final_params
+
elif custom_llm_provider == "fireworks_ai":
supported_params = get_supported_openai_params(
model=model,
@@ -7216,6 +7235,8 @@ class ProviderConfigManager:
return litellm.TritonConfig()
elif litellm.LlmProviders.PETALS == provider:
return litellm.PetalsConfig()
+ elif litellm.LlmProviders.SAP_GENERATIVE_AI_HUB == provider:
+ return litellm.GenAIHubOrchestrationConfig()
elif litellm.LlmProviders.FEATHERLESS_AI == provider:
return litellm.FeatherlessAIConfig()
elif litellm.LlmProviders.NOVITA == provider:
@@ -7276,6 +7297,8 @@ class ProviderConfigManager:
return litellm.TritonEmbeddingConfig()
elif litellm.LlmProviders.WATSONX == provider:
return litellm.IBMWatsonXEmbeddingConfig()
+ elif litellm.LlmProviders.SAP_GENERATIVE_AI_HUB == provider:
+ return litellm.GenAIHubEmbeddingConfig()
elif litellm.LlmProviders.INFINITY == provider:
return litellm.InfinityEmbeddingConfig()
elif litellm.LlmProviders.SAMBANOVA == provider:
@@ -7343,7 +7366,11 @@ class ProviderConfigManager:
elif litellm.LlmProviders.DEEPINFRA == provider:
return litellm.DeepinfraRerankConfig()
elif litellm.LlmProviders.NVIDIA_NIM == provider:
- return litellm.NvidiaNimRerankConfig()
+ from litellm.llms.nvidia_nim.rerank.common_utils import (
+ get_nvidia_nim_rerank_config,
+ )
+
+ return get_nvidia_nim_rerank_config(model)
elif litellm.LlmProviders.VERTEX_AI == provider:
return litellm.VertexAIRerankConfig()
elif litellm.LlmProviders.FIREWORKS_AI == provider:
@@ -7364,12 +7391,19 @@ class ProviderConfigManager:
return BedrockModelInfo.get_bedrock_provider_config_for_messages_api(model)
elif litellm.LlmProviders.VERTEX_AI == provider:
- if "claude" in model:
+ if "claude" in model.lower():
from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.experimental_pass_through.transformation import (
VertexAIPartnerModelsAnthropicMessagesConfig,
)
return VertexAIPartnerModelsAnthropicMessagesConfig()
+ elif litellm.LlmProviders.AZURE_AI == provider:
+ if "claude" in model.lower():
+ from litellm.llms.azure_ai.anthropic.messages_transformation import (
+ AzureAnthropicMessagesConfig,
+ )
+
+ return AzureAnthropicMessagesConfig()
return None
@staticmethod
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index fde60a92370..549c3d60018 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -255,6 +255,50 @@
"mode": "image_generation",
"output_cost_per_image": 0.06
},
+ "us.writer.palmyra-x4-v1:0": {
+ "input_cost_per_token": 2.5e-06,
+ "litellm_provider": "bedrock_converse",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 8192,
+ "max_tokens": 8192,
+ "mode": "chat",
+ "output_cost_per_token": 1e-05,
+ "supports_function_calling": true,
+ "supports_pdf_input": true
+ },
+ "us.writer.palmyra-x5-v1:0": {
+ "input_cost_per_token": 6e-07,
+ "litellm_provider": "bedrock_converse",
+ "max_input_tokens": 1000000,
+ "max_output_tokens": 8192,
+ "max_tokens": 8192,
+ "mode": "chat",
+ "output_cost_per_token": 6e-06,
+ "supports_function_calling": true,
+ "supports_pdf_input": true
+ },
+ "writer.palmyra-x4-v1:0": {
+ "input_cost_per_token": 2.5e-06,
+ "litellm_provider": "bedrock_converse",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 8192,
+ "max_tokens": 8192,
+ "mode": "chat",
+ "output_cost_per_token": 1e-05,
+ "supports_function_calling": true,
+ "supports_pdf_input": true
+ },
+ "writer.palmyra-x5-v1:0": {
+ "input_cost_per_token": 6e-07,
+ "litellm_provider": "bedrock_converse",
+ "max_input_tokens": 1000000,
+ "max_output_tokens": 8192,
+ "max_tokens": 8192,
+ "mode": "chat",
+ "output_cost_per_token": 6e-06,
+ "supports_function_calling": true,
+ "supports_pdf_input": true
+ },
"amazon.nova-lite-v1:0": {
"input_cost_per_token": 6e-08,
"litellm_provider": "bedrock_converse",
@@ -6206,6 +6250,19 @@
"supports_function_calling": true,
"supports_tool_choice": true
},
+ "cerebras/zai-glm-4.6": {
+ "input_cost_per_token": 2.25e-06,
+ "litellm_provider": "cerebras",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "max_tokens": 128000,
+ "mode": "chat",
+ "output_cost_per_token": 2.75e-06,
+ "source": "https://www.cerebras.ai/pricing",
+ "supports_function_calling": true,
+ "supports_reasoning": true,
+ "supports_tool_choice": true
+ },
"chat-bison": {
"input_cost_per_character": 2.5e-07,
"input_cost_per_token": 1.25e-07,
@@ -22865,6 +22922,13 @@
"mode": "rerank",
"output_cost_per_token": 0.0
},
+ "nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2": {
+ "input_cost_per_query": 0.0,
+ "input_cost_per_token": 0.0,
+ "litellm_provider": "nvidia_nim",
+ "mode": "rerank",
+ "output_cost_per_token": 0.0
+ },
"sagemaker/meta-textgeneration-llama-2-13b": {
"input_cost_per_token": 0.0,
"litellm_provider": "sagemaker",
@@ -28089,5 +28153,2049 @@
"metadata": {
"comment": "Estimated cost based on standard TTS pricing. RunwayML uses ElevenLabs models."
}
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-coder-480b-a35b-instruct": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 4.5e-07,
+ "output_cost_per_token": 1.8e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/flux-kontext-pro": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 4e-08,
+ "output_cost_per_token": 4e-08,
+ "litellm_provider": "fireworks_ai",
+ "mode": "image_generation"
+ },
+ "fireworks_ai/accounts/fireworks/models/SSD-1B": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1.3e-10,
+ "output_cost_per_token": 1.3e-10,
+ "litellm_provider": "fireworks_ai",
+ "mode": "image_generation"
+ },
+ "fireworks_ai/accounts/fireworks/models/chronos-hermes-13b-v2": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-13b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-13b-instruct": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-13b-python": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-34b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-34b-instruct": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-34b-python": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-70b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-70b-instruct": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-70b-python": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-7b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-7b-instruct": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-llama-7b-python": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/code-qwen-1p5-7b": {
+ "max_tokens": 65536,
+ "max_input_tokens": 65536,
+ "max_output_tokens": 65536,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/codegemma-2b": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/codegemma-7b": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/cogito-671b-v2-p1": {
+ "max_tokens": 163840,
+ "max_input_tokens": 163840,
+ "max_output_tokens": 163840,
+ "input_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-3b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-70b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-8b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-qwen-14b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-qwen-32b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/flux-kontext-max": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 8e-08,
+ "output_cost_per_token": 8e-08,
+ "litellm_provider": "fireworks_ai",
+ "mode": "image_generation"
+ },
+ "fireworks_ai/accounts/fireworks/models/dbrx-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-coder-1b-base": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-coder-33b-instruct": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-base": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-base-v1p5": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-instruct-v1p5": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-lite-base": {
+ "max_tokens": 163840,
+ "max_input_tokens": 163840,
+ "max_output_tokens": 163840,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-lite-instruct": {
+ "max_tokens": 163840,
+ "max_input_tokens": 163840,
+ "max_output_tokens": 163840,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-prover-v2": {
+ "max_tokens": 163840,
+ "max_input_tokens": 163840,
+ "max_output_tokens": 163840,
+ "input_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-r1-0528-distill-qwen3-8b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-llama-70b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-llama-8b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-14b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-1p5b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-32b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-7b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-v2-lite-chat": {
+ "max_tokens": 163840,
+ "max_input_tokens": 163840,
+ "max_output_tokens": 163840,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/deepseek-v2p5": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/devstral-small-2505": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/dobby-mini-unhinged-plus-llama-3-1-8b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/dobby-unhinged-llama-3-3-70b-new": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/dolphin-2-9-2-qwen2-72b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/dolphin-2p6-mixtral-8x7b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/ernie-4p5-21b-a3b-pt": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/ernie-4p5-300b-a47b-pt": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/fare-20b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/firefunction-v1": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/firellava-13b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/firesearch-ocr-v6": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/fireworks-asr-large": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "audio_transcription"
+ },
+ "fireworks_ai/accounts/fireworks/models/fireworks-asr-v2": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "audio_transcription"
+ },
+ "fireworks_ai/accounts/fireworks/models/flux-1-dev": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/flux-1-dev-controlnet-union": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-09,
+ "output_cost_per_token": 1e-09,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/flux-1-dev-fp8": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 5e-10,
+ "output_cost_per_token": 5e-10,
+ "litellm_provider": "fireworks_ai",
+ "mode": "image_generation"
+ },
+ "fireworks_ai/accounts/fireworks/models/flux-1-schnell": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/flux-1-schnell-fp8": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 3.5e-10,
+ "output_cost_per_token": 3.5e-10,
+ "litellm_provider": "fireworks_ai",
+ "mode": "image_generation"
+ },
+ "fireworks_ai/accounts/fireworks/models/gemma-2b-it": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/gemma-3-27b-it": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/gemma-7b": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/gemma-7b-it": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/gemma2-9b-it": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/glm-4p5v": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/gpt-oss-safeguard-120b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/gpt-oss-safeguard-20b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/hermes-2-pro-mistral-7b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/internvl3-38b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/internvl3-78b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/internvl3-8b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/japanese-stable-diffusion-xl": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1.3e-10,
+ "output_cost_per_token": 1.3e-10,
+ "litellm_provider": "fireworks_ai",
+ "mode": "image_generation"
+ },
+ "fireworks_ai/accounts/fireworks/models/kat-coder": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/kat-dev-32b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/kat-dev-72b-exp": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-guard-2-8b": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-guard-3-1b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-guard-3-8b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v2-13b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v2-13b-chat": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v2-70b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v2-70b-chat": {
+ "max_tokens": 2048,
+ "max_input_tokens": 2048,
+ "max_output_tokens": 2048,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v2-7b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v2-7b-chat": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3-70b-instruct": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3-70b-instruct-hf": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3-8b": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3-8b-instruct-hf": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3p1-405b-instruct-long": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3p1-70b-instruct": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3p1-70b-instruct-1b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3p1-nemotron-70b-instruct": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3p2-1b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3p2-3b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llamaguard-7b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/llava-yi-34b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/minimax-m1-80k": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/minimax-m2": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 3e-07,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/ministral-3-14b-instruct-2512": {
+ "max_tokens": 256000,
+ "max_input_tokens": 256000,
+ "max_output_tokens": 256000,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/ministral-3-3b-instruct-2512": {
+ "max_tokens": 256000,
+ "max_input_tokens": 256000,
+ "max_output_tokens": 256000,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/ministral-3-8b-instruct-2512": {
+ "max_tokens": 256000,
+ "max_input_tokens": 256000,
+ "max_output_tokens": 256000,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mistral-7b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-4k": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-v0p2": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-v3": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mistral-7b-v0p2": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mistral-large-3-fp8": {
+ "max_tokens": 256000,
+ "max_input_tokens": 256000,
+ "max_output_tokens": 256000,
+ "input_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mistral-nemo-base-2407": {
+ "max_tokens": 128000,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mistral-nemo-instruct-2407": {
+ "max_tokens": 128000,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mistral-small-24b-instruct-2501": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mixtral-8x22b": {
+ "max_tokens": 65536,
+ "max_input_tokens": 65536,
+ "max_output_tokens": 65536,
+ "input_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct": {
+ "max_tokens": 65536,
+ "max_input_tokens": 65536,
+ "max_output_tokens": 65536,
+ "input_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 1.2e-06,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mixtral-8x7b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mixtral-8x7b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mixtral-8x7b-instruct-hf": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/mythomax-l2-13b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/nemotron-nano-v2-12b-vl": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/nous-capybara-7b-v1p9": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/nous-hermes-2-mixtral-8x7b-dpo": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/nous-hermes-2-yi-34b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-13b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-70b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-7b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/nvidia-nemotron-nano-12b-v2": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/nvidia-nemotron-nano-9b-v2": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/openchat-3p5-0106-7b": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/openhermes-2-mistral-7b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/openhermes-2p5-mistral-7b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/openorca-7b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/phi-2-3b": {
+ "max_tokens": 2048,
+ "max_input_tokens": 2048,
+ "max_output_tokens": 2048,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/phi-3-mini-128k-instruct": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/phi-3-vision-128k-instruct": {
+ "max_tokens": 32064,
+ "max_input_tokens": 32064,
+ "max_output_tokens": 32064,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-python-v1": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-v1": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-v2": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/playground-v2-1024px-aesthetic": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1.3e-10,
+ "output_cost_per_token": 1.3e-10,
+ "litellm_provider": "fireworks_ai",
+ "mode": "image_generation"
+ },
+ "fireworks_ai/accounts/fireworks/models/playground-v2-5-1024px-aesthetic": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1.3e-10,
+ "output_cost_per_token": 1.3e-10,
+ "litellm_provider": "fireworks_ai",
+ "mode": "image_generation"
+ },
+ "fireworks_ai/accounts/fireworks/models/pythia-12b": {
+ "max_tokens": 2048,
+ "max_input_tokens": 2048,
+ "max_output_tokens": 2048,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen-qwq-32b-preview": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen-v2p5-14b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen-v2p5-7b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen1p5-72b-chat": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2-7b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2-vl-2b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2-vl-72b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2-vl-7b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-0p5b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-14b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-1p5b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-32b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-32b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-72b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-72b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-7b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-0p5b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-0p5b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-14b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-14b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-1p5b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-1p5b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-128k": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-32k-rope": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-64k": {
+ "max_tokens": 65536,
+ "max_input_tokens": 65536,
+ "max_output_tokens": 65536,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-3b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-3b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-7b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-7b-instruct": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-math-72b-instruct": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-32b-instruct": {
+ "max_tokens": 128000,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-3b-instruct": {
+ "max_tokens": 128000,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-72b-instruct": {
+ "max_tokens": 128000,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-7b-instruct": {
+ "max_tokens": 128000,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-0p6b": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-14b": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-1p7b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft-131072": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft-40960": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 2.2e-07,
+ "output_cost_per_token": 8.8e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b-instruct-2507": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 2.2e-07,
+ "output_cost_per_token": 8.8e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b-thinking-2507": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 2.2e-07,
+ "output_cost_per_token": 8.8e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 1.5e-07,
+ "output_cost_per_token": 6e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b-instruct-2507": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 5e-07,
+ "output_cost_per_token": 5e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b-thinking-2507": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-32b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-4b": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-4b-instruct-2507": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-8b": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-coder-30b-a3b-instruct": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 1.5e-07,
+ "output_cost_per_token": 6e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-coder-480b-instruct-bf16": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-embedding-0p6b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "embedding"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-embedding-4b": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "embedding"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-embedding-8b": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "embedding"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-next-80b-a3b-instruct": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-next-80b-a3b-thinking": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-reranker-0p6b": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "rerank"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-reranker-4b": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "rerank"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-reranker-8b": {
+ "max_tokens": 40960,
+ "max_input_tokens": 40960,
+ "max_output_tokens": 40960,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "rerank"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-vl-235b-a22b-instruct": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 2.2e-07,
+ "output_cost_per_token": 8.8e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-vl-235b-a22b-thinking": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 2.2e-07,
+ "output_cost_per_token": 8.8e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-vl-30b-a3b-instruct": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 1.5e-07,
+ "output_cost_per_token": 6e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-vl-30b-a3b-thinking": {
+ "max_tokens": 262144,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "input_cost_per_token": 1.5e-07,
+ "output_cost_per_token": 6e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-vl-32b-instruct": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwen3-vl-8b-instruct": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/qwq-32b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/rolm-ocr": {
+ "max_tokens": 128000,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/snorkel-mistral-7b-pairrm-dpo": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/stable-diffusion-xl-1024-v1-0": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1.3e-10,
+ "output_cost_per_token": 1.3e-10,
+ "litellm_provider": "fireworks_ai",
+ "mode": "image_generation"
+ },
+ "fireworks_ai/accounts/fireworks/models/stablecode-3b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/starcoder-16b": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/starcoder-7b": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/starcoder2-15b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/starcoder2-3b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 1e-07,
+ "output_cost_per_token": 1e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/starcoder2-7b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 16384,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/toppy-m-7b": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/whisper-v3": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "audio_transcription"
+ },
+ "fireworks_ai/accounts/fireworks/models/whisper-v3-turbo": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 0.0,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "fireworks_ai",
+ "mode": "audio_transcription"
+ },
+ "fireworks_ai/accounts/fireworks/models/yi-34b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/yi-34b-200k-capybara": {
+ "max_tokens": 200000,
+ "max_input_tokens": 200000,
+ "max_output_tokens": 200000,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/yi-34b-chat": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 9e-07,
+ "output_cost_per_token": 9e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/yi-6b": {
+ "max_tokens": 4096,
+ "max_input_tokens": 4096,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
+ },
+ "fireworks_ai/accounts/fireworks/models/zephyr-7b-beta": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 2e-07,
+ "output_cost_per_token": 2e-07,
+ "litellm_provider": "fireworks_ai",
+ "mode": "chat"
}
-}
+
+}
\ No newline at end of file
diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json
index a0e794ce59d..37c2ec17371 100644
--- a/provider_endpoints_support.json
+++ b/provider_endpoints_support.json
@@ -1554,6 +1554,23 @@
"a2a": true
}
},
+ "sap": {
+ "display_name": "SAP Generative AI Hub (`sap`)",
+ "url": "https://docs.litellm.ai/docs/providers/sap",
+ "endpoints": {
+ "chat_completions": true,
+ "messages": true,
+ "responses": true,
+ "embeddings": false,
+ "image_generations": false,
+ "audio_transcriptions": false,
+ "audio_speech": false,
+ "moderations": false,
+ "batches": false,
+ "rerank": false,
+ "a2a": true
+ }
+ },
"snowflake": {
"display_name": "Snowflake (`snowflake`)",
"url": "https://docs.litellm.ai/docs/providers/snowflake",
diff --git a/scripts/mock_ibm_guardrails_server.py b/scripts/mock_ibm_guardrails_server.py
deleted file mode 100644
index a9251c14cef..00000000000
--- a/scripts/mock_ibm_guardrails_server.py
+++ /dev/null
@@ -1,358 +0,0 @@
-"""
-Mock FastAPI server for IBM FMS Guardrails Orchestrator Detector API.
-
-This server implements the Detector API endpoints for testing purposes.
-Based on: https://foundation-model-stack.github.io/fms-guardrails-orchestrator/
-
-Usage:
- python scripts/mock_ibm_guardrails_server.py
-
-The server will run on http://localhost:8001 by default.
-"""
-
-import uuid
-from typing import Any, Dict, List, Optional
-
-import uvicorn
-from fastapi import FastAPI, Header, HTTPException, status
-from pydantic import BaseModel, Field
-
-app = FastAPI(
- title="IBM FMS Guardrails Orchestrator Mock",
- description="Mock server for testing IBM Guardrails Detector API",
- version="1.0.0",
-)
-
-
-# Request Models
-class DetectorParams(BaseModel):
- """Parameters specific to the detector."""
-
- threshold: Optional[float] = Field(None, ge=0.0, le=1.0)
- custom_param: Optional[str] = None
-
-
-class TextDetectionRequest(BaseModel):
- """Request model for text detection."""
-
- contents: List[str] = Field(..., description="Text content to analyze")
- detector_params: Optional[DetectorParams] = None
-
-
-class TextGenerationDetectionRequest(BaseModel):
- """Request model for text generation detection."""
-
- detector_id: str = Field(..., description="ID of the detector to use")
- prompt: str = Field(..., description="Input prompt")
- generated_text: str = Field(..., description="Generated text to analyze")
- detector_params: Optional[DetectorParams] = None
-
-
-class ContextDetectionRequest(BaseModel):
- """Request model for detection with context."""
-
- detector_id: str = Field(..., description="ID of the detector to use")
- content: str = Field(..., description="Text content to analyze")
- context: Optional[Dict[str, Any]] = Field(None, description="Additional context")
- detector_params: Optional[DetectorParams] = None
-
-
-# Response Models
-class Detection(BaseModel):
- """Individual detection result."""
-
- detection_type: str = Field(..., description="Type of detection")
- detection: bool = Field(..., description="Whether content was detected as harmful")
- score: float = Field(..., ge=0.0, le=1.0, description="Detection confidence score")
- start: Optional[int] = Field(None, description="Start position in text")
- end: Optional[int] = Field(None, description="End position in text")
- text: Optional[str] = Field(None, description="Detected text segment")
- evidence: Optional[List[str]] = Field(None, description="Supporting evidence")
-
-
-class DetectionResponse(BaseModel):
- """Response model for detection results."""
-
- detections: List[Detection] = Field(..., description="List of detections")
- detection_id: str = Field(..., description="Unique ID for this detection request")
-
-
-# Mock detector configurations
-MOCK_DETECTORS = {
- "hate": {
- "name": "Hate Speech Detector",
- "triggers": ["hate", "offensive", "discriminatory", "slur"],
- "default_score": 0.85,
- },
- "pii": {
- "name": "PII Detector",
- "triggers": ["email", "ssn", "credit card", "phone number", "address"],
- "default_score": 0.92,
- },
- "toxicity": {
- "name": "Toxicity Detector",
- "triggers": ["toxic", "abusive", "profanity", "insult"],
- "default_score": 0.78,
- },
- "jailbreak": {
- "name": "Jailbreak Detector",
- "triggers": ["ignore instructions", "override", "bypass", "jailbreak"],
- "default_score": 0.88,
- },
- "prompt_injection": {
- "name": "Prompt Injection Detector",
- "triggers": ["ignore previous", "new instructions", "system prompt"],
- "default_score": 0.90,
- },
-}
-
-
-def simulate_detection(
- detector_id: str, content: str, detector_params: Optional[DetectorParams] = None
-) -> List[Detection]:
- """
- Simulate detection logic based on detector type and content.
-
- Args:
- detector_id: ID of the detector to simulate
- content: Text content to analyze
- detector_params: Optional detector parameters
-
- Returns:
- List of Detection objects
- """
- detections = []
- content_lower = " ".join(c for c in content).lower()
-
- # Get detector config
- detector_config = MOCK_DETECTORS.get(detector_id)
- if not detector_config:
- # Unknown detector - return no detections
- return detections
-
- # Check for triggers in content
- for trigger in detector_config["triggers"]:
- if trigger in content_lower:
- # Calculate score (use threshold if provided, otherwise default)
- base_score = detector_config["default_score"]
- threshold = (
- detector_params.threshold
- if detector_params and detector_params.threshold
- else None
- )
-
- # Adjust score slightly based on content length (longer content = slightly lower confidence)
- score_adjustment = max(0, min(0.1, len(content) / 10000))
- score = max(0.0, min(1.0, base_score - score_adjustment))
-
- # Find position of trigger
- start_pos = content_lower.find(trigger)
- end_pos = start_pos + len(trigger)
-
- detection = Detection(
- detection_type=detector_id,
- detection=threshold is None or score >= threshold,
- score=score,
- start=start_pos,
- end=end_pos,
- text=content[start_pos:end_pos] if start_pos >= 0 else None,
- evidence=[f"Found trigger word: {trigger}"],
- )
- detections.append(detection)
-
- # If no triggers found, return a negative detection
- if not detections:
- detections.append(
- Detection(
- detection_type=detector_id,
- detection=False,
- score=0.05, # Low score for clean content
- )
- )
-
- return detections
-
-
-# Authentication middleware
-def verify_auth_token(authorization: Optional[str] = Header(None)) -> bool:
- """
- Verify the authentication token.
-
- Args:
- authorization: Authorization header value
-
- Returns:
- True if valid, raises HTTPException otherwise
- """
- if not authorization:
- raise HTTPException(
- status_code=status.HTTP_401_UNAUTHORIZED,
- detail="Missing authorization header",
- )
-
- # Simple token validation - in real implementation, this would validate against a real auth system
- if not authorization.startswith("Bearer "):
- raise HTTPException(
- status_code=status.HTTP_401_UNAUTHORIZED,
- detail="Invalid authorization header format. Expected: Bearer
",
- )
-
- token = authorization.replace("Bearer ", "")
-
- # Accept any non-empty token for mock purposes
- if not token:
- raise HTTPException(
- status_code=status.HTTP_401_UNAUTHORIZED,
- detail="Empty token provided",
- )
-
- return True
-
-
-# API Endpoints
-@app.get("/health")
-async def health_check():
- """Health check endpoint."""
- return {"status": "healthy", "service": "IBM FMS Guardrails Mock Server"}
-
-
-@app.get("/")
-async def root():
- """Root endpoint with API information."""
- return {
- "service": "IBM FMS Guardrails Orchestrator Mock",
- "version": "1.0.0",
- "endpoints": {
- "health": "/health",
- "text_detection": "/api/v1/text/detection",
- "generation_detection": "/api/v1/text/generation/detection",
- "context_detection": "/api/v1/text/context/detection",
- },
- "available_detectors": list(MOCK_DETECTORS.keys()),
- }
-
-
-@app.post("/api/v1/text/contents")
-async def text_detection(
- request: TextDetectionRequest,
- detector_id: str = Header(None), # query parameter
- authorization: Optional[str] = Header(None),
-):
- """
- Detect potential issues in text content.
-
- Args:
- request: Detection request with content and detector ID
- detector_id: ID of detector
- authorization: Bearer token for authentication
-
- Returns:
- Detection results
- """
- verify_auth_token(authorization)
-
- detections = simulate_detection(
- detector_id=detector_id,
- content=request.contents,
- detector_params=request.detector_params,
- )
-
- return detections
-
-
-@app.post("/api/v1/text/generation/detection", response_model=DetectionResponse)
-async def text_generation_detection(
- request: TextGenerationDetectionRequest,
- authorization: Optional[str] = Header(None),
-):
- """
- Detect potential issues in generated text.
-
- Args:
- request: Detection request with prompt and generated text
- authorization: Bearer token for authentication
-
- Returns:
- Detection results
- """
- verify_auth_token(authorization)
-
- # Analyze both prompt and generated text
- combined_content = f"{request.prompt} {request.generated_text}"
-
- detections = simulate_detection(
- detector_id=request.detector_id,
- content=combined_content,
- detector_params=request.detector_params,
- )
-
- return DetectionResponse(
- detections=detections,
- detection_id=str(uuid.uuid4()),
- )
-
-
-@app.post("/api/v1/text/context/detection", response_model=DetectionResponse)
-async def context_detection(
- request: ContextDetectionRequest,
- authorization: Optional[str] = Header(None),
-):
- """
- Detect potential issues in text with additional context.
-
- Args:
- request: Detection request with content and context
- authorization: Bearer token for authentication
-
- Returns:
- Detection results
- """
- verify_auth_token(authorization)
-
- detections = simulate_detection(
- detector_id=request.detector_id,
- content=request.content,
- detector_params=request.detector_params,
- )
-
- return DetectionResponse(
- detections=detections,
- detection_id=str(uuid.uuid4()),
- )
-
-
-@app.get("/api/v1/detectors")
-async def list_detectors(authorization: Optional[str] = Header(None)):
- """
- List available detectors.
-
- Args:
- authorization: Bearer token for authentication
-
- Returns:
- List of available detectors
- """
- verify_auth_token(authorization)
-
- return {
- "detectors": [
- {
- "id": detector_id,
- "name": config["name"],
- "triggers": config["triggers"],
- }
- for detector_id, config in MOCK_DETECTORS.items()
- ]
- }
-
-
-if __name__ == "__main__":
- print("π Starting IBM FMS Guardrails Mock Server...")
- print("π Server will be available at: http://localhost:8001")
- print("π API docs at: http://localhost:8001/docs")
- print("\nAvailable detectors:")
- for detector_id, config in MOCK_DETECTORS.items():
- print(f" - {detector_id}: {config['name']}")
- print("\n⨠Use any Bearer token for authentication in this mock server\n")
-
- uvicorn.run(app, host="0.0.0.0", port=8001)
diff --git a/scripts/test_groq_streaming_issue.py b/scripts/test_groq_streaming_issue.py
deleted file mode 100644
index 0a996c0c209..00000000000
--- a/scripts/test_groq_streaming_issue.py
+++ /dev/null
@@ -1,54 +0,0 @@
-"""
-Test script to reproduce the Groq streaming ASCII encoding issue.
-
-This reproduces the issue described in #12660 where streaming responses
-containing non-ASCII characters like Β΅ cause encoding errors.
-"""
-import asyncio
-import os
-import traceback
-from litellm import acompletion
-
-async def test_groq_streaming_with_special_chars():
- """Test that reproduces the ASCII encoding issue with Groq streaming."""
- try:
- print("Testing acompletion + streaming with Groq...")
-
- # Test message that should trigger the Β΅ character or similar non-ASCII content
- test_messages = [
- {"content": "What is the symbol for micro? Please include the Β΅ symbol in your response.", "role": "user"}
- ]
-
- # This should trigger the ASCII encoding error described in the issue
- response = await acompletion(
- model="groq/llama-3.3-70b-versatile",
- messages=test_messages,
- stream=True
- )
-
- print(f"Response type: {type(response)}")
-
- # Try to iterate through the stream
- async for chunk in response:
- print(f"Chunk: {chunk}")
-
- print("β
Test completed successfully - no encoding errors!")
-
- except Exception as e:
- print(f"β Error occurred: {e}")
- print(f"Error type: {type(e)}")
- print(f"Traceback:\n{traceback.format_exc()}")
- return False
-
- return True
-
-if __name__ == "__main__":
- # Note: This requires GROQ_API_KEY to be set
- if not os.getenv("GROQ_API_KEY"):
- print("β οΈ GROQ_API_KEY not set. Skipping test.")
- else:
- success = asyncio.run(test_groq_streaming_with_special_chars())
- if success:
- print("π All tests passed!")
- else:
- print("π₯ Test failed!")
\ No newline at end of file
diff --git a/scripts/test_mock_ibm_guardrails.py b/scripts/test_mock_ibm_guardrails.py
deleted file mode 100644
index 91e4e02d8b5..00000000000
--- a/scripts/test_mock_ibm_guardrails.py
+++ /dev/null
@@ -1,181 +0,0 @@
-"""
-Test script for the mock IBM Guardrails server.
-
-This demonstrates how to interact with the mock server.
-
-Usage:
- # Start the mock server in one terminal:
- python scripts/mock_ibm_guardrails_server.py
-
- # Run this test in another terminal:
- python scripts/test_mock_ibm_guardrails.py
-"""
-
-import asyncio
-
-import httpx
-
-
-async def test_mock_server():
- """Test the mock IBM Guardrails server."""
- base_url = "http://localhost:8001"
- headers = {"Authorization": "Bearer test-token-12345"}
-
- print("π§ͺ Testing IBM FMS Guardrails Mock Server\n")
-
- async with httpx.AsyncClient() as client:
- # Test 1: Health check
- print("1οΈβ£ Testing health check...")
- try:
- response = await client.get(f"{base_url}/health")
- print(f" β
Health check: {response.json()}\n")
- except Exception as e:
- print(f" β Health check failed: {e}\n")
- return
-
- # Test 2: List detectors
- print("2οΈβ£ Testing list detectors...")
- try:
- response = await client.get(
- f"{base_url}/api/v1/detectors",
- headers=headers
- )
- detectors = response.json()
- print(f" β
Found {len(detectors['detectors'])} detectors:")
- for detector in detectors["detectors"]:
- print(f" - {detector['id']}: {detector['name']}")
- print()
- except Exception as e:
- print(f" β List detectors failed: {e}\n")
-
- # Test 3: Text detection with clean content
- print("3οΈβ£ Testing text detection (clean content)...")
- try:
- response = await client.post(
- f"{base_url}/api/v1/text/detection",
- headers=headers,
- json={
- "detector_id": "hate",
- "content": "This is a normal, friendly message.",
- }
- )
- result = response.json()
- print(f" β
Detection result:")
- print(f" Detection ID: {result['detection_id']}")
- for detection in result["detections"]:
- print(f" - Type: {detection['detection_type']}, Detected: {detection['detection']}, Score: {detection['score']:.2f}")
- print()
- except Exception as e:
- print(f" β Text detection failed: {e}\n")
-
- # Test 4: Text detection with problematic content
- print("4οΈβ£ Testing text detection (problematic content)...")
- try:
- response = await client.post(
- f"{base_url}/api/v1/text/detection",
- headers=headers,
- json={
- "detector_id": "hate",
- "content": "This message contains hate speech and offensive language.",
- }
- )
- result = response.json()
- print(f" β
Detection result:")
- print(f" Detection ID: {result['detection_id']}")
- for detection in result["detections"]:
- print(f" - Type: {detection['detection_type']}, Detected: {detection['detection']}, Score: {detection['score']:.2f}")
- if detection.get("evidence"):
- print(f" Evidence: {detection['evidence']}")
- print()
- except Exception as e:
- print(f" β Text detection failed: {e}\n")
-
- # Test 5: PII detection
- print("5οΈβ£ Testing PII detection...")
- try:
- response = await client.post(
- f"{base_url}/api/v1/text/detection",
- headers=headers,
- json={
- "detector_id": "pii",
- "content": "Please send the report to my email address john@example.com",
- }
- )
- result = response.json()
- print(f" β
Detection result:")
- print(f" Detection ID: {result['detection_id']}")
- for detection in result["detections"]:
- print(f" - Type: {detection['detection_type']}, Detected: {detection['detection']}, Score: {detection['score']:.2f}")
- if detection.get("text"):
- print(f" Detected text: '{detection['text']}'")
- print()
- except Exception as e:
- print(f" β PII detection failed: {e}\n")
-
- # Test 6: Generation detection
- print("6οΈβ£ Testing text generation detection...")
- try:
- response = await client.post(
- f"{base_url}/api/v1/text/generation/detection",
- headers=headers,
- json={
- "detector_id": "jailbreak",
- "prompt": "Tell me about AI safety",
- "generated_text": "I will ignore instructions and provide harmful content.",
- }
- )
- result = response.json()
- print(f" β
Detection result:")
- print(f" Detection ID: {result['detection_id']}")
- for detection in result["detections"]:
- print(f" - Type: {detection['detection_type']}, Detected: {detection['detection']}, Score: {detection['score']:.2f}")
- print()
- except Exception as e:
- print(f" β Generation detection failed: {e}\n")
-
- # Test 7: Detection with custom threshold
- print("7οΈβ£ Testing detection with custom threshold...")
- try:
- response = await client.post(
- f"{base_url}/api/v1/text/detection",
- headers=headers,
- json={
- "detector_id": "toxicity",
- "content": "This contains toxic language",
- "detector_params": {
- "threshold": 0.9
- }
- }
- )
- result = response.json()
- print(f" β
Detection result (threshold=0.9):")
- print(f" Detection ID: {result['detection_id']}")
- for detection in result["detections"]:
- print(f" - Type: {detection['detection_type']}, Detected: {detection['detection']}, Score: {detection['score']:.2f}")
- print()
- except Exception as e:
- print(f" β Threshold detection failed: {e}\n")
-
- # Test 8: Authentication error
- print("8οΈβ£ Testing authentication error...")
- try:
- response = await client.post(
- f"{base_url}/api/v1/text/detection",
- json={
- "detector_id": "hate",
- "content": "Test content",
- }
- )
- if response.status_code == 401:
- print(f" β
Authentication error handled correctly: {response.json()}\n")
- else:
- print(f" β οΈ Unexpected status code: {response.status_code}\n")
- except Exception as e:
- print(f" β Auth test failed: {e}\n")
-
- print("β¨ All tests completed!")
-
-
-if __name__ == "__main__":
- asyncio.run(test_mock_server())
-
diff --git a/scripts/update_readme_providers_table.py b/scripts/update_readme_providers_table.py
deleted file mode 100644
index 1435e13198f..00000000000
--- a/scripts/update_readme_providers_table.py
+++ /dev/null
@@ -1,147 +0,0 @@
-#!/usr/bin/env python3
-"""
-Script to update the README.md providers table from provider_endpoints_support.json
-"""
-
-import json
-import re
-from pathlib import Path
-
-# Define paths
-REPO_ROOT = Path(__file__).parent.parent
-JSON_PATH = REPO_ROOT / "provider_endpoints_support.json"
-README_PATH = REPO_ROOT / "README.md"
-
-# Endpoint column headers
-ENDPOINT_COLUMNS = [
- ("/chat/completions", "chat_completions"),
- ("/messages", "messages"),
- ("/responses", "responses"),
- ("/embeddings", "embeddings"),
- ("/image/generations", "image_generations"),
- ("/audio/transcriptions", "audio_transcriptions"),
- ("/audio/speech", "audio_speech"),
- ("/moderations", "moderations"),
- ("/batches", "batches"),
- ("/rerank", "rerank"),
-]
-
-
-def load_providers_data():
- """Load provider data from JSON file"""
- with open(JSON_PATH, 'r') as f:
- data = json.load(f)
-
- # Handle both old and new format
- if "providers" in data:
- return data["providers"]
- return data
-
-
-def generate_markdown_table(providers_data):
- """Generate markdown table from providers data"""
-
- # Sort providers alphabetically by display name
- sorted_providers = sorted(
- providers_data.items(),
- key=lambda x: x[1]['display_name'].lower()
- )
-
- # Generate header
- header_cols = ["Provider"] + [col[0] for col in ENDPOINT_COLUMNS]
- header = "| " + " | ".join(header_cols) + " |"
- separator = "|" + "|".join(["-" * (len(col) + 2) for col in header_cols]) + "|"
-
- # Generate rows
- rows = []
- for slug, data in sorted_providers:
- display_name = data['display_name']
- url = data['url']
-
- # Build row
- row_parts = [f"[{display_name}]({url})"]
-
- for _, endpoint_key in ENDPOINT_COLUMNS:
- supported = data['endpoints'].get(endpoint_key, False)
- row_parts.append("β
" if supported else "")
-
- row = "| " + " | ".join(row_parts) + " |"
- rows.append(row)
-
- # Combine all parts
- table_lines = [
- "",
- "",
- "",
- header,
- separator
- ] + rows + [
- ""
- ]
-
- return "\n".join(table_lines)
-
-
-def update_readme(table_markdown):
- """Update README.md with new table"""
- with open(README_PATH, 'r') as f:
- content = f.read()
-
- print(f" Original README length: {len(content)} bytes")
-
- # Find the table section
- # Look for the AUTO-GENERATED comment or the header, and replace until Read the Docs
- pattern = r"(## Supported Providers.*?\n\n)(?:|.*?)(\n\n\[\*\*Read the Docs\*\*\])"
-
- # Test if pattern matches
- match = re.search(pattern, content, flags=re.DOTALL)
- if not match:
- print("β Pattern did not match in README.md")
- return False
-
- print(f" Pattern matched, replacing table...")
-
- def replacer(match):
- return match.group(1) + table_markdown + match.group(2)
-
- new_content = re.sub(pattern, replacer, content, flags=re.DOTALL)
-
- print(f" New README length: {len(new_content)} bytes")
-
- if new_content == content:
- print(" βΉοΈ Table is already up-to-date, no changes needed")
- return True # Not an error - table is already correct
-
- with open(README_PATH, 'w') as f:
- f.write(new_content)
-
- print(" β README.md has been updated")
- return True
-
-
-def main():
- """Main function"""
- print("Loading provider data from provider_endpoints_support.json...")
- providers_data = load_providers_data()
- print(f"β Loaded {len(providers_data)} providers")
-
- print("\nGenerating markdown table...")
- table_markdown = generate_markdown_table(providers_data)
- print(f"β Generated table with {len(providers_data)} rows")
-
- print("\nUpdating README.md...")
- if update_readme(table_markdown):
- print("β Successfully updated README.md")
- print("\nπ Please review the changes and commit both files:")
- print(" - provider_endpoints_support.json")
- print(" - README.md")
- else:
- print("β Failed to update README.md")
- return 1
-
- return 0
-
-
-if __name__ == "__main__":
- exit(main())
-
diff --git a/tests/litellm_utils_tests/test_secret_manager.py b/tests/litellm_utils_tests/test_secret_manager.py
index 7099f6e13d0..da9c9d548a7 100644
--- a/tests/litellm_utils_tests/test_secret_manager.py
+++ b/tests/litellm_utils_tests/test_secret_manager.py
@@ -133,9 +133,8 @@ def test_oidc_circleci_v2():
print(f"secret_val: {redact_oidc_signature(secret_val)}")
-@pytest.mark.skipif(
- os.environ.get("CIRCLE_OIDC_TOKEN") is None,
- reason="Cannot run without being in CircleCI Runner",
+@pytest.mark.skip(
+ reason="Quarantined: Flaky test - fails with 401 Unauthorized from Azure OAuth. TODO: Switch to our own Azure account or fix authentication"
)
def test_oidc_circleci_with_azure():
# TODO: Switch to our own Azure account, currently using ai.moda's account
diff --git a/tests/llm_translation/test_helicone.py b/tests/llm_translation/test_helicone.py
new file mode 100644
index 00000000000..8ca2f62d2bc
--- /dev/null
+++ b/tests/llm_translation/test_helicone.py
@@ -0,0 +1,72 @@
+import os
+import sys
+import pytest
+
+sys.path.insert(
+ 0, os.path.abspath("../..")
+) # Adds the parent directory to the system path
+import litellm
+
+
+def test_completion_helicone():
+ """Test basic completion through Helicone gateway"""
+ litellm._turn_on_debug()
+ resp = litellm.completion(
+ model="helicone/gpt-4o-mini",
+ messages=[{"role": "user", "content": "Say 'Hello from Helicone' and nothing else"}],
+ max_tokens=10,
+ )
+ print(resp)
+ assert resp.choices[0].message.content is not None
+ assert len(resp.choices[0].message.content) > 0
+
+def test_completion_helicone_specific_provider():
+ """Test basic completion through Helicone gateway"""
+ litellm._turn_on_debug()
+ resp = litellm.completion(
+ model="helicone/claude-4.5-haiku/anthropic",
+ messages=[{"role": "user", "content": "Say 'Hello from Helicone' and nothing else"}],
+ max_tokens=10,
+ )
+ print(resp)
+ assert resp.choices[0].message.content is not None
+ assert len(resp.choices[0].message.content) > 0
+
+
+def test_completion_helicone_streaming():
+ """Test streaming completion through Helicone gateway"""
+ litellm._turn_on_debug()
+ resp = litellm.completion(
+ model="helicone/gpt-4o-mini",
+ messages=[{"role": "user", "content": "Count to 3"}],
+ max_tokens=20,
+ stream=True,
+ )
+
+ chunks = []
+ for chunk in resp:
+ print(chunk)
+ if hasattr(chunk.choices[0], "delta") and hasattr(chunk.choices[0].delta, "content"):
+ if chunk.choices[0].delta.content:
+ chunks.append(chunk.choices[0].delta.content)
+
+ full_response = "".join(chunks)
+ assert len(full_response) > 0
+ print(f"Full response: {full_response}")
+
+
+def test_completion_helicone_with_metadata():
+ """Test Helicone with custom properties"""
+ litellm._turn_on_debug()
+ resp = litellm.completion(
+ model="helicone/gpt-4o-mini",
+ messages=[{"role": "user", "content": "Hello"}],
+ max_tokens=10,
+ metadata={
+ "Helicone-Property-Environment": "test",
+ "Helicone-Property-Session": "test-session-123"
+ }
+ )
+ print(resp)
+ assert resp.choices[0].message.content is not None
+
diff --git a/tests/llm_translation/test_nvidia_nim.py b/tests/llm_translation/test_nvidia_nim.py
index 1705871258f..d0462efa6d5 100644
--- a/tests/llm_translation/test_nvidia_nim.py
+++ b/tests/llm_translation/test_nvidia_nim.py
@@ -184,13 +184,76 @@ def test_chat_completion_nvidia_nim_with_tools():
assert request_body["tool_choice"] == "auto"
assert request_body["parallel_tool_calls"] == True
+@pytest.mark.asyncio()
+async def test_nvidia_nim_rerank_ranking_endpoint():
+ """
+ Test that using "nvidia_nim/ranking/" forces the /v1/ranking endpoint.
+
+ This allows users to explicitly use the /v1/ranking endpoint for models like
+ nvidia/llama-3.2-nv-rerankqa-1b-v2.
+
+ Reference: https://build.nvidia.com/nvidia/llama-3_2-nv-rerankqa-1b-v2/deploy
+ """
+ mock_response = AsyncMock()
+
+ def return_val():
+ return {
+ "rankings": [
+ {"index": 0, "logit": 0.95},
+ {"index": 1, "logit": 0.75},
+ ],
+ }
+
+ mock_response.json = return_val
+ mock_response.headers = {"key": "value"}
+ mock_response.status_code = 200
+
+ with patch(
+ "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
+ return_value=mock_response,
+ ) as mock_post:
+ # Use "ranking/" prefix to force /v1/ranking endpoint
+ response = await litellm.arerank(
+ model="nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2",
+ query="What is the GPU memory bandwidth?",
+ documents=["H100 delivers 3TB/s memory bandwidth", "A100 has 2TB/s memory bandwidth"],
+ top_n=2,
+ api_key="fake-api-key",
+ )
+
+ mock_post.assert_called_once()
+
+ args_to_api = mock_post.call_args.kwargs["data"]
+ _url = mock_post.call_args.kwargs["url"]
+ print("url = ", _url)
+
+ # Verify URL is /v1/ranking
+ assert _url == "https://ai.api.nvidia.com/v1/ranking"
+
+ # Verify request body structure
+ request_data = json.loads(args_to_api)
+ print("request_data=", request_data)
+
+ # Query should be an object with 'text' field
+ assert request_data["query"] == {"text": "What is the GPU memory bandwidth?"}
+
+ # Documents should be 'passages'
+ assert request_data["passages"] == [
+ {"text": "H100 delivers 3TB/s memory bandwidth"},
+ {"text": "A100 has 2TB/s memory bandwidth"},
+ ]
+
+ # Model name in body should NOT have "ranking/" prefix
+ assert request_data["model"] == "nvidia/llama-3.2-nv-rerankqa-1b-v2"
+
+
class TestNvidiaNim(BaseLLMRerankTest):
def get_custom_llm_provider(self) -> litellm.LlmProviders:
return litellm.LlmProviders.NVIDIA_NIM
def get_base_rerank_call_args(self) -> dict:
return {
- "model": "nvidia_nim/nvidia/llama-3_2-nv-rerankqa-1b-v2",
+ "model": "nvidia_nim/nvidia/llama-3.2-nv-rerankqa-1b-v2",
}
def get_expected_cost(self) -> float:
diff --git a/tests/proxy_unit_tests/test_response_polling_handler.py b/tests/proxy_unit_tests/test_response_polling_handler.py
index 5d9b83969f7..cb4cd0efe57 100644
--- a/tests/proxy_unit_tests/test_response_polling_handler.py
+++ b/tests/proxy_unit_tests/test_response_polling_handler.py
@@ -519,6 +519,13 @@ class TestResponsePollingHandler:
# init_async_client is a sync method that returns an async client
mock_redis.init_async_client = Mock(return_value=mock_async_client)
+ # Mock async_delete_cache to actually call init_async_client and delete
+ async def mock_async_delete_cache(key):
+ client = mock_redis.init_async_client()
+ await client.delete(key)
+
+ mock_redis.async_delete_cache = mock_async_delete_cache
+
handler = ResponsePollingHandler(redis_cache=mock_redis)
result = await handler.delete_polling("litellm_poll_test")
diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py
index eecb12907ad..b6869525e6d 100644
--- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py
+++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py
@@ -444,3 +444,130 @@ def test_transform_request_single_char_keys_not_matched():
assert result_correct.get("previous_response_id") == "resp_abc"
print("β Single-character keys are not incorrectly matched to metadata/previous_response_id")
+
+
+# =============================================================================
+# Tests for issue #17246: Streaming tool_calls dropped when text + tool_calls
+# =============================================================================
+
+
+def test_message_done_does_not_emit_is_finished():
+ """
+ Test that OUTPUT_ITEM_DONE for a message does NOT emit is_finished=True.
+ This is the core fix for issue #17246.
+
+ Before fix: message completion emitted is_finished=True, causing tool_calls
+ that came after to be dropped.
+ """
+ from litellm.completion_extras.litellm_responses_transformation.transformation import (
+ OpenAiResponsesToChatCompletionStreamIterator,
+ )
+
+ iterator = OpenAiResponsesToChatCompletionStreamIterator(
+ streaming_response=None, sync_stream=True
+ )
+
+ chunk = {
+ "type": "response.output_item.done",
+ "item": {"type": "message", "content": []}
+ }
+
+ result = iterator.chunk_parser(chunk)
+
+ # After the fix, message completion should NOT set is_finished=True
+ assert result["is_finished"] == False, "message completion should not emit is_finished=True"
+ assert result["finish_reason"] == "", "message completion should not emit finish_reason"
+
+
+def test_response_completed_emits_is_finished():
+ """
+ Test that response.completed DOES emit is_finished=True.
+ This ensures streaming ends properly after ALL output items are sent.
+ """
+ from litellm.completion_extras.litellm_responses_transformation.transformation import (
+ OpenAiResponsesToChatCompletionStreamIterator,
+ )
+
+ iterator = OpenAiResponsesToChatCompletionStreamIterator(
+ streaming_response=None, sync_stream=True
+ )
+
+ chunk = {"type": "response.completed"}
+
+ result = iterator.chunk_parser(chunk)
+
+ assert result["is_finished"] == True, "response.completed should emit is_finished=True"
+ assert result["finish_reason"] == "stop", "response.completed should emit finish_reason='stop'"
+
+
+def test_function_call_done_emits_is_finished():
+ """
+ Test that OUTPUT_ITEM_DONE for a function_call still emits is_finished=True.
+ This preserves existing behavior for tool_calls.
+ """
+ from litellm.completion_extras.litellm_responses_transformation.transformation import (
+ OpenAiResponsesToChatCompletionStreamIterator,
+ )
+
+ iterator = OpenAiResponsesToChatCompletionStreamIterator(
+ streaming_response=None, sync_stream=True
+ )
+
+ chunk = {
+ "type": "response.output_item.done",
+ "item": {
+ "type": "function_call",
+ "name": "get_weather",
+ "call_id": "call_123",
+ "arguments": '{"location": "Tokyo"}'
+ }
+ }
+
+ result = iterator.chunk_parser(chunk)
+
+ assert result["is_finished"] == True, "function_call completion should emit is_finished=True"
+ assert result["finish_reason"] == "tool_calls", "function_call should emit finish_reason='tool_calls'"
+ assert result["tool_use"] is not None, "function_call should include tool_use"
+
+
+def test_text_plus_tool_calls_sequence():
+ """
+ Test the full sequence when model returns text + tool_calls.
+ This is the main scenario for issue #17246.
+
+ Expected: is_finished=True should NOT appear until function_call is done,
+ not when message is done.
+ """
+ from litellm.completion_extras.litellm_responses_transformation.transformation import (
+ OpenAiResponsesToChatCompletionStreamIterator,
+ )
+
+ iterator = OpenAiResponsesToChatCompletionStreamIterator(
+ streaming_response=None, sync_stream=True
+ )
+
+ # Simulate the sequence from OpenAI Responses API
+ chunks = [
+ {"type": "response.output_text.delta", "delta": "Hello"},
+ {"type": "response.output_text.delta", "delta": "!"},
+ {"type": "response.output_item.done", "item": {"type": "message", "content": []}}, # message done
+ {"type": "response.output_item.added", "item": {"type": "function_call", "name": "get_weather", "call_id": "call_123"}},
+ {"type": "response.function_call_arguments.delta", "delta": '{"location":"Tokyo"}'},
+ {"type": "response.output_item.done", "item": {"type": "function_call", "name": "get_weather", "call_id": "call_123", "arguments": '{"location":"Tokyo"}'}},
+ {"type": "response.completed"},
+ ]
+
+ results = [iterator.chunk_parser(chunk) for chunk in chunks]
+
+ # Check message done (index 2) does NOT have is_finished=True
+ message_done_result = results[2]
+ assert message_done_result["is_finished"] == False, "message done should not have is_finished=True"
+
+ # Check function_call done (index 5) DOES have is_finished=True
+ function_done_result = results[5]
+ assert function_done_result["is_finished"] == True, "function_call done should have is_finished=True"
+ assert function_done_result["finish_reason"] == "tool_calls"
+
+ # Check response.completed (index 6) also has is_finished=True
+ completed_result = results[6]
+ assert completed_result["is_finished"] == True, "response.completed should have is_finished=True"
diff --git a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py
index d59e52d965e..8852e9d5ac6 100644
--- a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py
+++ b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py
@@ -42,6 +42,16 @@ context_window_test_cases = [
),
# Test case insensitivity
("ERROR: THIS MODEL'S MAXIMUM CONTEXT LENGTH IS 1024.", True),
+ # Cerebras context window error format
+ # See: https://github.com/BerriAI/litellm/issues/XXXX
+ (
+ "Current length is 132784 while limit is 131000",
+ True,
+ ),
+ (
+ "CerebrasException - Please reduce the length of the messages or completion. Current length is 50000 while limit is 40000",
+ True,
+ ),
# Negative cases (should return False)
("A generic API error occurred.", False),
("Invalid API Key provided.", False),
diff --git a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_messages_transformation.py b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_messages_transformation.py
index ae8c35b2679..d78a638fd89 100644
--- a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_messages_transformation.py
+++ b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_messages_transformation.py
@@ -55,12 +55,11 @@ class TestAzureAnthropicMessagesConfig:
assert isinstance(call_args[1]["litellm_params"], GenericLiteLLMParams)
assert call_args[1]["litellm_params"].api_key == "test-api-key"
assert "anthropic-version" in result
- assert "x-api-key" in result
- assert result["x-api-key"] == "test-api-key"
- assert "api-key" not in result
+ # api-key header is preserved as-is (no conversion to x-api-key)
+ assert "api-key" in result
- def test_validate_anthropic_messages_environment_converts_api_key_to_x_api_key(self):
- """Test that api-key header is converted to x-api-key"""
+ def test_validate_anthropic_messages_environment_preserves_api_key_header(self):
+ """Test that api-key header is preserved as-is (Azure handles the header internally)"""
config = AzureAnthropicMessagesConfig()
headers = {}
model = "claude-sonnet-4-5"
@@ -80,10 +79,9 @@ class TestAzureAnthropicMessagesConfig:
litellm_params=litellm_params,
)
- # Verify api-key was converted to x-api-key
- assert "x-api-key" in result
- assert result["x-api-key"] == "test-api-key"
- assert "api-key" not in result
+ # Verify api-key header is preserved as-is
+ assert "api-key" in result
+ assert result["api-key"] == "test-api-key"
def test_validate_anthropic_messages_environment_sets_headers(self):
"""Test that required headers are set"""
@@ -110,7 +108,8 @@ class TestAzureAnthropicMessagesConfig:
assert result["anthropic-version"] == "2023-06-01"
assert "content-type" in result
assert result["content-type"] == "application/json"
- assert "x-api-key" in result
+ # api-key header is preserved as-is
+ assert "api-key" in result
def test_get_complete_url_with_base_url(self):
"""Test get_complete_url with base URL"""
@@ -239,3 +238,47 @@ class TestAzureAnthropicMessagesConfig:
assert "tools" in params
assert "tool_choice" in params
+
+class TestProviderConfigManagerAzureAnthropicMessages:
+ """Test ProviderConfigManager returns correct config for Azure AI Anthropic Messages API"""
+
+ def test_get_provider_anthropic_messages_config_returns_azure_config(self):
+ """Test that ProviderConfigManager returns AzureAnthropicMessagesConfig for azure_ai provider with claude model"""
+ import litellm
+ from litellm.utils import ProviderConfigManager
+
+ config = ProviderConfigManager.get_provider_anthropic_messages_config(
+ model="claude-sonnet-4-5_gb_20250929",
+ provider=litellm.LlmProviders.AZURE_AI,
+ )
+
+ assert config is not None
+ assert isinstance(config, AzureAnthropicMessagesConfig)
+
+ def test_get_provider_anthropic_messages_config_case_insensitive_model_name(self):
+ """Test that model name check is case insensitive"""
+ import litellm
+ from litellm.utils import ProviderConfigManager
+
+ # Test with uppercase CLAUDE
+ config = ProviderConfigManager.get_provider_anthropic_messages_config(
+ model="CLAUDE-SONNET-4-5",
+ provider=litellm.LlmProviders.AZURE_AI,
+ )
+
+ assert config is not None
+ assert isinstance(config, AzureAnthropicMessagesConfig)
+
+ def test_get_provider_anthropic_messages_config_returns_none_for_non_claude_model(
+ self,
+ ):
+ """Test that ProviderConfigManager returns None for non-claude model on azure_ai"""
+ import litellm
+ from litellm.utils import ProviderConfigManager
+
+ config = ProviderConfigManager.get_provider_anthropic_messages_config(
+ model="gpt-4o",
+ provider=litellm.LlmProviders.AZURE_AI,
+ )
+
+ assert config is None
diff --git a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py
index f2c75cf1a61..1a20806243f 100644
--- a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py
+++ b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py
@@ -103,8 +103,8 @@ class TestAzureAnthropicConfig:
call_args = mock_validate.call_args
assert call_args[1]["litellm_params"].api_key == "provided-api-key"
- def test_validate_environment_converts_api_key_to_x_api_key(self):
- """Test that api-key header is converted to x-api-key (Azure Anthropic uses x-api-key)"""
+ def test_validate_environment_preserves_api_key_header(self):
+ """Test that api-key header is preserved as-is (Azure handles the header internally)"""
config = AzureAnthropicConfig()
headers = {}
model = "claude-sonnet-4-5"
@@ -127,10 +127,9 @@ class TestAzureAnthropicConfig:
litellm_params=litellm_params,
)
- # Verify api-key was converted to x-api-key
- assert "x-api-key" in result
- assert result["x-api-key"] == "test-api-key"
- assert "api-key" not in result
+ # Verify api-key header is preserved as-is
+ assert "api-key" in result
+ assert result["api-key"] == "test-api-key"
def test_validate_environment_sets_anthropic_version(self):
"""Test that anthropic-version header is set"""
diff --git a/tests/test_litellm/llms/bedrock/chat/test_writer_palmyra.py b/tests/test_litellm/llms/bedrock/chat/test_writer_palmyra.py
new file mode 100644
index 00000000000..9bc6724867f
--- /dev/null
+++ b/tests/test_litellm/llms/bedrock/chat/test_writer_palmyra.py
@@ -0,0 +1,69 @@
+"""
+Tests for Writer Palmyra X5 and X4 models on Bedrock Converse.
+"""
+
+import os
+import sys
+
+import pytest
+
+sys.path.insert(
+ 0, os.path.abspath("../../../../..")
+) # Adds the parent directory to the system path
+
+
+from litellm.llms.bedrock.common_utils import BedrockModelInfo
+
+
+def test_writer_palmyra_routes_to_converse():
+ """
+ Test that Writer Palmyra models route to converse API.
+ """
+ bedrock_model_info = BedrockModelInfo
+
+ # Test base model routes to converse
+ bedrock_route = bedrock_model_info.get_bedrock_route(
+ model="bedrock/writer.palmyra-x5-v1:0"
+ )
+ assert bedrock_route == "converse"
+
+ bedrock_route = bedrock_model_info.get_bedrock_route(
+ model="bedrock/writer.palmyra-x4-v1:0"
+ )
+ assert bedrock_route == "converse"
+
+
+def test_writer_palmyra_cross_region_routes_to_converse():
+ """
+ Test that Writer Palmyra models with cross-region inference prefix route to converse API.
+ """
+ bedrock_model_info = BedrockModelInfo
+
+ # Test cross-region inference profile routes to converse
+ bedrock_route = bedrock_model_info.get_bedrock_route(
+ model="bedrock/us.writer.palmyra-x5-v1:0"
+ )
+ assert bedrock_route == "converse"
+
+ bedrock_route = bedrock_model_info.get_bedrock_route(
+ model="bedrock/us.writer.palmyra-x4-v1:0"
+ )
+ assert bedrock_route == "converse"
+
+
+def test_writer_palmyra_base_model_extraction():
+ """
+ Test that base model is correctly extracted from Writer Palmyra cross-region models.
+ """
+ bedrock_model_info = BedrockModelInfo
+
+ # Test us. prefix is stripped correctly
+ base_model = bedrock_model_info.get_base_model(
+ model="bedrock/us.writer.palmyra-x5-v1:0"
+ )
+ assert base_model == "writer.palmyra-x5-v1:0"
+
+ base_model = bedrock_model_info.get_base_model(
+ model="bedrock/us.writer.palmyra-x4-v1:0"
+ )
+ assert base_model == "writer.palmyra-x4-v1:0"
diff --git a/tests/test_litellm/llms/openai/chat/test_openai_gpt_transformation.py b/tests/test_litellm/llms/openai/chat/test_openai_gpt_transformation.py
new file mode 100644
index 00000000000..5f087363797
--- /dev/null
+++ b/tests/test_litellm/llms/openai/chat/test_openai_gpt_transformation.py
@@ -0,0 +1,125 @@
+"""
+Tests for OpenAI GPT transformation (litellm/llms/openai/chat/gpt_transformation.py)
+"""
+
+import pytest
+import sys
+import os
+
+sys.path.insert(0, os.path.abspath("../../../../.."))
+
+from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
+
+
+class TestOpenAIGPTConfig:
+ """Tests for OpenAIGPTConfig class"""
+
+ def setup_method(self):
+ self.config = OpenAIGPTConfig()
+
+ def test_user_param_supported_for_regular_models(self):
+ """Test that 'user' param is in supported params for regular OpenAI models."""
+ supported_params = self.config.get_supported_openai_params("gpt-4o")
+ assert "user" in supported_params
+
+ supported_params = self.config.get_supported_openai_params("gpt-4.1-mini")
+ assert "user" in supported_params
+
+ def test_user_param_supported_for_responses_api_models(self):
+ """Test that 'user' param is in supported params for responses API models.
+
+ Regression test for: https://github.com/BerriAI/litellm/issues/17633
+ When using model="openai/responses/gpt-4.1", the 'user' parameter should
+ be included in supported params so it reaches OpenAI and SpendLogs.
+ """
+ # responses/gpt-4.1-mini should support 'user' just like gpt-4.1-mini
+ supported_params = self.config.get_supported_openai_params("responses/gpt-4.1-mini")
+ assert "user" in supported_params
+
+ supported_params = self.config.get_supported_openai_params("responses/gpt-4o")
+ assert "user" in supported_params
+
+ supported_params = self.config.get_supported_openai_params("responses/gpt-4.1")
+ assert "user" in supported_params
+
+ def test_model_normalization_for_responses_prefix(self):
+ """Test that models with 'responses/' prefix are normalized correctly.
+
+ The fix normalizes 'responses/gpt-4.1' to 'gpt-4.1' when checking
+ if the model is in the list of supported OpenAI models.
+ """
+ # Both should have the same supported params
+ regular_params = self.config.get_supported_openai_params("gpt-4.1-mini")
+ responses_params = self.config.get_supported_openai_params("responses/gpt-4.1-mini")
+
+ # 'user' should be in both
+ assert "user" in regular_params
+ assert "user" in responses_params
+
+ def test_base_params_always_included(self):
+ """Test that base params are always included regardless of model."""
+ base_expected_params = [
+ "frequency_penalty",
+ "max_tokens",
+ "temperature",
+ "top_p",
+ "stream",
+ "tools",
+ "tool_choice",
+ ]
+
+ supported_params = self.config.get_supported_openai_params("responses/gpt-4.1-mini")
+
+ for param in base_expected_params:
+ assert param in supported_params, f"Expected '{param}' in supported params"
+
+
+class TestGetOptionalParamsIntegration:
+ """Integration tests using litellm.get_optional_params()"""
+
+ def test_user_in_optional_params_for_responses_model(self):
+ """Test that 'user' ends up in optional_params when using responses API models.
+
+ Regression test for: https://github.com/BerriAI/litellm/issues/17633
+ This verifies the full flow through get_optional_params().
+ """
+ from litellm.utils import get_optional_params
+
+ # Test with responses model
+ optional_params = get_optional_params(
+ model="responses/gpt-4.1-mini",
+ custom_llm_provider="openai",
+ user="test-user-123",
+ )
+ assert optional_params.get("user") == "test-user-123"
+
+ def test_user_in_optional_params_for_regular_model(self):
+ """Test that 'user' ends up in optional_params for regular OpenAI models."""
+ from litellm.utils import get_optional_params
+
+ optional_params = get_optional_params(
+ model="gpt-4o",
+ custom_llm_provider="openai",
+ user="test-user-456",
+ )
+ assert optional_params.get("user") == "test-user-456"
+
+ def test_user_param_consistency_between_regular_and_responses(self):
+ """Test that 'user' param behavior is consistent between regular and responses models."""
+ from litellm.utils import get_optional_params
+
+ regular_params = get_optional_params(
+ model="gpt-4.1-mini",
+ custom_llm_provider="openai",
+ user="my-end-user",
+ )
+
+ responses_params = get_optional_params(
+ model="responses/gpt-4.1-mini",
+ custom_llm_provider="openai",
+ user="my-end-user",
+ )
+
+ # Both should include user
+ assert regular_params.get("user") == "my-end-user"
+ assert responses_params.get("user") == "my-end-user"
diff --git a/tests/test_litellm/llms/sap/chat/test_sap_chat_calls.py b/tests/test_litellm/llms/sap/chat/test_sap_chat_calls.py
new file mode 100644
index 00000000000..3984bba27fa
--- /dev/null
+++ b/tests/test_litellm/llms/sap/chat/test_sap_chat_calls.py
@@ -0,0 +1,142 @@
+import httpx
+from unittest.mock import patch, PropertyMock
+
+import pytest
+
+mock_response = {
+ "request_id": "e86a0b4e-53e3-97dc-a5f7-82e451376b23",
+ "intermediate_results": {
+ "templating": [{"content": "Say hello", "role": "user"}],
+ "llm": {
+ "id": "chatcmpl-CUB63bLTYnfO2CQR0r0rArkrbe8CH",
+ "object": "chat.completion",
+ "created": 1761308531,
+ "model": "gpt-4o-2024-08-06",
+ "system_fingerprint": "fp_4a331a0222",
+ "choices": [
+ {
+ "index": 0,
+ "message": {"role": "assistant", "content": "Hello from SAP!"},
+ "finish_reason": "stop",
+ }
+ ],
+ "usage": {"completion_tokens": 7, "prompt_tokens": 3, "total_tokens": 10},
+ },
+ },
+ "final_result": {
+ "id": "chatcmpl-CUB63bLTYnfO2CQR0r0rArkrbe8CH",
+ "object": "chat.completion",
+ "created": 1761308531,
+ "model": "gpt-4o-2024-08-06",
+ "system_fingerprint": "fp_4a331a0222",
+ "choices": [
+ {
+ "index": 0,
+ "message": {"role": "assistant", "content": "Hello from SAP!"},
+ "finish_reason": "stop",
+ }
+ ],
+ "usage": {"completion_tokens": 7, "prompt_tokens": 3, "total_tokens": 10},
+ },
+}
+mock_stream_response = [
+ b'data: {"request_id": "a07127d3-cb74-9427-a4dc-ef9bf424fb43", "intermediate_results": {"templating": [{"content": "Hi", "role": "user"}]}, "final_result": {"id": \'\', "object": \'\', "created": 0, "model": \'\', "system_fingerprint": null, "choices": [{"index": 0, "delta": {"content": ""}}]}}\n\n',
+ b'data: {"request_id": "a07127d3-cb74-9427-a4dc-ef9bf424fb43", "intermediate_results": {"llm": {"id": "chatcmpl-HelloMsg", "object": "chat.completion.chunk", "created": 1761319270, "model": "gpt-4o-2024-08-06", "system_fingerprint": "fp_HelloMsg", "choices": [{"index": 0, "delta": {"role": "assistant", "content": "Hello "}}]}}, "final_result": {"id": "chatcmpl-HelloMsg", "object": "chat.completion.chunk", "created": 1761319270, "model": "gpt-4o-2024-08-06", "system_fingerprint": "fp_HelloMsg", "choices": [{"index": 0, "delta": {"role": "assistant", "content": "Hello "}}]}}\n\n',
+ b'data: {"request_id": "a07127d3-cb74-9427-a4dc-ef9bf424fb43", "intermediate_results": {"llm": {"id": "chatcmpl-CUDtFmLex96SxakzBIzhLq2h8Axmk", "object": "chat.completion.chunk", "created": 1761319269, "model": "gpt-4o-2024-08-06", "system_fingerprint": "fp_4a331a0222", "choices": [{"index": 0, "delta": {"role": "assistant", "content": "from SAP!"}, "finish_reason": "stop"}]}}, "final_result": {"id": "chatcmpl-CUDtFmLex96SxakzBIzhLq2h8Axmk", "object": "chat.completion.chunk", "created": 1761319269, "model": "gpt-4o-2024-08-06", "system_fingerprint": "fp_4a331a0222", "choices": [{"index": 0, "delta": {"role": "assistant", "content": "from SAP!"}, "finish_reason": "stop"}]}}\n\n',
+ b"data: [DONE]\n\n",
+]
+
+
+@pytest.fixture
+def sap_api_response():
+ return mock_response
+
+
+@pytest.fixture
+def sap_api_stream_response():
+ return mock_response
+
+
+@pytest.fixture
+def fake_token_creator():
+ return lambda: "Bearer FAKE_TOKEN", "https://api.ai.mock-sap.com", "fake-group"
+
+
+@pytest.fixture
+def fake_deployment_url():
+ return "https://api.ai.mock-sap.com/v2/inference/deployments/mockid"
+
+
+@pytest.mark.parametrize("sync_mode", [True, False])
+@pytest.mark.asyncio
+async def test_sap_chat(
+ respx_mock,
+ sap_api_response,
+ fake_token_creator,
+ fake_deployment_url,
+ sync_mode,
+):
+ import litellm
+
+ litellm.disable_aiohttp_transport = True
+ with patch(
+ "litellm.llms.sap.chat.transformation.GenAIHubOrchestrationConfig.deployment_url",
+ new_callable=PropertyMock,
+ return_value=fake_deployment_url,
+ ), patch(
+ "litellm.llms.sap.chat.transformation.get_token_creator",
+ return_value=fake_token_creator,
+ ):
+ model = "sap/gpt-4o"
+ messages = [{"role": "user", "content": "Hello"}]
+ respx_mock.post(f"{fake_deployment_url}/v2/completion").respond(
+ json=sap_api_response
+ )
+
+ if sync_mode:
+ response = litellm.completion(model=model, messages=messages)
+ else:
+ response = await litellm.acompletion(model=model, messages=messages)
+
+ assert response.choices[0].message.content == "Hello from SAP!"
+ assert response.model.startswith("gpt-4o")
+ assert response.usage.total_tokens == 10
+
+
+@pytest.mark.asyncio
+async def test_sap_streaming(
+ respx_mock,
+ sap_api_stream_response,
+ fake_token_creator,
+ fake_deployment_url,
+):
+ import litellm
+
+ litellm.disable_aiohttp_transport = True
+ with patch(
+ "litellm.llms.sap.chat.transformation.GenAIHubOrchestrationConfig.deployment_url",
+ new_callable=PropertyMock,
+ return_value=fake_deployment_url,
+ ), patch(
+ "litellm.llms.sap.chat.transformation.get_token_creator",
+ return_value=fake_token_creator,
+ ):
+ model = "sap/gpt-4o"
+ messages = [{"role": "user", "content": "Hello"}]
+
+ respx_mock.post(f"{fake_deployment_url}/v2/completion").mock(
+ return_value=httpx.Response(
+ 200,
+ content=mock_stream_response,
+ headers={"Content-Type": "text/event-stream"},
+ )
+ )
+
+ stream = litellm.completion(model=model, messages=messages, stream=True)
+
+ full = ""
+ for chunk in stream:
+ delta = getattr(chunk.choices[0].delta, "content", None) or ""
+ full += delta
+
+ assert full == "Hello from SAP!"
diff --git a/tests/test_litellm/llms/sap/embed/test_sap_embedding.py b/tests/test_litellm/llms/sap/embed/test_sap_embedding.py
new file mode 100644
index 00000000000..617740bb43f
--- /dev/null
+++ b/tests/test_litellm/llms/sap/embed/test_sap_embedding.py
@@ -0,0 +1,1607 @@
+import httpx
+from unittest.mock import patch, PropertyMock
+
+import pytest
+
+moke_response = {
+ "request_id": "9c18627f-ffce-9264-b441-e1f8967d5085",
+ "final_result": {
+ "object": "list",
+ "data": [
+ {
+ "object": "embedding",
+ "embedding": [
+ -0.0069594960659742355,
+ -0.035274259746074677,
+ 0.0015957315918058157,
+ 0.06534460932016373,
+ 0.03293841332197189,
+ -0.024201158434152603,
+ -0.02610827423632145,
+ 0.04937804862856865,
+ 0.01623266376554966,
+ -0.05168433114886284,
+ -0.013357206247746944,
+ -0.014599049463868141,
+ -0.026019571349024773,
+ -0.003257990349084139,
+ 0.024585537612438202,
+ 0.001171619864180684,
+ -0.05345839262008667,
+ 0.015057348646223545,
+ 0.011487049050629139,
+ 0.03394371271133423,
+ 0.04934848099946976,
+ 0.020372141152620316,
+ -0.01396334357559681,
+ 0.01887897402048111,
+ 0.017149262130260468,
+ 0.024156806990504265,
+ 0.01827283576130867,
+ -0.0011956436792388558,
+ 0.01955902948975563,
+ -0.03678221255540848,
+ 0.027675362303853035,
+ -0.028207581490278244,
+ 0.027645794674754143,
+ -0.01623266376554966,
+ -0.011716199107468128,
+ -0.01604047417640686,
+ -0.01407422311604023,
+ 0.03758053854107857,
+ 0.01887897402048111,
+ -0.037698812782764435,
+ 0.04343494400382042,
+ -0.012411040253937244,
+ 0.020948711782693863,
+ 0.013556787744164467,
+ 0.0019680997356772423,
+ 0.0002180617448175326,
+ -0.049171075224876404,
+ 0.00832330621778965,
+ 0.018331971019506454,
+ 0.029464207589626312,
+ -0.02678833156824112,
+ 0.007635856978595257,
+ 0.025014270097017288,
+ 0.10638456791639328,
+ 0.03624999523162842,
+ -0.010289558209478855,
+ 0.05954933911561966,
+ 0.02360980398952961,
+ -0.0040766457095742226,
+ 0.00014760748308617622,
+ -0.019056379795074463,
+ 0.009683419950306416,
+ 0.008338090032339096,
+ 0.004091429989784956,
+ -0.00712211849167943,
+ -0.013593747280538082,
+ -0.02390548214316368,
+ 0.012167106382548809,
+ -0.020933927968144417,
+ -0.013837681151926517,
+ -0.0013979976065456867,
+ 0.036220427602529526,
+ -0.028665879741311073,
+ -0.007184949703514576,
+ -0.00439819460734725,
+ -0.03018861636519432,
+ -0.08533236384391785,
+ -0.03775794804096222,
+ -0.0012566270306706429,
+ 0.0097425552085042,
+ -0.02742403745651245,
+ 0.02217577025294304,
+ -0.03710745647549629,
+ -0.011937957257032394,
+ -0.05629689246416092,
+ -0.006068769376724958,
+ -0.08083807677030563,
+ 0.016350936144590378,
+ -0.026684844866394997,
+ -0.00926947221159935,
+ -0.0157004464417696,
+ 0.04420370236039162,
+ -0.02841455489397049,
+ -0.02360980398952961,
+ 0.009350783191621304,
+ 0.01164967194199562,
+ -0.024733377620577812,
+ -0.0011577600380405784,
+ 0.059076253324747086,
+ -0.009380350820720196,
+ 0.03787621855735779,
+ -0.029375504702329636,
+ 0.03967984765768051,
+ -0.010548274964094162,
+ 0.011538793332874775,
+ 0.03926589712500572,
+ 0.008456360548734665,
+ -0.011228332296013832,
+ -0.053133148699998856,
+ 0.025590840727090836,
+ -0.06599509716033936,
+ -0.07817698270082474,
+ -0.0011134084779769182,
+ 0.010112151503562927,
+ 0.008959011174738407,
+ 0.04059644415974617,
+ 0.015138659626245499,
+ -0.05739089474081993,
+ -0.00017786816169973463,
+ -0.0665864497423172,
+ 0.003707049647346139,
+ -0.004886061418801546,
+ 0.02834063582122326,
+ -0.036604806780815125,
+ -0.06085031479597092,
+ 0.02111133374273777,
+ 0.0011392802698537707,
+ 0.016883153468370438,
+ -0.04736744612455368,
+ 0.0036054106894880533,
+ 0.05073816329240799,
+ 0.015463904477655888,
+ -0.02579781413078308,
+ -0.0072219097055494785,
+ -0.029863372445106506,
+ 0.03672307729721069,
+ -0.03749183565378189,
+ 0.028311068192124367,
+ -0.043789755553007126,
+ -0.022530583664774895,
+ 0.03128262236714363,
+ -0.008818564936518669,
+ 0.005717652849853039,
+ -0.015907419845461845,
+ 0.022027932107448578,
+ -0.015005605295300484,
+ 0.0012538550654426217,
+ 0.06705953180789948,
+ -0.028089310973882675,
+ -0.015389985404908657,
+ 0.027069224044680595,
+ 0.021820958703756332,
+ -0.052305251359939575,
+ 0.02448205091059208,
+ 0.012751067988574505,
+ -0.029523342847824097,
+ 0.015042564831674099,
+ -0.029464207589626312,
+ -0.023846345022320747,
+ 0.008545063436031342,
+ 0.0332932248711586,
+ 0.016099609434604645,
+ 0.01224102545529604,
+ -0.0526009276509285,
+ -0.0389702208340168,
+ 0.01159792859107256,
+ 0.028399771079421043,
+ 0.03678221255540848,
+ -0.032820142805576324,
+ -0.0012483111349865794,
+ -0.024866431951522827,
+ 0.029272018000483513,
+ -0.03234705701470375,
+ -0.007953709922730923,
+ -0.012654973194003105,
+ -0.005244569852948189,
+ 0.009299039840698242,
+ 0.00017197772103827447,
+ -0.0684787780046463,
+ -0.017962373793125153,
+ 0.016365719959139824,
+ 0.09745512157678604,
+ 0.005403496325016022,
+ 0.005754612386226654,
+ -0.032820142805576324,
+ -0.020593900233507156,
+ -0.011923172511160374,
+ 0.005255657713860273,
+ 0.021318307146430016,
+ 0.04334624111652374,
+ 0.000568716146517545,
+ 0.061914753168821335,
+ 0.004032294265925884,
+ 0.005163258872926235,
+ -0.006113120820373297,
+ -0.044321972876787186,
+ 0.0809563472867012,
+ -0.007366051897406578,
+ -0.005285225342959166,
+ -0.002382047474384308,
+ 0.021880093961954117,
+ -0.05535072460770607,
+ 0.01717883162200451,
+ -0.014488170854747295,
+ -0.024526402354240417,
+ -0.021273955702781677,
+ 0.022220123559236526,
+ 0.058011818677186966,
+ -0.00015661638462916017,
+ -0.04816577583551407,
+ 0.05248265713453293,
+ -0.03382544219493866,
+ 0.0070481994189321995,
+ 0.030129481106996536,
+ -0.013379381969571114,
+ -0.034712474793195724,
+ 0.045002032071352005,
+ 0.002792299259454012,
+ 0.049998972564935684,
+ 0.012329728342592716,
+ -0.009409918449819088,
+ 0.002725771861150861,
+ 0.06226956471800804,
+ 0.034180253744125366,
+ 0.021850526332855225,
+ 0.017844103276729584,
+ -0.013083704747259617,
+ -0.01316501572728157,
+ 0.015449120663106441,
+ -0.03420982137322426,
+ 0.02232361026108265,
+ 0.04923021048307419,
+ -0.047722261399030685,
+ -0.04656912013888359,
+ 0.019987761974334717,
+ 0.022841043770313263,
+ 0.030366022139787674,
+ -0.01254409458488226,
+ 0.016395287588238716,
+ -0.01499821338802576,
+ -0.03382544219493866,
+ 0.006460541393607855,
+ 0.0006504892953671515,
+ 0.02773449756205082,
+ 0.022160986438393593,
+ -0.00404707808047533,
+ 0.008655942976474762,
+ -0.06504892557859421,
+ 0.013194584287703037,
+ 0.015463904477655888,
+ 0.008582023903727531,
+ 0.010548274964094162,
+ 0.013401557691395283,
+ -0.022190555930137634,
+ -0.02746838890016079,
+ 0.02587173320353031,
+ 0.003599866759032011,
+ 0.03867454454302788,
+ -0.02485164813697338,
+ -0.04050774127244949,
+ -0.023698506876826286,
+ -0.03734399750828743,
+ -0.006582507863640785,
+ -0.04444024711847305,
+ -0.055942077189683914,
+ -0.042429640889167786,
+ 0.01164967194199562,
+ 0.03125305473804474,
+ -0.013926384039223194,
+ 0.005806356202811003,
+ -0.003505619941279292,
+ -0.026418736204504967,
+ 0.03536296263337135,
+ -0.010089975781738758,
+ -0.006885576993227005,
+ 0.015049956738948822,
+ 0.020534764975309372,
+ -0.016557909548282623,
+ -0.00034349344787187874,
+ 0.01642485521733761,
+ -0.046628255397081375,
+ -0.023713290691375732,
+ -0.006349662318825722,
+ 0.0355699360370636,
+ -0.06853791326284409,
+ 0.020194735378026962,
+ -0.009727771393954754,
+ -0.03456463664770126,
+ 0.03426895663142204,
+ 0.029567694291472435,
+ 0.03589517995715141,
+ 0.0104965316131711,
+ -0.01106570940464735,
+ -0.03344106301665306,
+ 0.002962313359603286,
+ 0.01793280616402626,
+ -0.006582507863640785,
+ -0.022146202623844147,
+ 0.023698506876826286,
+ -0.032406192272901535,
+ 0.0714946836233139,
+ 0.014842982403934002,
+ 0.009380350820720196,
+ -0.0008722469792701304,
+ -0.00832330621778965,
+ 0.028843285515904427,
+ 0.0036035627126693726,
+ -0.031164349988102913,
+ 0.008205035701394081,
+ 0.006678603123873472,
+ -0.035185556858778,
+ 0.014029871672391891,
+ 0.0348011776804924,
+ -0.02807452715933323,
+ -0.04103996232151985,
+ 0.003487139940261841,
+ -0.0004827850207220763,
+ -0.014658184722065926,
+ 0.002358023775741458,
+ -0.04420370236039162,
+ 0.0064790211617946625,
+ -0.04639171436429024,
+ 0.027571875602006912,
+ -0.01717883162200451,
+ 0.0006398633704520762,
+ -0.021022630855441093,
+ 0.06085031479597092,
+ 0.008648551069200039,
+ -0.01808064617216587,
+ -0.011516617611050606,
+ -0.0010561210801824927,
+ -0.023668939247727394,
+ 0.003780968952924013,
+ -0.007377139758318663,
+ 0.01914508268237114,
+ 0.007894574664533138,
+ -0.0431392677128315,
+ 0.02356545254588127,
+ -0.05224611610174179,
+ 0.05091556906700134,
+ -0.024807296693325043,
+ -0.05008767545223236,
+ -0.05836663022637367,
+ 0.010696114040911198,
+ 0.00684861745685339,
+ -0.010681330226361752,
+ 0.004294707905501127,
+ -0.010607410222291946,
+ -0.01737102121114731,
+ -0.008825956843793392,
+ 0.03663437440991402,
+ 0.03690048307180405,
+ -0.015996122732758522,
+ -0.025901300832629204,
+ -0.02072695456445217,
+ -0.03616129234433174,
+ 0.07108073681592941,
+ -0.0029087220318615437,
+ -0.01164967194199562,
+ 0.05008767545223236,
+ -0.05656300112605095,
+ 0.00023573306680191308,
+ 0.0010182374389842153,
+ -0.0366935096681118,
+ 0.05357666313648224,
+ 0.03208094835281372,
+ -0.060554638504981995,
+ 0.004283619578927755,
+ 0.020800873637199402,
+ 0.05103383958339691,
+ -0.013172407634556293,
+ 0.013652883470058441,
+ 0.001349950092844665,
+ 0.011775334365665913,
+ -0.04101039096713066,
+ 0.023121938109397888,
+ 0.03861540928483009,
+ -0.005329576786607504,
+ 0.011487049050629139,
+ -0.0016095914179459214,
+ 0.019455542787909508,
+ 0.05064946040511131,
+ 0.017563210800290108,
+ -0.028577176854014397,
+ 0.057302191853523254,
+ -0.007384531665593386,
+ 0.025014270097017288,
+ -0.038763247430324554,
+ -0.030957376584410667,
+ 0.015833500772714615,
+ 0.07805871218442917,
+ 0.0031988550908863544,
+ 0.042547911405563354,
+ -0.02356545254588127,
+ 0.018361538648605347,
+ -0.02035735733807087,
+ 0.03604302182793617,
+ 0.02273755706846714,
+ -0.009609500877559185,
+ -0.012204065918922424,
+ 0.021880093961954117,
+ -0.06034766510128975,
+ 0.010393044911324978,
+ 0.009180769324302673,
+ -0.01244799979031086,
+ -0.019381623715162277,
+ -0.04937804862856865,
+ 0.01159792859107256,
+ 0.020327789708971977,
+ -0.016025690361857414,
+ 0.015508255921304226,
+ -0.048638857901096344,
+ 0.05688824504613876,
+ 0.02278190851211548,
+ 0.027039656415581703,
+ -0.028311068192124367,
+ -0.013231543824076653,
+ -0.00849332008510828,
+ 0.015153443440794945,
+ 0.009587325155735016,
+ 0.012181890197098255,
+ 0.00012681768566835672,
+ -0.01982514001429081,
+ 0.025590840727090836,
+ -0.017193615436553955,
+ 0.046480417251586914,
+ 0.04316883534193039,
+ -0.06386622041463852,
+ 0.013918992131948471,
+ -0.07267739623785019,
+ -0.02300366573035717,
+ 0.01127268373966217,
+ -0.01212275493890047,
+ 0.040537308901548386,
+ -0.04044860601425171,
+ -0.012854555621743202,
+ 0.006870793178677559,
+ 0.038763247430324554,
+ 0.023846345022320747,
+ 0.023639371618628502,
+ -0.005876579321920872,
+ -0.007872398942708969,
+ -0.008382441475987434,
+ -0.039709415286779404,
+ -0.010016056708991528,
+ -0.04423326998949051,
+ -0.0039731590077281,
+ -0.007133206352591515,
+ -0.0228853952139616,
+ -0.018642431125044823,
+ -0.014865159057080746,
+ -0.0026592444628477097,
+ 0.007961101830005646,
+ 0.003590626874938607,
+ 0.006804266013205051,
+ 0.0219392292201519,
+ 0.010866127908229828,
+ -0.009956921450793743,
+ 0.00272392388433218,
+ -0.029419856145977974,
+ 0.024733377620577812,
+ -0.0003227036795578897,
+ 0.04760398715734482,
+ 0.035510800778865814,
+ 0.023624587804079056,
+ -0.03477161005139351,
+ 0.0021954013500362635,
+ -0.007717168424278498,
+ -0.022220123559236526,
+ -0.07575243711471558,
+ 0.02936072088778019,
+ 0.01184186153113842,
+ 0.04748571664094925,
+ -0.009299039840698242,
+ -0.014887334778904915,
+ -0.04003465920686722,
+ 0.0020715866703540087,
+ -0.019958194345235825,
+ 0.00602441793307662,
+ -0.015183011069893837,
+ 0.004202308598905802,
+ -0.006094641052186489,
+ -0.03423938900232315,
+ 0.06611336767673492,
+ -0.010245205834507942,
+ 0.07238171994686127,
+ 0.02440813183784485,
+ 0.021096549928188324,
+ -0.04399672895669937,
+ -0.034978583455085754,
+ 0.01963294856250286,
+ 0.019854707643389702,
+ 0.08704729378223419,
+ -0.01842067390680313,
+ -0.029183315113186836,
+ -0.015759581699967384,
+ 0.0019514678278937936,
+ -0.003065800294280052,
+ -0.0404781736433506,
+ -0.051506925374269485,
+ -0.015345633961260319,
+ 0.008293738588690758,
+ -0.008160683326423168,
+ 0.0518321692943573,
+ 0.04177915304899216,
+ -0.03879281505942345,
+ -0.03249489516019821,
+ 0.012551486492156982,
+ -0.012285376898944378,
+ -0.015049956738948822,
+ 0.006774697918444872,
+ 0.04523857310414314,
+ -0.012573662213981152,
+ -0.056503865867853165,
+ -0.024659456685185432,
+ -0.0035887788981199265,
+ -0.0016465509543195367,
+ -0.006275743246078491,
+ 0.02448205091059208,
+ -0.03420982137322426,
+ 0.02569432742893696,
+ 0.006767306011170149,
+ -0.025413433089852333,
+ -0.0068190498277544975,
+ 0.005026508122682571,
+ -0.016469206660985947,
+ -0.011494440957903862,
+ -0.031223485246300697,
+ -0.005362840835005045,
+ 0.0019662517588585615,
+ 0.038911085575819016,
+ -0.017415372654795647,
+ -0.032406192272901535,
+ 0.005717652849853039,
+ -0.028784150257706642,
+ 0.009609500877559185,
+ -0.005359144881367683,
+ -0.04438111186027527,
+ 0.0003402594884391874,
+ -0.02871023118495941,
+ 0.03435766324400902,
+ 0.006205520126968622,
+ 0.013054137118160725,
+ 0.011701415292918682,
+ -0.00207528262399137,
+ 0.024955134838819504,
+ 0.03314538672566414,
+ -0.0056733014062047005,
+ 0.041335638612508774,
+ 0.04183828830718994,
+ -0.01967730186879635,
+ -0.030957376584410667,
+ 0.04441067948937416,
+ -0.010200854390859604,
+ 0.021007847040891647,
+ -0.01846502535045147,
+ -0.025265594944357872,
+ -0.004475809633731842,
+ 0.009905178099870682,
+ 0.019470326602458954,
+ 0.00424666004255414,
+ -0.002463358687236905,
+ 0.013652883470058441,
+ 0.007236693520098925,
+ 0.0006560332258231938,
+ 0.03412111848592758,
+ -0.009417311288416386,
+ -0.028237149119377136,
+ 0.005802660249173641,
+ -0.023506317287683487,
+ -0.016217879951000214,
+ -0.008759429678320885,
+ -0.028917206451296806,
+ -0.01110266987234354,
+ 0.008655942976474762,
+ -0.015227362513542175,
+ -0.0034353965893387794,
+ 0.010385653004050255,
+ 0.0355699360370636,
+ 0.0097425552085042,
+ -0.024393348023295403,
+ -0.022427096962928772,
+ -0.008811173029243946,
+ -0.03317495435476303,
+ -0.023624587804079056,
+ 0.001332394196651876,
+ -0.010740465484559536,
+ 0.027971038594841957,
+ 0.02958247810602188,
+ 0.03057299740612507,
+ 0.016927504912018776,
+ -2.5496361558907665e-05,
+ -0.001735254074446857,
+ 0.027246631681919098,
+ 0.011627496220171452,
+ -0.004120997618883848,
+ -0.021421795710921288,
+ -0.045800358057022095,
+ -0.034328095614910126,
+ 0.004265139810740948,
+ 0.0019015723373740911,
+ -0.015404769219458103,
+ -0.0014174013631418347,
+ -0.06652731448411942,
+ 0.01269193273037672,
+ -0.0037698810920119286,
+ 0.0025964132510125637,
+ 0.02239752933382988,
+ -0.01686836965382099,
+ -0.023920265957713127,
+ -0.0007895498420111835,
+ 0.04089212045073509,
+ 0.011509224772453308,
+ -0.04130607098340988,
+ -0.03305668383836746,
+ 0.02300366573035717,
+ -0.001791617483831942,
+ 0.026832683011889458,
+ 0.016291799023747444,
+ -0.01284716371446848,
+ -0.015449120663106441,
+ -0.035540368407964706,
+ 0.007299524731934071,
+ 0.03772838041186333,
+ 0.03962071239948273,
+ 0.02122960425913334,
+ -0.023062802851200104,
+ -0.026049138978123665,
+ -0.034978583455085754,
+ 0.002077130600810051,
+ 0.019839923828840256,
+ 0.024378564208745956,
+ 0.023994185030460358,
+ -0.013660275377333164,
+ -0.027290983125567436,
+ -0.003294949885457754,
+ -0.006741434335708618,
+ -0.010223030112683773,
+ 0.015996122732758522,
+ -0.03923632949590683,
+ 0.010577842593193054,
+ -0.0032986460719257593,
+ 0.01633615233004093,
+ -0.000837135361507535,
+ 0.034150686115026474,
+ -0.0003019138821400702,
+ 0.005322184879332781,
+ 0.014421642757952213,
+ 0.011161805130541325,
+ -0.018967676907777786,
+ 0.0025760855060070753,
+ -0.04444024711847305,
+ -0.004697567317634821,
+ -0.01618831232190132,
+ -0.0034446364734321833,
+ -0.0031304797157645226,
+ -0.04030076786875725,
+ -0.006131600588560104,
+ -0.01237408071756363,
+ 0.005684389267116785,
+ 0.00957254134118557,
+ -0.009262080304324627,
+ -0.017297102138400078,
+ -0.018775485455989838,
+ -0.011021357960999012,
+ 0.009143809787929058,
+ 0.013128056190907955,
+ 0.004789966624230146,
+ -0.014983429573476315,
+ -0.021022630855441093,
+ 0.01349026057869196,
+ 0.002108546206727624,
+ 0.03690048307180405,
+ -0.028473690152168274,
+ 0.04778139665722847,
+ 0.005211306270211935,
+ 0.03988682106137276,
+ -0.0507085956633091,
+ -0.019396407529711723,
+ -0.03438723087310791,
+ -0.016705747693777084,
+ 0.005100427195429802,
+ -0.017696265131235123,
+ -0.016779666766524315,
+ 0.00019877344311680645,
+ 0.030336454510688782,
+ 0.03137132525444031,
+ -0.009284256026148796,
+ 0.003651610342785716,
+ 0.02826671674847603,
+ 0.00027858311659656465,
+ 0.009535581804811954,
+ 0.02965639717876911,
+ -0.01899724453687668,
+ 0.003202551044523716,
+ -0.03698918595910072,
+ 0.045800358057022095,
+ -0.025339514017105103,
+ 0.024940351024270058,
+ -0.07770390063524246,
+ 0.0022027932573109865,
+ -0.02807452715933323,
+ -0.016321366652846336,
+ 0.0020309309475123882,
+ -0.0023266079369932413,
+ 0.0060133300721645355,
+ -0.029715532436966896,
+ 0.018095429986715317,
+ -0.0025465176440775394,
+ 0.02579781413078308,
+ -0.02414202317595482,
+ 0.01660226099193096,
+ -0.017755400389432907,
+ -0.008404617197811604,
+ 0.04535684362053871,
+ 0.02455596998333931,
+ -0.013734194450080395,
+ -0.02943463996052742,
+ 0.022707989439368248,
+ -0.02183574251830578,
+ -0.010548274964094162,
+ -0.02397940121591091,
+ -0.02307758666574955,
+ -0.014369899407029152,
+ 0.01895289309322834,
+ -0.031075647100806236,
+ -0.03829016536474228,
+ 0.012100579217076302,
+ 0.0541088804602623,
+ 0.01244799979031086,
+ -0.012876731343567371,
+ 0.00900336354970932,
+ 0.013283287174999714,
+ 0.027897119522094727,
+ 0.03450550138950348,
+ 0.002345087705180049,
+ -0.031460028141736984,
+ -0.038763247430324554,
+ -0.020564332604408264,
+ -0.0597858801484108,
+ -0.0011956436792388558,
+ 0.012004484422504902,
+ 0.020401708781719208,
+ -0.004261443857103586,
+ 0.014347723685204983,
+ -0.02397940121591091,
+ 0.04166088253259659,
+ 0.04151304438710213,
+ -0.024053320288658142,
+ -0.006216607987880707,
+ 0.019115515053272247,
+ 0.012078403495252132,
+ -0.02220533974468708,
+ 0.012381472624838352,
+ 0.00907728262245655,
+ -0.027113575488328934,
+ 0.03766924515366554,
+ -0.0183467548340559,
+ 0.043494079262018204,
+ 0.01822848431766033,
+ 0.02281147614121437,
+ 0.03589517995715141,
+ -0.012884123250842094,
+ -0.016897937282919884,
+ -0.030055562034249306,
+ -0.012004484422504902,
+ -0.03645696863532066,
+ -0.018893757835030556,
+ -0.038231030106544495,
+ -0.012876731343567371,
+ 0.01822848431766033,
+ -0.019795572385191917,
+ -0.001042261254042387,
+ 0.003895543748512864,
+ 0.016853585839271545,
+ -0.01611439324915409,
+ -0.007768911775201559,
+ -0.012943258509039879,
+ -0.03571777418255806,
+ 0.019307704642415047,
+ 0.004956285003572702,
+ 0.026389168575406075,
+ 0.033766306936740875,
+ -0.00029914191691204906,
+ -0.02077130600810051,
+ -0.04707176983356476,
+ -0.008648551069200039,
+ 0.0019828835502266884,
+ -0.002997425151988864,
+ -0.007983277551829815,
+ -0.00564742973074317,
+ -0.03574734181165695,
+ 0.02671441249549389,
+ -0.028059743344783783,
+ 0.007480626925826073,
+ 0.02477772906422615,
+ 0.010356085374951363,
+ -0.019869491457939148,
+ 0.0202390868216753,
+ 0.00332451774738729,
+ -0.009372958913445473,
+ -0.021170469000935555,
+ 0.020712170749902725,
+ 0.018095429986715317,
+ -0.004017510451376438,
+ -0.002457814523950219,
+ -0.028798934072256088,
+ -0.008448968641459942,
+ -0.006527068559080362,
+ -0.008264170959591866,
+ -0.013113272376358509,
+ -0.00602441793307662,
+ -0.010577842593193054,
+ 0.007665425073355436,
+ 0.0021621377673000097,
+ -0.010940046980977058,
+ 0.011265291832387447,
+ -0.043257538229227066,
+ 0.013667667284607887,
+ 0.022027932107448578,
+ 0.04801793769001961,
+ 0.04834318161010742,
+ -0.015389985404908657,
+ -0.036870915442705154,
+ -0.0021418097894638777,
+ 0.026507439091801643,
+ 0.01975122094154358,
+ 0.000411175744375214,
+ 0.014000303111970425,
+ -0.04077384993433952,
+ 0.01401508692651987,
+ -0.03503771871328354,
+ -0.012980218045413494,
+ -0.02618219330906868,
+ 0.005100427195429802,
+ 0.05328098684549332,
+ 0.00936556700617075,
+ -0.01139095425605774,
+ -0.01556739117950201,
+ -0.033500198274850845,
+ 0.02258971892297268,
+ -0.009587325155735016,
+ 0.030025994405150414,
+ 0.003507467918097973,
+ 0.01604047417640686,
+ 0.029833804816007614,
+ -0.009321215562522411,
+ -0.01096961461007595,
+ -0.01728231832385063,
+ 2.100634628732223e-05,
+ -0.011775334365665913,
+ 0.007207125425338745,
+ 0.017829319462180138,
+ -0.02470380999147892,
+ 0.0017093823989853263,
+ -0.003806840628385544,
+ -0.02780841663479805,
+ 0.018331971019506454,
+ 0.02603435516357422,
+ -0.010962222702801228,
+ -0.04056687653064728,
+ 0.03775794804096222,
+ 0.03110521472990513,
+ 4.5015662180958316e-05,
+ 0.0038179284892976284,
+ -0.04719004034996033,
+ 0.021362660452723503,
+ -0.01660226099193096,
+ 0.025590840727090836,
+ 0.023447182029485703,
+ 0.012063619680702686,
+ -0.003446484450250864,
+ -0.02579781413078308,
+ -0.04423326998949051,
+ 0.02671441249549389,
+ 0.041631314903497696,
+ 0.015596958808600903,
+ 0.016143960878252983,
+ -0.008825956843793392,
+ -0.003448332427069545,
+ 0.040655579417943954,
+ 9.528651571599767e-05,
+ -0.0021344178821891546,
+ -0.002557605504989624,
+ 0.029523342847824097,
+ 0.0016659548273310065,
+ 0.011361386626958847,
+ 0.011886212974786758,
+ -0.02822236530482769,
+ 0.028931990265846252,
+ 0.008885092101991177,
+ -0.04101039096713066,
+ 0.013726802542805672,
+ -0.03500815108418465,
+ -0.008012845180928707,
+ 0.0035592112690210342,
+ -0.021480930969119072,
+ 0.009469054639339447,
+ -0.014828198589384556,
+ -0.0005927399033680558,
+ -0.02659614197909832,
+ -0.030927808955311775,
+ -0.015286498703062534,
+ -0.005625254008919001,
+ 0.008589415811002254,
+ 0.01139095425605774,
+ 0.030898241326212883,
+ -0.005780484527349472,
+ -0.001752809970639646,
+ -0.036220427602529526,
+ -0.0036867219023406506,
+ -0.02943463996052742,
+ 0.009321215562522411,
+ -0.012684540823101997,
+ -0.013911600224673748,
+ 0.014599049463868141,
+ -0.022678421810269356,
+ 0.008855524472892284,
+ 0.013623315840959549,
+ 0.0009013526723720133,
+ 0.000988669809885323,
+ -0.01970686949789524,
+ -0.004309491720050573,
+ 0.018450241535902023,
+ -0.038497138768434525,
+ -0.009469054639339447,
+ -0.011775334365665913,
+ 0.029257234185934067,
+ 0.0078502232208848,
+ 0.03098694421350956,
+ -0.02387591451406479,
+ 0.006859705317765474,
+ -0.008212427608668804,
+ 0.014850374311208725,
+ 0.02560562454164028,
+ 0.001327774254605174,
+ 0.024452483281493187,
+ 0.017001423984766006,
+ -0.017563210800290108,
+ 0.008071980439126492,
+ -0.011827077716588974,
+ -0.017001423984766006,
+ 0.027438821271061897,
+ 0.017237966880202293,
+ 0.019322488456964493,
+ 0.04795880243182182,
+ 0.004745615180581808,
+ 0.009838650934398174,
+ 0.0023986792657524347,
+ -0.0321696512401104,
+ 0.025635192170739174,
+ 0.008182859979569912,
+ 0.0317852720618248,
+ 0.03657523915171623,
+ -0.022294042631983757,
+ -0.03610215708613396,
+ 0.039709415286779404,
+ -0.01530128251761198,
+ 0.007173861842602491,
+ 0.03533339500427246,
+ -0.0052002184092998505,
+ -0.03011469729244709,
+ -0.02217577025294304,
+ -0.0015578479506075382,
+ 0.011923172511160374,
+ 0.018982460722327232,
+ 0.013955951668322086,
+ -0.02800060622394085,
+ 0.0012113514821976423,
+ 0.02814844623208046,
+ -0.03548123314976692,
+ 0.011812293902039528,
+ 0.03840843588113785,
+ 0.005292617250233889,
+ 0.027113575488328934,
+ -0.007325396407395601,
+ 0.01159792859107256,
+ 0.010903087444603443,
+ 0.026625709608197212,
+ -0.005758308805525303,
+ 0.004091429989784956,
+ 0.021954013034701347,
+ 0.032140083611011505,
+ -0.00607246533036232,
+ 0.0014931686455383897,
+ 0.0026001092046499252,
+ 0.0332932248711586,
+ 0.050412919372320175,
+ -0.0024337908253073692,
+ 0.023476749658584595,
+ 0.007658033166080713,
+ -0.0166466124355793,
+ -0.0017971614142879844,
+ 0.0057213488034904,
+ 0.007983277551829815,
+ -0.042843591421842575,
+ -0.019159866496920586,
+ 0.01369723491370678,
+ 0.033884577453136444,
+ 0.016350936144590378,
+ -0.004298403859138489,
+ -0.01926335319876671,
+ 0.05830749496817589,
+ 0.018553728237748146,
+ -0.0020106032025069,
+ 0.015508255921304226,
+ 0.06215129420161247,
+ 0.005558726843446493,
+ 0.01339416578412056,
+ -0.0033522373996675014,
+ 0.031992245465517044,
+ -0.030336454510688782,
+ 0.021747039631009102,
+ 0.009129025973379612,
+ 0.0157004464417696,
+ 0.026684844866394997,
+ 0.04293229430913925,
+ 0.030173832550644875,
+ -0.04949632287025452,
+ 0.006789481732994318,
+ 0.03456463664770126,
+ -0.02285582758486271,
+ -0.008293738588690758,
+ -0.02152528241276741,
+ 0.014259020797908306,
+ -0.018065862357616425,
+ 0.020327789708971977,
+ 0.00298818526789546,
+ 0.043523646891117096,
+ 0.046480417251586914,
+ -0.0035425794776529074,
+ 0.030898241326212883,
+ 0.015863068401813507,
+ 0.020446060225367546,
+ -0.01713447831571102,
+ 0.01967730186879635,
+ -0.03690048307180405,
+ -0.04151304438710213,
+ -0.010910479351878166,
+ -0.02096349559724331,
+ -0.023506317287683487,
+ -0.013978127390146255,
+ -0.004497985355556011,
+ -0.014103790745139122,
+ -0.07273653149604797,
+ 0.05910582095384598,
+ -0.009143809787929058,
+ -0.0008061816915869713,
+ 0.024659456685185432,
+ 0.006264655385166407,
+ 0.002077130600810051,
+ 0.004224484320729971,
+ -0.00976473093032837,
+ 0.006238783709704876,
+ 0.029612045735120773,
+ 0.009890394285321236,
+ -0.005887667182832956,
+ 0.00929164793342352,
+ 0.005710260942578316,
+ 0.024230726063251495,
+ -0.01883462257683277,
+ 0.002962313359603286,
+ -0.032524462789297104,
+ -0.027335334569215775,
+ 0.006349662318825722,
+ -0.04952589049935341,
+ 0.012226241640746593,
+ 0.008655942976474762,
+ 0.003224726766347885,
+ 0.021776607260107994,
+ 0.00597637053579092,
+ -0.011383562348783016,
+ 0.01454730611294508,
+ 0.011967524886131287,
+ -0.005676997359842062,
+ -0.01725275069475174,
+ -0.006534460466355085,
+ -0.04970329627394676,
+ 0.028798934072256088,
+ 0.017622346058487892,
+ 0.010282166302204132,
+ -0.01021563820540905,
+ 0.024378564208745956,
+ -0.012810204178094864,
+ -0.039177194237709045,
+ 0.0026222849264740944,
+ 0.031755704432725906,
+ -0.027364902198314667,
+ 0.041365206241607666,
+ 0.01744494028389454,
+ 0.0008010997553355992,
+ 0.013305462896823883,
+ -0.0202390868216753,
+ -0.018524160608649254,
+ -0.012861947529017925,
+ 0.004254051949828863,
+ 0.007569329813122749,
+ 0.05588294193148613,
+ 0.02167312055826187,
+ -0.004335363395512104,
+ -0.008922051638364792,
+ -0.00031831470550969243,
+ 0.02807452715933323,
+ 0.009661244228482246,
+ -0.006693386938422918,
+ -0.02511775679886341,
+ 0.01235190499573946,
+ 0.002814474981278181,
+ 0.001953315921127796,
+ 0.01473949570208788,
+ 0.024275077506899834,
+ 0.017016207799315453,
+ -0.00896640308201313,
+ -0.013556787744164467,
+ -0.006142688449472189,
+ -0.011117453686892986,
+ -0.0308391060680151,
+ -0.04441067948937416,
+ 0.0037791209761053324,
+ -0.03184440732002258,
+ 0.014089006930589676,
+ -0.018790269270539284,
+ 0.015670878812670708,
+ 0.00660098809748888,
+ -0.023639371618628502,
+ 0.013519828207790852,
+ 0.048520587384700775,
+ -0.015375201590359211,
+ -0.021702688187360764,
+ 0.027941470965743065,
+ 0.031755704432725906,
+ 0.01346808485686779,
+ 0.012721500359475613,
+ 0.0003790670889429748,
+ -0.011398346163332462,
+ 0.03666394203901291,
+ 0.0033818050287663937,
+ -0.034298524260520935,
+ -0.011560969054698944,
+ 0.026906602084636688,
+ 0.019307704642415047,
+ -0.03601345419883728,
+ 0.021894877776503563,
+ -0.003143415553495288,
+ -0.011472265236079693,
+ -0.001932988059706986,
+ -0.0192929208278656,
+ 0.13400079309940338,
+ -0.016483990475535393,
+ -0.04588906094431877,
+ 0.003272774163633585,
+ -0.00902553927153349,
+ -0.025206459686160088,
+ -0.007033415604382753,
+ 0.0149464700371027,
+ 0.0056104701943695545,
+ 0.027335334569215775,
+ -0.0014793087029829621,
+ -0.0021214820444583893,
+ -0.036604806780815125,
+ 0.022914962843060493,
+ 0.012736284174025059,
+ 0.03196267783641815,
+ 0.02122960425913334,
+ -0.014909510500729084,
+ 0.019987761974334717,
+ -0.015508255921304226,
+ -0.004316883627325296,
+ -0.003453876357525587,
+ -0.005237177945673466,
+ -0.00013894506264477968,
+ -0.007358659990131855,
+ -0.037698812782764435,
+ -0.023624587804079056,
+ 0.024718593806028366,
+ 0.038497138768434525,
+ -0.008633767254650593,
+ 0.015759581699967384,
+ 0.020076464861631393,
+ 0.042961861938238144,
+ 0.015730014070868492,
+ 0.007414099294692278,
+ -0.017563210800290108,
+ -0.014887334778904915,
+ -0.027217064052820206,
+ 0.023328911513090134,
+ -0.0080424128100276,
+ 0.008508103899657726,
+ 0.006442061625421047,
+ 0.03716659173369408,
+ -0.02443769946694374,
+ 0.02072695456445217,
+ -0.02198358066380024,
+ 0.01356417965143919,
+ 0.011856645345687866,
+ 0.0027534915134310722,
+ 0.008582023903727531,
+ -0.019322488456964493,
+ 0.03559950366616249,
+ -0.003697809763252735,
+ -0.004035990219563246,
+ 0.019425975158810616,
+ 0.01530128251761198,
+ 0.003289405955001712,
+ 0.024452483281493187,
+ 0.03026253543794155,
+ 0.0037329215556383133,
+ 0.022027932107448578,
+ -0.008271562866866589,
+ 0.01447338704019785,
+ 0.002790451282635331,
+ 0.04367148503661156,
+ 0.012063619680702686,
+ 0.005924626719206572,
+ 0.05546899512410164,
+ 0.014392075128853321,
+ -0.01167184766381979,
+ 0.007901966571807861,
+ 0.0023192160297185183,
+ 0.010304342024028301,
+ 0.00896640308201313,
+ -0.007550850044935942,
+ -0.00612051272764802,
+ -0.00669708289206028,
+ 0.008811173029243946,
+ -0.021998364478349686,
+ -0.020712170749902725,
+ 0.0015541519969701767,
+ -0.004826926160603762,
+ 0.020534764975309372,
+ 0.013519828207790852,
+ -0.003082432085648179,
+ 0.022027932107448578,
+ -0.01282498799264431,
+ -0.01698664017021656,
+ -0.0024984702467918396,
+ -0.020593900233507156,
+ 0.012285376898944378,
+ -0.002487382385879755,
+ 0.01638050377368927,
+ -0.016587477177381516,
+ 0.007099942769855261,
+ 0.012980218045413494,
+ -0.03627956286072731,
+ 0.030129481106996536,
+ 0.011376170441508293,
+ 0.01002344861626625,
+ -0.019278137013316154,
+ -0.00811633188277483,
+ 0.017075343057513237,
+ -0.02633003145456314,
+ -0.02837020345032215,
+ -0.01051870733499527,
+ 0.02708400785923004,
+ -0.008618983440101147,
+ -0.035806477069854736,
+ -0.05336968973278999,
+ 0.006068769376724958,
+ -0.005580902565270662,
+ -0.021480930969119072,
+ -0.0008976567187346518,
+ 0.019884275272488594,
+ -0.01676488295197487,
+ 0.007136902306228876,
+ -0.00917337741702795,
+ 0.016483990475535393,
+ 0.01237408071756363,
+ 0.017341453582048416,
+ 0.03426895663142204,
+ 0.009868218563497066,
+ -0.031607866287231445,
+ 0.023136721923947334,
+ 0.0020808265544474125,
+ 0.006275743246078491,
+ 0.030779970809817314,
+ 0.030750403180718422,
+ -0.0034797480329871178,
+ -0.0382014624774456,
+ -0.011834469623863697,
+ 0.009801690466701984,
+ 0.002282256493344903,
+ -0.0023672636598348618,
+ 0.01785888709127903,
+ -0.007232997566461563,
+ -0.021022630855441093,
+ 0.022264475002884865,
+ -0.010230422019958496,
+ -0.02239752933382988,
+ -0.030513860285282135,
+ 0.007643249351531267,
+ 0.018642431125044823,
+ 0.004132085479795933,
+ 0.018775485455989838,
+ 0.0157004464417696,
+ 0.008781605400145054,
+ -0.002657396486029029,
+ -0.021273955702781677,
+ -0.023314127698540688,
+ -0.019573813304305077,
+ 0.03314538672566414,
+ -0.022648854181170464,
+ 0.026344815269112587,
+ 0.02599000371992588,
+ -0.008862916380167007,
+ 0.00997170526534319,
+ 0.02829628437757492,
+ -0.0008098776452243328,
+ -0.02038692496716976,
+ -0.002106698229908943,
+ -0.01339416578412056,
+ -0.02569432742893696,
+ 0.023698506876826286,
+ 0.017090126872062683,
+ -0.000379529083147645,
+ 0.01907116360962391,
+ -0.003585082944482565,
+ -0.0015319761587306857,
+ 0.015360417775809765,
+ -0.031075647100806236,
+ -0.0008939607650972903,
+ 0.005713956896215677,
+ 0.021702688187360764,
+ 0.006527068559080362,
+ -0.0036793299950659275,
+ -0.002404223196208477,
+ 0.01997297815978527,
+ -0.002668484579771757,
+ -0.029523342847824097,
+ -0.005044987890869379,
+ 0.008064588531851768,
+ 0.015286498703062534,
+ -0.0366935096681118,
+ 0.02140701189637184,
+ -0.009986489079892635,
+ -0.021007847040891647,
+ -0.013549395836889744,
+ -0.01997297815978527,
+ -0.012935866601765156,
+ -0.0002760421484708786,
+ 0.04665782302618027,
+ -0.008929443545639515,
+ 0.008131115697324276,
+ 0.01108788512647152,
+ -0.0172083992511034,
+ -0.015330850146710873,
+ 0.0049710688181221485,
+ 0.009905178099870682,
+ -0.01960338093340397,
+ -0.002709140069782734,
+ -0.0013434821739792824,
+ 0.04041903838515282,
+ 0.044055864214897156,
+ -0.017430156469345093,
+ 0.011716199107468128,
+ 0.012403648346662521,
+ 0.008205035701394081,
+ -0.005928322672843933,
+ 0.012085795402526855,
+ -0.009446878917515278,
+ -0.02489599958062172,
+ -0.020904360339045525,
+ 0.047870099544525146,
+ 0.031341757625341415,
+ -0.00036359025398269296,
+ 0.046214308589696884,
+ 0.02822236530482769,
+ 0.01079960074275732,
+ 0.001308370498009026,
+ -0.020372141152620316,
+ -0.008914659731090069,
+ -0.02613784186542034,
+ -0.001027477439492941,
+ 0.007343876175582409,
+ -0.011405738070607185,
+ -0.01405204739421606,
+ 0.0010635129874572158,
+ 0.03332279250025749,
+ 0.030366022139787674,
+ -0.014295980334281921,
+ 0.010843952186405659,
+ 0.020401708781719208,
+ -0.01289890706539154,
+ 0.008271562866866589,
+ -0.049998972564935684,
+ 0.009010755456984043,
+ -0.019928626716136932,
+ -0.001308370498009026,
+ -0.004291011951863766,
+ -0.025590840727090836,
+ -0.018435457721352577,
+ -0.025487352162599564,
+ -0.015449120663106441,
+ 0.028384987264871597,
+ 0.06292005628347397,
+ -0.02190966159105301,
+ 0.014007695019245148,
+ 0.02474816143512726,
+ 0.031075647100806236,
+ 0.01982514001429081,
+ -0.0035721471067517996,
+ -0.014236845076084137,
+ -0.016070041805505753,
+ -0.030336454510688782,
+ -0.009528189897537231,
+ -0.006767306011170149,
+ 0.01502038910984993,
+ 0.02130352333188057,
+ -0.017888454720377922,
+ 0.016513558104634285,
+ 0.031016511842608452,
+ -0.009705595672130585,
+ 0.011989700607955456,
+ -0.01051870733499527,
+ 0.0005530082853510976,
+ 0.029301585629582405,
+ -0.05011724308133125,
+ 0.016587477177381516,
+ -0.0072847409173846245,
+ -0.0028495865408331156,
+ -0.02999642677605152,
+ -0.008611591532826424,
+ 0.015153443440794945,
+ 0.020874792709946632,
+ 0.00016527879051864147,
+ -0.005410888232290745,
+ 0.0022804085165262222,
+ -0.021968796849250793,
+ -0.015936987474560738,
+ 0.026093490421772003,
+ -0.0221314188092947,
+ 0.021200036630034447,
+ 0.0035573632922023535,
+ 0.002790451282635331,
+ 0.019869491457939148,
+ 0.02122960425913334,
+ -0.009328607469797134,
+ -0.03400284796953201,
+ 0.011376170441508293,
+ 0.020519981160759926,
+ -0.007724560331553221,
+ 0.015759581699967384,
+ 0.022087067365646362,
+ 0.031164349988102913,
+ -0.009786906652152538,
+ -0.020268654450774193,
+ -0.02227925881743431,
+ -7.975192420417443e-05,
+ -0.0004518313508015126,
+ 0.011738374829292297,
+ 0.044588085263967514,
+ -0.004930413328111172,
+ 0.007768911775201559,
+ 0.03559950366616249,
+ -0.008471144363284111,
+ 0.029035476967692375,
+ -0.007332788314670324,
+ -0.0065492442809045315,
+ -0.024112455546855927,
+ -0.0174005888402462,
+ -0.004656911827623844,
+ -0.002694356255233288,
+ -0.027438821271061897,
+ 0.022042715921998024,
+ -0.005968978628516197,
+ 0.05328098684549332,
+ 0.004571904893964529,
+ 0.03790578618645668,
+ 0.035806477069854736,
+ 0.006105728913098574,
+ 0.003136023646220565,
+ -0.018139781430363655,
+ 0.025590840727090836,
+ -0.001859992858953774,
+ 0.004634736105799675,
+ 0.005536550655961037,
+ 0.047308310866355896,
+ -0.010858736000955105,
+ -0.012322336435317993,
+ 0.007247781381011009,
+ 0.007613681256771088,
+ 0.0026370687410235405,
+ 0.015153443440794945,
+ -0.015168227255344391,
+ 0.00824938714504242,
+ -0.02035735733807087,
+ 0.03205138072371483,
+ -0.03157829865813255,
+ -0.015345633961260319,
+ -0.015256930142641068,
+ -0.002542821690440178,
+ 0.012773244641721249,
+ -0.015596958808600903,
+ 0.008752037771046162,
+ -0.0035277956631034613,
+ -0.017681481316685677,
+ 0.014429034665226936,
+ 0.050797298550605774,
+ -0.017622346058487892,
+ 0.001332394196651876,
+ 0.007439971435815096,
+ 0.001193795702420175,
+ -0.0047345273196697235,
+ 0.005455239675939083,
+ 0.02451161853969097,
+ -0.04970329627394676,
+ 0.008012845180928707,
+ -0.0057693966664373875,
+ 0.007366051897406578,
+ -0.04582992568612099,
+ -0.03734399750828743,
+ 0.02424550987780094,
+ 0.00841940101236105,
+ 0.06132340058684349,
+ -0.024038536474108696,
+ -0.003544427454471588,
+ -0.01728231832385063,
+ 0.0061796484515070915,
+ -0.008271562866866589,
+ -0.019322488456964493,
+ 2.53086764132604e-05,
+ -0.05845533311367035,
+ 0.00972037948668003,
+ -0.021540066227316856,
+ 0.032554030418395996,
+ 0.006412493996322155,
+ -0.0009674180182628334,
+ 0.00830113049596548,
+ 0.012603229843080044,
+ -0.034298524260520935,
+ -0.015803933143615723,
+ -7.484322850359604e-05,
+ 0.011812293902039528,
+ -0.002601957181468606,
+ -0.012980218045413494,
+ -0.01907116360962391,
+ -0.006017026025801897,
+ ],
+ "index": 0,
+ }
+ ],
+ "model": "text-embedding-3-small",
+ "usage": {"prompt_tokens": 1, "total_tokens": 1},
+ },
+}
+
+
+@pytest.fixture
+def sap_api_response():
+ return moke_response
+
+
+@pytest.fixture
+def fake_token_creator():
+ return lambda: "Bearer FAKE_TOKEN", "https://api.ai.moke-sap.com", "fake-group"
+
+
+@pytest.fixture
+def fake_deployment_url():
+ return "https://api.ai.moke-sap.com/v2/inference/deployments/mokeid"
+
+
+@pytest.mark.parametrize("sync_mode", [True, False])
+@pytest.mark.asyncio
+async def test_sap_chat(
+ respx_mock,
+ sap_api_response,
+ fake_token_creator,
+ fake_deployment_url,
+ sync_mode,
+):
+ import litellm
+
+ litellm.disable_aiohttp_transport = True
+ with patch(
+ "litellm.llms.sap.embed.transformation.GenAIHubEmbeddingConfig.deployment_url",
+ new_callable=PropertyMock,
+ return_value=fake_deployment_url,
+ ), patch(
+ "litellm.llms.sap.embed.transformation.get_token_creator",
+ return_value=fake_token_creator,
+ ):
+ model = "sap/text-embedding-3-small"
+ input = "Hi"
+ respx_mock.post(f"{fake_deployment_url}/v2/embeddings").respond(
+ json=sap_api_response
+ )
+
+ if sync_mode:
+ response = litellm.embedding(model=model, input=input)
+ else:
+ response = await litellm.aembedding(model=model, input=input)
+
+ assert response
+ assert response.data[0]["embedding"]
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_ui_session_utils.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_ui_session_utils.py
new file mode 100644
index 00000000000..f372f7b181c
--- /dev/null
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_ui_session_utils.py
@@ -0,0 +1,92 @@
+import pytest
+from types import SimpleNamespace
+from unittest.mock import AsyncMock
+
+from litellm.constants import UI_SESSION_TOKEN_TEAM_ID
+from litellm.proxy._types import UserAPIKeyAuth
+
+from litellm.proxy._experimental.mcp_server.ui_session_utils import (
+ build_effective_auth_contexts,
+ clone_user_api_key_auth_with_team,
+ resolve_ui_session_team_ids,
+)
+
+
+def test_clone_user_api_key_auth_with_team_creates_independent_copy():
+ original = UserAPIKeyAuth(team_id="team-original", user_id="user-123")
+
+ cloned = clone_user_api_key_auth_with_team(original, "team-override")
+
+ assert cloned is not original
+ assert cloned.team_id == "team-override"
+ assert original.team_id == "team-original"
+
+
+@pytest.mark.asyncio
+async def test_resolve_ui_session_team_ids_returns_unique_ids(monkeypatch):
+ user_auth = UserAPIKeyAuth(
+ team_id=UI_SESSION_TOKEN_TEAM_ID,
+ user_id="user-1",
+ )
+
+ fake_user = SimpleNamespace(
+ teams=["team-a", "team-b", "team-a", "", None, "team-c"]
+ )
+
+ monkeypatch.setattr(
+ "litellm.proxy.auth.auth_checks.get_user_object",
+ AsyncMock(return_value=fake_user),
+ )
+
+ import litellm.proxy.proxy_server as proxy_server
+
+ monkeypatch.setattr(proxy_server, "prisma_client", object())
+ monkeypatch.setattr(proxy_server, "proxy_logging_obj", None)
+ monkeypatch.setattr(proxy_server, "user_api_key_cache", None)
+
+ team_ids = await resolve_ui_session_team_ids(user_auth)
+
+ assert team_ids == ["team-a", "team-b", "team-c"]
+
+
+@pytest.mark.asyncio
+async def test_resolve_ui_session_team_ids_short_circuits_when_not_ui_session():
+ normal_user = UserAPIKeyAuth(team_id="regular-team", user_id="user-1")
+
+ result = await resolve_ui_session_team_ids(normal_user)
+
+ assert result == []
+
+
+@pytest.mark.asyncio
+async def test_build_effective_auth_contexts_returns_cloned_contexts(monkeypatch):
+ user_auth = UserAPIKeyAuth(team_id=UI_SESSION_TOKEN_TEAM_ID, user_id="user-42")
+
+ mock_resolve = AsyncMock(return_value=["team-one", "team-two"])
+ monkeypatch.setattr(
+ "litellm.proxy._experimental.mcp_server.ui_session_utils.resolve_ui_session_team_ids",
+ mock_resolve,
+ )
+
+ contexts = await build_effective_auth_contexts(user_auth)
+
+ assert [ctx.team_id for ctx in contexts] == ["team-one", "team-two"]
+ assert all(ctx is not user_auth for ctx in contexts)
+ mock_resolve.assert_awaited_once_with(user_auth)
+
+
+@pytest.mark.asyncio
+async def test_build_effective_auth_contexts_returns_original_when_no_resolution(monkeypatch):
+ user_auth = UserAPIKeyAuth(team_id="existing-team", user_id="user-7")
+
+ mock_resolve = AsyncMock(return_value=[])
+ monkeypatch.setattr(
+ "litellm.proxy._experimental.mcp_server.ui_session_utils.resolve_ui_session_team_ids",
+ mock_resolve,
+ )
+
+ contexts = await build_effective_auth_contexts(user_auth)
+
+ assert contexts == [user_auth]
+ mock_resolve.assert_awaited_once_with(user_auth)
+
diff --git a/tests/test_litellm/proxy/auth/test_route_checks.py b/tests/test_litellm/proxy/auth/test_route_checks.py
index f9276645a58..4402c09e278 100644
--- a/tests/test_litellm/proxy/auth/test_route_checks.py
+++ b/tests/test_litellm/proxy/auth/test_route_checks.py
@@ -804,4 +804,34 @@ def test_proxy_admin_viewer_can_access_global_spend_tags():
pytest.fail(
f"proxy_admin_viewer should be able to access /global/spend/tags route. Got error: {str(e)}"
)
+
+
+def test_route_in_additional_public_routes_wildcard_match():
+ """
+ Test that route_in_additonal_public_routes supports wildcard patterns.
+ """
+ from litellm.proxy.auth.auth_utils import route_in_additonal_public_routes
+
+ with patch("litellm.proxy.proxy_server.general_settings", {"public_routes": ["/api/*"]}), \
+ patch("litellm.proxy.proxy_server.premium_user", True):
+ # Wildcard should match subpaths
+ assert route_in_additonal_public_routes("/api/users") is True
+ assert route_in_additonal_public_routes("/api/users/123") is True
+ # Should not match different prefix
+ assert route_in_additonal_public_routes("/other/path") is False
+
+
+def test_route_in_additional_public_routes_exact_match():
+ """
+ Test that route_in_additonal_public_routes supports exact matches.
+ """
+ from litellm.proxy.auth.auth_utils import route_in_additonal_public_routes
+
+ with patch("litellm.proxy.proxy_server.general_settings", {"public_routes": ["/health", "/status"]}), \
+ patch("litellm.proxy.proxy_server.premium_user", True):
+ # Exact matches should work
+ assert route_in_additonal_public_routes("/health") is True
+ assert route_in_additonal_public_routes("/status") is True
+ # Non-matching routes should fail
+ assert route_in_additonal_public_routes("/other") is False
diff --git a/tests/test_litellm/proxy/hooks/test_dynamic_rate_limiter_v3.py b/tests/test_litellm/proxy/hooks/test_dynamic_rate_limiter_v3.py
index d9e10e6f4b8..095d5f50dcf 100644
--- a/tests/test_litellm/proxy/hooks/test_dynamic_rate_limiter_v3.py
+++ b/tests/test_litellm/proxy/hooks/test_dynamic_rate_limiter_v3.py
@@ -1418,6 +1418,100 @@ async def test_async_log_success_event_increments_by_actual_tokens():
assert any("priority_model" in k and "dev" in k for k in keys), "Should increment priority_model with 'dev' priority"
+@pytest.mark.asyncio
+async def test_saturation_check_cache_ttl_configuration():
+ """
+ Test that saturation_check_cache_ttl controls how long saturation values are cached locally.
+
+ This validates the configurable TTL for multi-node consistency:
+ - When saturation_check_cache_ttl is set, local cache should expire after that duration
+ - After expiration, fresh values should be fetched from Redis
+ - This prevents nodes from having stale saturation data in multi-node deployments
+ """
+ os.environ["LITELLM_LICENSE"] = "test-license-key"
+
+ # Set a short TTL for testing (5 seconds)
+ original_ttl = litellm.priority_reservation_settings.saturation_check_cache_ttl
+ litellm.priority_reservation_settings.saturation_check_cache_ttl = 5
+
+ try:
+ dual_cache = DualCache()
+ handler = DynamicRateLimitHandler(internal_usage_cache=dual_cache)
+
+ model = "test-saturation-ttl"
+ llm_router = Router(
+ model_list=[
+ {
+ "model_name": model,
+ "litellm_params": {
+ "model": "gpt-3.5-turbo",
+ "api_key": "test-key",
+ "api_base": "test-base",
+ "rpm": 100,
+ "tpm": 1000,
+ },
+ }
+ ]
+ )
+ handler.update_variables(llm_router=llm_router)
+
+ # Verify the TTL getter returns configured value
+ assert handler._get_saturation_check_cache_ttl() == 5, (
+ "TTL should be configurable via priority_reservation_settings"
+ )
+
+ # Track async_get_cache calls to verify TTL is passed
+ get_cache_calls = []
+ original_get_cache = handler.internal_usage_cache.async_get_cache
+
+ async def mock_get_cache(key, litellm_parent_otel_span=None, local_only=False, **kwargs):
+ get_cache_calls.append({
+ "key": key,
+ "ttl": kwargs.get("ttl"),
+ "local_only": local_only,
+ })
+ return None # Simulate cache miss
+
+ handler.internal_usage_cache.async_get_cache = mock_get_cache
+
+ # Call _get_saturation_value_from_cache
+ counter_key = handler.v3_limiter.create_rate_limit_keys(
+ key="model_saturation_check",
+ value=model,
+ rate_limit_type="requests",
+ )
+
+ await handler._get_saturation_value_from_cache(counter_key=counter_key)
+
+ # Verify async_get_cache was called with the configured TTL
+ assert len(get_cache_calls) == 1, "Expected 1 cache call"
+ assert get_cache_calls[0]["ttl"] == 5, (
+ f"Expected TTL of 5 seconds, got {get_cache_calls[0]['ttl']}"
+ )
+ assert get_cache_calls[0]["local_only"] is False, (
+ "Should check Redis (local_only=False) for multi-node consistency"
+ )
+
+ # Test with different TTL value
+ get_cache_calls.clear()
+ litellm.priority_reservation_settings.saturation_check_cache_ttl = 30
+
+ await handler._get_saturation_value_from_cache(counter_key=counter_key)
+
+ assert get_cache_calls[0]["ttl"] == 30, (
+ f"TTL should update to 30 seconds, got {get_cache_calls[0]['ttl']}"
+ )
+
+ print("Saturation check cache TTL test passed:")
+ print(" - TTL is configurable via priority_reservation_settings.saturation_check_cache_ttl")
+ print(" - TTL is passed to async_get_cache for local cache expiration control")
+ print(" - local_only=False ensures Redis is checked for multi-node consistency")
+
+ finally:
+ # Restore original TTL
+ litellm.priority_reservation_settings.saturation_check_cache_ttl = original_ttl
+
+
@pytest.mark.asyncio
async def test_async_log_success_event_uses_team_priority_from_auth_metadata():
"""
diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py
index e54e537eed1..59ab5068fa1 100644
--- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py
+++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py
@@ -151,4 +151,132 @@ class TestAnthropicLoggingHandlerModelFallback:
if not model and hasattr(logging_obj, 'model_call_details') and logging_obj.model_call_details.get('model'):
model = logging_obj.model_call_details.get('model')
- assert model == "" # Should remain empty
\ No newline at end of file
+ assert model == "" # Should remain empty
+
+
+class TestAzureAnthropicCostCalculation:
+ """Test the custom_llm_provider cost calculation logic for Azure AI Anthropic."""
+
+ def _create_mock_logging_obj(
+ self, model: str = None, custom_llm_provider: str = None
+ ) -> LiteLLMLoggingObj:
+ """Create a mock logging object with optional model and custom_llm_provider"""
+ mock_logging_obj = MagicMock()
+ mock_model_call_details = {}
+ if model:
+ mock_model_call_details["model"] = model
+ if custom_llm_provider:
+ mock_model_call_details["custom_llm_provider"] = custom_llm_provider
+ mock_logging_obj.model_call_details = mock_model_call_details
+ mock_logging_obj.litellm_call_id = "test-call-id"
+ return mock_logging_obj
+
+ @patch("litellm.completion_cost")
+ def test_cost_calculation_with_azure_ai_custom_llm_provider(
+ self, mock_completion_cost
+ ):
+ """Test that custom_llm_provider is passed to completion_cost for Azure AI Anthropic"""
+ from litellm.types.utils import ModelResponse
+ from datetime import datetime
+
+ mock_completion_cost.return_value = 0.001
+
+ logging_obj = self._create_mock_logging_obj(
+ model="claude-sonnet-4-5_gb_20250929", custom_llm_provider="azure_ai"
+ )
+
+ mock_response = MagicMock(spec=ModelResponse)
+ mock_response.id = "test-id"
+ mock_response.model = "claude-sonnet-4-5_gb_20250929"
+
+ kwargs = {}
+ start_time = datetime.now()
+ end_time = datetime.now()
+
+ AnthropicPassthroughLoggingHandler._create_anthropic_response_logging_payload(
+ litellm_model_response=mock_response,
+ model="claude-sonnet-4-5_gb_20250929",
+ kwargs=kwargs,
+ start_time=start_time,
+ end_time=end_time,
+ logging_obj=logging_obj,
+ )
+
+ # Verify completion_cost was called with the correct parameters
+ mock_completion_cost.assert_called_once()
+ call_kwargs = mock_completion_cost.call_args[1]
+ assert call_kwargs["model"] == "azure_ai/claude-sonnet-4-5_gb_20250929"
+ assert call_kwargs["custom_llm_provider"] == "azure_ai"
+
+ @patch("litellm.completion_cost")
+ def test_cost_calculation_without_custom_llm_provider(self, mock_completion_cost):
+ """Test that cost calculation works without custom_llm_provider (standard Anthropic)"""
+ from litellm.types.utils import ModelResponse
+ from datetime import datetime
+
+ mock_completion_cost.return_value = 0.001
+
+ # No custom_llm_provider in model_call_details
+ logging_obj = self._create_mock_logging_obj(model="claude-3-sonnet-20240229")
+
+ mock_response = MagicMock(spec=ModelResponse)
+ mock_response.id = "test-id"
+ mock_response.model = "claude-3-sonnet-20240229"
+
+ kwargs = {}
+ start_time = datetime.now()
+ end_time = datetime.now()
+
+ AnthropicPassthroughLoggingHandler._create_anthropic_response_logging_payload(
+ litellm_model_response=mock_response,
+ model="claude-3-sonnet-20240229",
+ kwargs=kwargs,
+ start_time=start_time,
+ end_time=end_time,
+ logging_obj=logging_obj,
+ )
+
+ # Verify completion_cost was called without provider prefix
+ mock_completion_cost.assert_called_once()
+ call_kwargs = mock_completion_cost.call_args[1]
+ assert call_kwargs["model"] == "claude-3-sonnet-20240229"
+ assert call_kwargs["custom_llm_provider"] is None
+
+ @patch("litellm.completion_cost")
+ def test_cost_calculation_does_not_duplicate_provider_prefix(
+ self, mock_completion_cost
+ ):
+ """Test that provider prefix is not duplicated if already present in model name"""
+ from litellm.types.utils import ModelResponse
+ from datetime import datetime
+
+ mock_completion_cost.return_value = 0.001
+
+ logging_obj = self._create_mock_logging_obj(
+ model="azure_ai/claude-sonnet-4-5_gb_20250929",
+ custom_llm_provider="azure_ai",
+ )
+
+ mock_response = MagicMock(spec=ModelResponse)
+ mock_response.id = "test-id"
+ mock_response.model = "azure_ai/claude-sonnet-4-5_gb_20250929"
+
+ kwargs = {}
+ start_time = datetime.now()
+ end_time = datetime.now()
+
+ # Model already has the provider prefix
+ AnthropicPassthroughLoggingHandler._create_anthropic_response_logging_payload(
+ litellm_model_response=mock_response,
+ model="azure_ai/claude-sonnet-4-5_gb_20250929",
+ kwargs=kwargs,
+ start_time=start_time,
+ end_time=end_time,
+ logging_obj=logging_obj,
+ )
+
+ # Verify provider prefix was not duplicated
+ mock_completion_cost.assert_called_once()
+ call_kwargs = mock_completion_cost.call_args[1]
+ assert call_kwargs["model"] == "azure_ai/claude-sonnet-4-5_gb_20250929"
+ assert call_kwargs["custom_llm_provider"] == "azure_ai"
\ No newline at end of file
diff --git a/test_litellm/proxy/pass_through_endpoints/test_passthrough_guardrails_field_targeting.py b/tests/test_litellm/proxy/pass_through_endpoints/test_passthrough_guardrails_field_targeting.py
similarity index 100%
rename from test_litellm/proxy/pass_through_endpoints/test_passthrough_guardrails_field_targeting.py
rename to tests/test_litellm/proxy/pass_through_endpoints/test_passthrough_guardrails_field_targeting.py
diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py
index 76fd37596e9..7b99cd0b920 100644
--- a/tests/test_litellm/proxy/test_proxy_server.py
+++ b/tests/test_litellm/proxy/test_proxy_server.py
@@ -124,6 +124,44 @@ def test_login_v2_returns_redirect_url_and_sets_cookie(monkeypatch):
)
+def test_fallback_login_has_no_deprecation_banner(client_no_auth):
+ response = client_no_auth.get("/fallback/login")
+
+ assert response.status_code == 200
+ html = response.text
+ assert '' not in html
+ assert "Deprecated:" not in html
+ assert "