Spend
${formatNumberWithCommas(currentKeyData.spend, 4)}
-
- of{" "}
- {currentKeyData.max_budget !== null
- ? `$${formatNumberWithCommas(currentKeyData.max_budget, 2)}`
- : "Unlimited"}
-
+ of {budgetDisplay}
From 4b22aa1fcaf9e7b07ba4396146b2d5aa34e453d9 Mon Sep 17 00:00:00 2001
From: jesco
Date: Wed, 24 Jun 2026 07:03:52 -0400
Subject: [PATCH 23/30] fix(anthropic): emit replayable streaming thinking
blocks (#31022)
---
litellm/llms/anthropic/chat/handler.py | 59 ++++----
.../chat/test_anthropic_chat_handler.py | 131 ++++++++++++++++++
2 files changed, 159 insertions(+), 31 deletions(-)
diff --git a/litellm/llms/anthropic/chat/handler.py b/litellm/llms/anthropic/chat/handler.py
index 5d14f3cc4ae..d2e5b89371e 100644
--- a/litellm/llms/anthropic/chat/handler.py
+++ b/litellm/llms/anthropic/chat/handler.py
@@ -629,6 +629,7 @@ class ModelResponseIterator:
Optional[ChatCompletionToolCallChunk],
List[Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]],
Dict[str, Any],
+ Optional[str],
]:
"""
Helper function to handle the content block delta
@@ -636,6 +637,7 @@ class ModelResponseIterator:
text = ""
tool_use: Optional[ChatCompletionToolCallChunk] = None
provider_specific_fields = {}
+ reasoning_content: Optional[str] = None
content_block = ContentBlockDelta(**chunk) # type: ignore
thinking_blocks: List[
Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]
@@ -670,14 +672,24 @@ class ModelResponseIterator:
thinking_content = content_block["delta"].get("thinking")
if isinstance(thinking_content, str) and thinking_content:
self.reasoning_content_chunks.append(thinking_content)
- thinking_blocks = [
- ChatCompletionThinkingBlock(
- type="thinking",
- thinking=thinking_content or "",
- signature=str(content_block["delta"].get("signature") or ""),
- )
- ]
- provider_specific_fields["thinking_blocks"] = thinking_blocks
+ reasoning_content = thinking_content
+
+ signature = content_block["delta"].get("signature")
+ if isinstance(signature, str) and signature:
+ thinking_blocks = [
+ ChatCompletionThinkingBlock(
+ type="thinking",
+ thinking="".join(
+ cast(str, block["delta"].get("thinking"))
+ for block in self.content_blocks
+ if isinstance(block["delta"].get("thinking"), str)
+ ),
+ signature=signature,
+ )
+ ]
+ provider_specific_fields["thinking_blocks"] = thinking_blocks
+ if reasoning_content is None:
+ reasoning_content = ""
elif (
"content" in content_block["delta"]
and content_block["delta"].get("type") == "compaction_delta"
@@ -688,25 +700,13 @@ class ModelResponseIterator:
"content": content_block["delta"]["content"],
}
- return text, tool_use, thinking_blocks, provider_specific_fields
-
- def _handle_reasoning_content(
- self,
- thinking_blocks: List[
- Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]
- ],
- ) -> Optional[str]:
- """
- Handle the reasoning content
- """
- reasoning_content = None
- for block in thinking_blocks:
- thinking_content = cast(Optional[str], block.get("thinking"))
- if reasoning_content is None:
- reasoning_content = ""
- if thinking_content is not None:
- reasoning_content += thinking_content
- return reasoning_content
+ return (
+ text,
+ tool_use,
+ thinking_blocks,
+ provider_specific_fields,
+ reasoning_content,
+ )
def _handle_redacted_thinking_content(
self,
@@ -802,11 +802,8 @@ class ModelResponseIterator:
tool_use,
thinking_blocks,
provider_specific_fields,
+ reasoning_content,
) = self._content_block_delta_helper(chunk=chunk)
- if thinking_blocks:
- reasoning_content = self._handle_reasoning_content(
- thinking_blocks=thinking_blocks
- )
elif type_chunk == "content_block_start":
"""
event: content_block_start
diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_handler.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_handler.py
index 2fdd639e74d..0b1aaf87516 100644
--- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_handler.py
+++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_handler.py
@@ -74,6 +74,137 @@ def test_redacted_thinking_content_block_delta():
assert "thinking_blocks" in model_response.choices[0].delta.provider_specific_fields
+def test_streaming_thinking_blocks_are_replayable_after_signature_delta():
+ model_response_iterator = ModelResponseIterator(
+ streaming_response=MagicMock(), sync_stream=True, json_mode=False
+ )
+ chunks = [
+ {
+ "type": "content_block_start",
+ "index": 0,
+ "content_block": {"type": "thinking", "thinking": ""},
+ },
+ {
+ "type": "content_block_delta",
+ "index": 0,
+ "delta": {"type": "thinking_delta", "thinking": "Step 1. "},
+ },
+ {
+ "type": "content_block_delta",
+ "index": 0,
+ "delta": {"type": "thinking_delta", "thinking": "Step 2."},
+ },
+ {
+ "type": "content_block_delta",
+ "index": 0,
+ "delta": {"type": "signature_delta", "signature": "sig-final"},
+ },
+ ]
+
+ parsed_chunks = [
+ model_response_iterator.chunk_parser(chunk=chunk) for chunk in chunks
+ ]
+ reasoning_content = "".join(
+ getattr(chunk.choices[0].delta, "reasoning_content", None) or ""
+ for chunk in parsed_chunks
+ )
+ thinking_blocks = tuple(
+ block
+ for chunk in parsed_chunks
+ for block in (getattr(chunk.choices[0].delta, "thinking_blocks", None) or [])
+ )
+ expected_thinking_block = {
+ "type": "thinking",
+ "thinking": "Step 1. Step 2.",
+ "signature": "sig-final",
+ }
+
+ assert reasoning_content == "Step 1. Step 2."
+ assert thinking_blocks == (expected_thinking_block,)
+ assert parsed_chunks[-1].choices[0].delta.provider_specific_fields == {
+ "thinking_blocks": [expected_thinking_block]
+ }
+
+
+def test_streaming_unsigned_thinking_deltas_keep_reasoning_content():
+ model_response_iterator = ModelResponseIterator(
+ streaming_response=MagicMock(), sync_stream=True, json_mode=False
+ )
+ chunks = [
+ {
+ "type": "content_block_start",
+ "index": 0,
+ "content_block": {"type": "thinking", "thinking": ""},
+ },
+ {
+ "type": "content_block_delta",
+ "index": 0,
+ "delta": {"type": "thinking_delta", "thinking": "Step 1. "},
+ },
+ {
+ "type": "content_block_delta",
+ "index": 0,
+ "delta": {"type": "thinking_delta", "thinking": "Step 2."},
+ },
+ {"type": "content_block_stop", "index": 0},
+ ]
+
+ parsed_chunks = [
+ model_response_iterator.chunk_parser(chunk=chunk) for chunk in chunks
+ ]
+ reasoning_content = "".join(
+ getattr(chunk.choices[0].delta, "reasoning_content", None) or ""
+ for chunk in parsed_chunks
+ )
+ thinking_blocks = tuple(
+ block
+ for chunk in parsed_chunks
+ for block in (getattr(chunk.choices[0].delta, "thinking_blocks", None) or [])
+ )
+
+ assert reasoning_content == "Step 1. Step 2."
+ assert thinking_blocks == ()
+
+
+def test_streaming_truncated_thinking_deltas_keep_reasoning_content():
+ model_response_iterator = ModelResponseIterator(
+ streaming_response=MagicMock(), sync_stream=True, json_mode=False
+ )
+ chunks = [
+ {
+ "type": "content_block_start",
+ "index": 0,
+ "content_block": {"type": "thinking", "thinking": ""},
+ },
+ {
+ "type": "content_block_delta",
+ "index": 0,
+ "delta": {"type": "thinking_delta", "thinking": "Step 1. "},
+ },
+ {
+ "type": "content_block_delta",
+ "index": 0,
+ "delta": {"type": "thinking_delta", "thinking": "Step 2."},
+ },
+ ]
+
+ parsed_chunks = [
+ model_response_iterator.chunk_parser(chunk=chunk) for chunk in chunks
+ ]
+ reasoning_content = "".join(
+ getattr(chunk.choices[0].delta, "reasoning_content", None) or ""
+ for chunk in parsed_chunks
+ )
+ thinking_blocks = tuple(
+ block
+ for chunk in parsed_chunks
+ for block in (getattr(chunk.choices[0].delta, "thinking_blocks", None) or [])
+ )
+
+ assert reasoning_content == "Step 1. Step 2."
+ assert thinking_blocks == ()
+
+
def test_handle_json_mode_chunk_response_format_tool():
model_response_iterator = ModelResponseIterator(
streaming_response=MagicMock(), sync_stream=True, json_mode=True
From 7ee492f749f2a573bacbded0c060b0ea719fd431 Mon Sep 17 00:00:00 2001
From: Rick <26716961+Bytechoreographer@users.noreply.github.com>
Date: Wed, 24 Jun 2026 19:05:34 +0800
Subject: [PATCH 24/30] feat(proxy): read cold-storage prompts back in the logs
detail view (#30364)
* feat(proxy): read cold-storage prompts back in the logs detail view
When a deployment offloads prompts and responses to cold storage instead of
Postgres, the spend-log row holds only "{}" placeholders plus a
metadata.cold_storage_object_key pointer, so the UI logs detail drawer showed
nothing. The detail endpoint only read the placeholder columns and never
fetched the object back.
Resolve the payload per row based on actual content, not a config flag: if
Postgres has content, return it; otherwise read the exact stored object key and
fetch from the configured cold storage backend through ColdStorageHandler.
Reading the persisted key is a single GET. The key embeds a microsecond
timestamp that cannot be reconstructed from the millisecond-precision startTime
column, and listing the day's prefix to match on request_id would be too
expensive for this per-open path.
Also teach the detail drawer's pretty-view parser to accept a bare messages
array. The cold storage payload carries the prompt as a top-level messages list
with no proxy_server_request, so without this the output rendered while the
input stayed blank.
ColdStorageHandler gains an optional injected logger so the resolver can be unit
tested without monkeypatching. Postgres-stored prompts are unaffected: the fast
path returns the existing columns and the request-body object still renders the
same way.
* Update litellm/proxy/spend_tracking/spend_management_endpoints.py
Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com>
* test(proxy): cover ColdStorageHandler resolution paths and cold-storage fetch failure
Add unit tests for ColdStorageHandler (injected logger, graceful None when no
logger is configured, and resolution of a configured logger from the callback
registry) and a regression test asserting a cold storage backend exception
degrades to the Postgres values instead of surfacing a 500.
---------
Co-authored-by: Bytechoreographer
Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com>
---
.../spend_tracking/cold_storage_handler.py | 37 ++-
.../spend_management_endpoints.py | 121 ++++++-
.../test_spend_management_endpoints.py | 295 ++++++++++++++++++
.../PrettyMessagesView.test.tsx | 11 +
.../LogDetailsDrawer/prettyMessagesUtils.ts | 24 +-
5 files changed, 447 insertions(+), 41 deletions(-)
diff --git a/litellm/proxy/spend_tracking/cold_storage_handler.py b/litellm/proxy/spend_tracking/cold_storage_handler.py
index 57c41bafccd..3974d4df618 100644
--- a/litellm/proxy/spend_tracking/cold_storage_handler.py
+++ b/litellm/proxy/spend_tracking/cold_storage_handler.py
@@ -16,8 +16,14 @@ class ColdStorageHandler:
This class is responsible for handling Getting/Setting the proxy server request from cold storage.
It allows fetching a dict of the proxy server request from s3 or GCS bucket.
+
+ The cold storage logger can be injected for testing; when omitted it is
+ resolved from the configured ``litellm.cold_storage_custom_logger``.
"""
+ def __init__(self, cold_storage_logger: Optional[CustomLogger] = None):
+ self._injected_cold_storage_logger = cold_storage_logger
+
async def get_proxy_server_request_from_cold_storage_with_object_key(
self,
object_key: str,
@@ -31,33 +37,26 @@ class ColdStorageHandler:
Returns:
Optional[dict]: The proxy server request dict or None if not found
"""
-
- # select the custom logger to use for cold storage
- custom_logger_name: Optional[_custom_logger_compatible_callbacks_literal] = (
- self._select_custom_logger_for_cold_storage()
+ custom_logger = (
+ self._injected_cold_storage_logger or self._resolve_cold_storage_logger()
)
-
- # if no custom logger name is configured, return None
- if custom_logger_name is None:
+ if custom_logger is None:
return None
- # get the active/initialized custom logger
- custom_logger: Optional[CustomLogger] = (
+ return await custom_logger.get_proxy_server_request_from_cold_storage_with_object_key(
+ object_key=object_key,
+ )
+
+ def _resolve_cold_storage_logger(self) -> Optional[CustomLogger]:
+ custom_logger_name = self._select_custom_logger_for_cold_storage()
+ if custom_logger_name is None:
+ return None
+ return (
litellm.logging_callback_manager.get_active_custom_logger_for_callback_name(
custom_logger_name
)
)
- # if no custom logger is found, return None
- if custom_logger is None:
- return None
-
- proxy_server_request = await custom_logger.get_proxy_server_request_from_cold_storage_with_object_key(
- object_key=object_key,
- )
-
- return proxy_server_request
-
def _select_custom_logger_for_cold_storage(
self,
) -> Optional[_custom_logger_compatible_callbacks_literal]:
diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py
index 0ba77dcd2f0..48f12d44370 100644
--- a/litellm/proxy/spend_tracking/spend_management_endpoints.py
+++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py
@@ -3,7 +3,17 @@ import collections
import json
import os
from datetime import datetime, timedelta, timezone
-from typing import TYPE_CHECKING, Any, Dict, List, Literal, Optional
+from typing import (
+ TYPE_CHECKING,
+ Any,
+ Dict,
+ List,
+ Literal,
+ Mapping,
+ NamedTuple,
+ Optional,
+ Union,
+)
import fastapi
from fastapi import APIRouter, Depends, HTTPException, Request, status
@@ -29,6 +39,7 @@ from litellm.repositories.verification_token_repository import (
if TYPE_CHECKING:
from litellm.proxy.proxy_server import PrismaClient
+ from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler
else:
PrismaClient = Any
@@ -2175,6 +2186,89 @@ async def ui_view_spend_logs(
raise handle_exception_on_proxy(e)
+class RequestResponsePayload(NamedTuple):
+ messages: Optional[Union[str, list, dict]]
+ response: Optional[Union[str, list, dict]]
+ proxy_server_request: Optional[Union[str, dict]]
+
+
+_EMPTY_SPEND_LOG_VALUES = frozenset({"", "{}", "[]", "null"})
+
+
+def _spend_log_field_has_content(value: Optional[Union[str, list, dict]]) -> bool:
+ if value is None:
+ return False
+ if isinstance(value, str):
+ return value.strip() not in _EMPTY_SPEND_LOG_VALUES
+ if isinstance(value, (list, dict)):
+ return len(value) > 0
+ return True
+
+
+def _cold_storage_object_key_from_metadata(
+ metadata: Optional[Union[str, dict]],
+) -> Optional[str]:
+ if isinstance(metadata, str):
+ try:
+ metadata = json.loads(metadata)
+ except (json.JSONDecodeError, TypeError):
+ return None
+ if not isinstance(metadata, dict):
+ return None
+ object_key = metadata.get("cold_storage_object_key")
+ return object_key if isinstance(object_key, str) and object_key else None
+
+
+async def _resolve_request_response_payload(
+ row: Mapping[str, Any],
+ cold_storage_handler: "ColdStorageHandler",
+) -> RequestResponsePayload:
+ """
+ Decide where the prompt/response come from for a single spend-log row.
+
+ PG holds the content when ``store_prompts_in_spend_logs`` is on; otherwise it
+ holds ``"{}"`` placeholders and the real payload lives in cold storage keyed
+ by ``metadata.cold_storage_object_key``. The choice is made on actual row
+ content, not config flags, so historical and mixed-storage rows both resolve
+ correctly.
+ """
+ messages = row.get("messages")
+ response = row.get("response")
+ proxy_server_request = row.get("proxy_server_request")
+
+ pg_payload = RequestResponsePayload(messages, response, proxy_server_request)
+ if (
+ _spend_log_field_has_content(messages)
+ or _spend_log_field_has_content(response)
+ or _spend_log_field_has_content(proxy_server_request)
+ ):
+ return pg_payload
+
+ object_key = _cold_storage_object_key_from_metadata(row.get("metadata"))
+ if object_key is None:
+ return pg_payload
+
+ try:
+ payload = await cold_storage_handler.get_proxy_server_request_from_cold_storage_with_object_key(
+ object_key=object_key
+ )
+ except Exception:
+ verbose_proxy_logger.warning(
+ "Failed to fetch cold storage payload for key %s; falling back to DB values",
+ object_key,
+ exc_info=True,
+ )
+ return pg_payload
+ if payload is None:
+ return pg_payload
+
+ return RequestResponsePayload(
+ messages=payload.get("messages"),
+ response=payload.get("response"),
+ proxy_server_request=payload.get("proxy_server_request"),
+ )
+
+
@router.get(
"/spend/logs/ui/{request_id}",
tags=["Budget & Spend Tracking"],
@@ -2241,26 +2335,27 @@ async def ui_view_request_response_for_request_id(
if payload is not None:
return payload
- # Fallback: fetch heavy columns directly from the database.
- # The list endpoint (/spend/logs/ui) intentionally excludes messages,
- # response, and proxy_server_request for performance. When no custom
- # logger (S3, GCS, etc.) is configured, we still need to serve these
- # fields from the DB for the detail/drawer view.
+ # Fallback: the list endpoint omits the heavy columns for performance, so
+ # serve them here. When prompts were offloaded to cold storage the DB holds
+ # only placeholders, so _resolve_request_response_payload fetches the real
+ # payload from the configured cold storage backend by object key.
if prisma_client is not None:
+ from litellm.proxy.spend_tracking.cold_storage_handler import (
+ ColdStorageHandler,
+ )
+
sql_query = """
- SELECT messages, response, proxy_server_request
+ SELECT messages, response, proxy_server_request, metadata
FROM "LiteLLM_SpendLogs"
WHERE request_id = $1
LIMIT 1
"""
db_result = await prisma_client.db.query_raw(sql_query, request_id)
if db_result and len(db_result) > 0:
- row = db_result[0]
- return {
- "messages": row.get("messages"),
- "response": row.get("response"),
- "proxy_server_request": row.get("proxy_server_request"),
- }
+ resolved = await _resolve_request_response_payload(
+ db_result[0], cold_storage_handler=ColdStorageHandler()
+ )
+ return resolved._asdict()
return None
diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py
index 0b583129591..b9716c22cee 100644
--- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py
+++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py
@@ -3786,3 +3786,298 @@ async def test_ui_view_spend_logs_metadata_invalid_json_falls_back_to_empty_dict
assert body["data"][0]["metadata"] == {}
finally:
app.dependency_overrides.pop(ps.user_api_key_auth, None)
+
+
+class _FakeColdStorageLogger:
+ """Injectable cold storage logger that records the object key it was asked for."""
+
+ def __init__(self, payload):
+ self._payload = payload
+ self.requested_object_keys = []
+
+ async def get_proxy_server_request_from_cold_storage_with_object_key(
+ self, object_key
+ ):
+ self.requested_object_keys.append(object_key)
+ return self._payload
+
+
+def _cold_storage_handler(payload):
+ from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler
+
+ logger = _FakeColdStorageLogger(payload)
+ return ColdStorageHandler(cold_storage_logger=logger), logger
+
+
+@pytest.mark.parametrize(
+ "value, expected",
+ [
+ (None, False),
+ ("", False),
+ (" ", False),
+ ("{}", False),
+ ("[]", False),
+ ("null", False),
+ ('{"a": 1}', True),
+ ({}, False),
+ ({"a": 1}, True),
+ ([], False),
+ ([1], True),
+ (5, True),
+ ],
+)
+def test_spend_log_field_has_content(value, expected):
+ assert spend_management_endpoints._spend_log_field_has_content(value) is expected
+
+
+@pytest.mark.parametrize(
+ "metadata, expected",
+ [
+ (None, None),
+ ("{}", None),
+ ("not-json", None),
+ ({"cold_storage_object_key": ""}, None),
+ ({"cold_storage_object_key": "k/req-1.json"}, "k/req-1.json"),
+ ('{"cold_storage_object_key": "k/req-2.json"}', "k/req-2.json"),
+ ],
+)
+def test_cold_storage_object_key_from_metadata(metadata, expected):
+ assert (
+ spend_management_endpoints._cold_storage_object_key_from_metadata(metadata)
+ == expected
+ )
+
+
+@pytest.mark.asyncio
+async def test_resolve_payload_prefers_pg_and_skips_cold_storage():
+ handler, logger = _cold_storage_handler({"messages": "X", "response": "Y"})
+ row = {
+ "messages": "{}",
+ "response": '{"choices": [{"message": {"content": "hi"}}]}',
+ "proxy_server_request": "{}",
+ "metadata": {"cold_storage_object_key": "k/req.json"},
+ }
+
+ resolved = await spend_management_endpoints._resolve_request_response_payload(
+ row, cold_storage_handler=handler
+ )
+
+ assert resolved.response == '{"choices": [{"message": {"content": "hi"}}]}'
+ assert logger.requested_object_keys == []
+
+
+@pytest.mark.asyncio
+async def test_resolve_payload_fetches_from_cold_storage_when_pg_empty():
+ cold_payload = {
+ "messages": [{"role": "user", "content": "what is 2+2"}],
+ "response": {"choices": [{"message": {"content": "4"}}]},
+ "proxy_server_request": {"body": {"model": "gpt-4o-mini"}},
+ }
+ handler, logger = _cold_storage_handler(cold_payload)
+ row = {
+ "messages": "{}",
+ "response": "{}",
+ "proxy_server_request": "{}",
+ "metadata": {"cold_storage_object_key": "llm-gateway/prod/req-42.json"},
+ }
+
+ resolved = await spend_management_endpoints._resolve_request_response_payload(
+ row, cold_storage_handler=handler
+ )
+
+ assert logger.requested_object_keys == ["llm-gateway/prod/req-42.json"]
+ assert resolved.messages == cold_payload["messages"]
+ assert resolved.response == cold_payload["response"]
+ assert resolved.proxy_server_request == cold_payload["proxy_server_request"]
+
+
+@pytest.mark.asyncio
+async def test_resolve_payload_metadata_as_json_string():
+ cold_payload = {"messages": "in", "response": "out", "proxy_server_request": None}
+ handler, logger = _cold_storage_handler(cold_payload)
+ row = {
+ "messages": "{}",
+ "response": "{}",
+ "proxy_server_request": "{}",
+ "metadata": json.dumps({"cold_storage_object_key": "k/str-meta.json"}),
+ }
+
+ resolved = await spend_management_endpoints._resolve_request_response_payload(
+ row, cold_storage_handler=handler
+ )
+
+ assert logger.requested_object_keys == ["k/str-meta.json"]
+ assert resolved.response == "out"
+
+
+@pytest.mark.asyncio
+async def test_resolve_payload_no_object_key_returns_empty_without_fetch():
+ handler, logger = _cold_storage_handler({"messages": "should-not-be-used"})
+ row = {
+ "messages": "{}",
+ "response": "{}",
+ "proxy_server_request": "{}",
+ "metadata": {},
+ }
+
+ resolved = await spend_management_endpoints._resolve_request_response_payload(
+ row, cold_storage_handler=handler
+ )
+
+ assert logger.requested_object_keys == []
+ assert resolved == spend_management_endpoints.RequestResponsePayload(
+ "{}", "{}", "{}"
+ )
+
+
+@pytest.mark.asyncio
+async def test_resolve_payload_cold_storage_miss_falls_back_to_pg_values():
+ handler, logger = _cold_storage_handler(None)
+ row = {
+ "messages": "{}",
+ "response": "{}",
+ "proxy_server_request": "{}",
+ "metadata": {"cold_storage_object_key": "k/missing.json"},
+ }
+
+ resolved = await spend_management_endpoints._resolve_request_response_payload(
+ row, cold_storage_handler=handler
+ )
+
+ assert logger.requested_object_keys == ["k/missing.json"]
+ assert resolved == spend_management_endpoints.RequestResponsePayload(
+ "{}", "{}", "{}"
+ )
+
+
+@pytest.mark.asyncio
+async def test_resolve_payload_cold_storage_exception_falls_back_to_pg_values():
+ """A backend error during fetch degrades to PG values instead of bubbling a 500."""
+
+ class _RaisingLogger:
+ async def get_proxy_server_request_from_cold_storage_with_object_key(
+ self, object_key
+ ):
+ raise RuntimeError("cold storage backend unavailable")
+
+ from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler
+
+ handler = ColdStorageHandler(cold_storage_logger=_RaisingLogger())
+ row = {
+ "messages": "{}",
+ "response": "{}",
+ "proxy_server_request": "{}",
+ "metadata": {"cold_storage_object_key": "k/boom.json"},
+ }
+
+ resolved = await spend_management_endpoints._resolve_request_response_payload(
+ row, cold_storage_handler=handler
+ )
+
+ assert resolved == spend_management_endpoints.RequestResponsePayload(
+ "{}", "{}", "{}"
+ )
+
+
+@pytest.mark.asyncio
+async def test_cold_storage_handler_uses_injected_logger():
+ from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler
+
+ logger = _FakeColdStorageLogger({"messages": "in", "response": "out"})
+ handler = ColdStorageHandler(cold_storage_logger=logger)
+
+ result = await handler.get_proxy_server_request_from_cold_storage_with_object_key(
+ object_key="k/req.json"
+ )
+
+ assert result == {"messages": "in", "response": "out"}
+ assert logger.requested_object_keys == ["k/req.json"]
+
+
+@pytest.mark.asyncio
+async def test_cold_storage_handler_returns_none_when_no_logger_configured(monkeypatch):
+ from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler
+
+ monkeypatch.setattr(litellm, "cold_storage_custom_logger", None, raising=False)
+ handler = ColdStorageHandler()
+
+ result = await handler.get_proxy_server_request_from_cold_storage_with_object_key(
+ object_key="k/req.json"
+ )
+
+ assert result is None
+
+
+@pytest.mark.asyncio
+async def test_cold_storage_handler_resolves_configured_logger_from_registry(monkeypatch):
+ from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler
+
+ logger = _FakeColdStorageLogger({"messages": "from-registry"})
+ monkeypatch.setattr(litellm, "cold_storage_custom_logger", "s3_v2", raising=False)
+ monkeypatch.setattr(
+ litellm.logging_callback_manager,
+ "get_active_custom_logger_for_callback_name",
+ lambda name: logger if name == "s3_v2" else None,
+ )
+ handler = ColdStorageHandler()
+
+ result = await handler.get_proxy_server_request_from_cold_storage_with_object_key(
+ object_key="k/req.json"
+ )
+
+ assert result == {"messages": "from-registry"}
+ assert logger.requested_object_keys == ["k/req.json"]
+
+
+def test_ui_view_request_response_reads_from_cold_storage(client, monkeypatch):
+ """End-to-end: a placeholder row with a cold_storage_object_key is served from
+ cold storage through the detail endpoint."""
+ from types import SimpleNamespace
+
+ placeholder_row = {
+ "messages": "{}",
+ "response": "{}",
+ "proxy_server_request": "{}",
+ "metadata": {"cold_storage_object_key": "k/cold.json"},
+ }
+
+ async def _query_raw(_sql, *_args):
+ return [placeholder_row]
+
+ fake_prisma = SimpleNamespace(db=SimpleNamespace(query_raw=_query_raw))
+ monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", fake_prisma)
+
+ cold_logger = _FakeColdStorageLogger(
+ {
+ "messages": [{"role": "user", "content": "hi"}],
+ "response": {"choices": [{"message": {"content": "hello"}}]},
+ "proxy_server_request": None,
+ }
+ )
+ monkeypatch.setattr(litellm, "cold_storage_custom_logger", "s3_v2", raising=False)
+ monkeypatch.setattr(
+ litellm.logging_callback_manager,
+ "get_active_additional_logging_utils_from_custom_logger",
+ lambda: [],
+ )
+ monkeypatch.setattr(
+ litellm.logging_callback_manager,
+ "get_active_custom_logger_for_callback_name",
+ lambda name: cold_logger if name == "s3_v2" else None,
+ )
+
+ app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth(
+ user_role=LitellmUserRoles.PROXY_ADMIN, user_id="admin_1"
+ )
+ try:
+ response = client.get(
+ "/spend/logs/ui/req-cold",
+ headers={"Authorization": "Bearer sk-test"},
+ )
+ assert response.status_code == 200
+ body = response.json()
+ assert body["messages"] == [{"role": "user", "content": "hi"}]
+ assert body["response"] == {"choices": [{"message": {"content": "hello"}}]}
+ assert cold_logger.requested_object_keys == ["k/cold.json"]
+ finally:
+ app.dependency_overrides.pop(ps.user_api_key_auth, None)
diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/PrettyMessagesView.test.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/PrettyMessagesView.test.tsx
index b73dcafcdc3..e7295ed7a72 100644
--- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/PrettyMessagesView.test.tsx
+++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/PrettyMessagesView.test.tsx
@@ -27,6 +27,17 @@ describe("PrettyMessagesView", () => {
expect(screen.getByText("Hi there!")).toBeInTheDocument();
});
+ it("renders input when request is a bare messages array (cold storage payload)", () => {
+ const request = [{ role: "user", content: "Write me a poem" }];
+ const response = {
+ choices: [{ message: { role: "assistant", content: "A quiet moment." } }],
+ };
+
+ render();
+ expect(screen.getByText("Write me a poem")).toBeInTheDocument();
+ expect(screen.getByText("A quiet moment.")).toBeInTheDocument();
+ });
+
it("should render the realtime pretty view for realtime API responses", () => {
const request = {};
const response = {
diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/prettyMessagesUtils.ts b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/prettyMessagesUtils.ts
index 32ae294b1ee..09b8f551c1d 100644
--- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/prettyMessagesUtils.ts
+++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/prettyMessagesUtils.ts
@@ -39,18 +39,24 @@ export const ROLE_STYLES: Record = {
* Parse request messages and response message from log data
*/
export const parseMessages = (request: any, response: any): ParsedMessages => {
- // Parse request messages
+ // Parse request messages. `request` is either the raw request body
+ // ({ messages: [...] }) or, when prompts come from cold storage, the bare
+ // messages array itself.
const requestMessages: ParsedMessage[] = [];
- if (request?.messages && Array.isArray(request.messages)) {
- request.messages.forEach((msg: any) => {
- requestMessages.push({
- role: msg.role || "user",
- content: parseMessageContent(msg.content),
- toolCallId: msg.tool_call_id,
- });
+ const requestMessageList = Array.isArray(request)
+ ? request
+ : Array.isArray(request?.messages)
+ ? request.messages
+ : [];
+
+ requestMessageList.forEach((msg: any) => {
+ requestMessages.push({
+ role: msg.role || "user",
+ content: parseMessageContent(msg.content),
+ toolCallId: msg.tool_call_id,
});
- }
+ });
// Parse response message
let responseMessage: ParsedMessage | null = null;
From 1eb6bdde9c46aa4417e4ffb049ac777ea4cd1ec6 Mon Sep 17 00:00:00 2001
From: Praveen Ghuge
Date: Wed, 24 Jun 2026 16:36:52 +0530
Subject: [PATCH 25/30] fix(mavvrik): advance metricsMarker after upload; fix
scheduler startup (#31068)
* fix(mavvrik): advance metricsMarker after upload + fix scheduler startup
Two bugs fixed:
1. deliver() never called PATCH /metrics/agent/ai/{connectionId} after a
successful GCS upload, so metricsMarker stayed at 0 and every daily run
re-exported the same dates in an infinite catch-up loop.
Fix: add _update_metrics_marker(date_epoch) called at the end of deliver()
after _upload_to_gcs() succeeds. A 4xx warns but does not raise (the GCS
file is already committed). A 410 raises consistent with the rest of the
destination.
2. init_mavvrik_focus_background_job runs at proxy startup before any LLM call
has triggered lazy instantiation of MavvrikFocusLogger, so it found no
logger instance and silently skipped registering the daily export job.
Fix: if no instance is found but "mavvrik" is in litellm.callbacks, call
_init_custom_logger_compatible_class to force instantiation before
the APScheduler job is registered.
* fix(mavvrik): catch up from earliest window when metricsMarker=0
When the connector is freshly registered, metricsMarker=0 parses to None.
The catch-up block was guarded by `if last_ingested and ...` which skipped
it entirely for None, so only yesterday was exported instead of the full
_MAX_CATCHUP_DAYS window.
Fix: treat None as being _MAX_CATCHUP_DAYS behind (start from earliest_catchup).
The existing > 7 day warning only fires for non-None markers that are old.
* fix(mavvrik): use now as end_time for yesterday's export window
LiteLLM_DailyUserSpend rows for a given date get their updated_at
bumped by the spend flush job throughout the next morning. The core
database query filters on updated_at, so capping end_time at midnight
(yesterday + 1 day) missed any spend rows flushed after midnight.
Fix: pass now (cron fire time) as end_time for the daily "yesterday"
window so all fully-settled rows are captured regardless of when the
flush job ran.
Verified: claude-3-5-sonnet BilledCost went from 0.0 to ~$2.40 per
row in the exported FOCUS CSV.
* fix(mavvrik): also use now as end_time for catch-up windows
* fix(mavvrik_focus): pass required args to _init_custom_logger_compatible_class
Calling it with only logging_integration raised TypeError at proxy startup
because internal_usage_cache and llm_router have no defaults. Also fix test
name to reflect the actual status code (5xx not 4xx) used in the mock.
* ci: retrigger CI run
---
.../focus/destinations/mavvrik_destination.py | 36 ++++++--
.../mavvrik_focus/mavvrik_focus_logger.py | 51 ++++++++---
.../focus/test_mavvrik_destination.py | 86 ++++++++++++++++---
3 files changed, 143 insertions(+), 30 deletions(-)
diff --git a/litellm/integrations/focus/destinations/mavvrik_destination.py b/litellm/integrations/focus/destinations/mavvrik_destination.py
index 1e3c98b9a70..659f608a3e1 100644
--- a/litellm/integrations/focus/destinations/mavvrik_destination.py
+++ b/litellm/integrations/focus/destinations/mavvrik_destination.py
@@ -3,6 +3,7 @@
Flow:
1. GET /metrics/agent/ai/{connection_id}/upload-url → GCS signed URL
2. PUT with CSV content
+ 3. PATCH /metrics/agent/ai/{connection_id} → advance metricsMarker
"""
from __future__ import annotations
@@ -127,8 +128,6 @@ class FocusMavvrikDestination(FocusDestination):
timeout=30.0,
)
if resp.status_code == 410:
- # Connector has been disconnected in Mavvrik — reset flag so next
- # delivery attempt re-registers after it becomes active again.
self._registered = False
raise RuntimeError(
"Mavvrik FOCUS destination: connector is disconnected (410). "
@@ -273,14 +272,35 @@ class FocusMavvrikDestination(FocusDestination):
pass
raise
+ async def _update_metrics_marker(self, date_epoch: int) -> None:
+ """PATCH agent endpoint to advance metricsMarker after a successful upload."""
+ resp = await self._http.client.request(
+ method="PATCH",
+ url=self._agent_url,
+ headers=self._auth_headers,
+ json={"metricsMarker": date_epoch},
+ timeout=30.0,
+ )
+ if resp.status_code == 410:
+ self._registered = False
+ raise RuntimeError(
+ "Mavvrik FOCUS destination: connector is disconnected (410). "
+ "Re-enable the connection in the Mavvrik dashboard."
+ )
+ if resp.status_code >= 400:
+ verbose_logger.warning(
+ "Mavvrik FOCUS destination: failed to update metricsMarker (%s): %s",
+ resp.status_code,
+ resp.text[:200],
+ )
+ return
+ verbose_logger.debug(
+ "Mavvrik FOCUS destination: metricsMarker advanced to %s", date_epoch
+ )
+
async def get_metrics_marker(self) -> Optional[int]:
"""Register with Mavvrik and return the current metricsMarker.
- The metricsMarker is a Unix timestamp (seconds) representing the last
- date Mavvrik has successfully ingested. Called on every scheduled run
- so the logger can detect and catch up any dates missed due to previous
- export failures.
-
Always calls the Mavvrik register API — unlike deliver() which skips
registration once _registered is True, catch-up requires a fresh
marker value on every run.
@@ -328,6 +348,7 @@ class FocusMavvrikDestination(FocusDestination):
return
date_str = time_window.start_time.strftime("%Y-%m-%d")
+ date_epoch = int(time_window.start_time.timestamp())
verbose_logger.debug(
"Mavvrik FOCUS destination: uploading %d bytes for date=%s (%s)",
@@ -339,6 +360,7 @@ class FocusMavvrikDestination(FocusDestination):
await self._ensure_registered()
signed_url = await self._get_signed_url(date_str)
await self._upload_to_gcs(signed_url, content)
+ await self._update_metrics_marker(date_epoch)
verbose_logger.debug(
"Mavvrik FOCUS destination: upload complete for date=%s", date_str
diff --git a/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py b/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py
index 47d3e1da7bc..bbc9d1a6330 100644
--- a/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py
+++ b/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py
@@ -149,8 +149,8 @@ class MavvrikFocusLogger(FocusLogger):
On each run:
1. Register with Mavvrik → get metricsMarker (last successfully ingested date)
- 2. If metricsMarker is behind yesterday, catch up missed dates (capped at
- _MAX_CATCHUP_DAYS to avoid runaway loops on long outages)
+ 2. If metricsMarker is behind yesterday (or 0/None for a fresh connector),
+ catch up missed dates (capped at _MAX_CATCHUP_DAYS)
3. Export yesterday (today's daily window)
This ensures a failed export on day N is automatically retried on day N+1
@@ -177,13 +177,21 @@ class MavvrikFocusLogger(FocusLogger):
last_ingested = _parse_metrics_marker(marker)
- # Catch up missed dates, capped at _MAX_CATCHUP_DAYS
- if last_ingested and last_ingested < yesterday:
- # Never go further back than _MAX_CATCHUP_DAYS from yesterday
- earliest_catchup = yesterday - timedelta(days=self._MAX_CATCHUP_DAYS - 1)
- catch_up_date = max(last_ingested + timedelta(days=1), earliest_catchup)
+ # Catch up missed dates, capped at _MAX_CATCHUP_DAYS.
+ # last_ingested=None means metricsMarker=0 (fresh connector, never ingested) —
+ # treat the same as being _MAX_CATCHUP_DAYS behind so we export all available history.
+ earliest_catchup = yesterday - timedelta(days=self._MAX_CATCHUP_DAYS - 1)
+ if last_ingested is None or last_ingested < yesterday:
+ catch_up_date = (
+ earliest_catchup
+ if last_ingested is None
+ else max(last_ingested + timedelta(days=1), earliest_catchup)
+ )
- if last_ingested + timedelta(days=1) < earliest_catchup:
+ if (
+ last_ingested is not None
+ and last_ingested + timedelta(days=1) < earliest_catchup
+ ):
verbose_proxy_logger.warning(
"Mavvrik FOCUS export: metricsMarker is more than %d days behind "
"(%s). Catching up from %s only; earlier data will not be re-exported.",
@@ -197,18 +205,24 @@ class MavvrikFocusLogger(FocusLogger):
"Mavvrik FOCUS export: catching up missed date %s",
catch_up_date.date(),
)
+ # Use now as end_time for catch-up windows too — rows for old dates
+ # may have been flushed to DB well after their calendar day ended.
+ catch_up_end = min(catch_up_date + timedelta(days=1), now)
window = FocusTimeWindow(
start_time=catch_up_date,
- end_time=catch_up_date + timedelta(days=1),
+ end_time=catch_up_end,
frequency="daily",
)
await self._export_window(window=window, limit=None)
catch_up_date += timedelta(days=1)
- # Export yesterday's window (the normal daily run)
+ # Export yesterday's window (the normal daily run).
+ # Use `now` as end_time so spend rows flushed after midnight are included.
+ # LiteLLM's DailyUserSpend rows for a given date keep getting updated_at
+ # bumped as the flush job runs; capping at midnight would miss those updates.
window = FocusTimeWindow(
start_time=yesterday,
- end_time=yesterday + timedelta(days=1),
+ end_time=now,
frequency="daily",
)
await self._export_window(window=window, limit=None)
@@ -253,6 +267,21 @@ class MavvrikFocusLogger(FocusLogger):
)
if type(cb) is MavvrikFocusLogger
]
+ if not loggers and "mavvrik" in litellm.callbacks:
+ # The logger is registered as the string "mavvrik" but hasn't been
+ # instantiated yet (lazy init happens on first LLM call). Force it now
+ # so the scheduler can register the daily export job at startup.
+ from litellm.litellm_core_utils.litellm_logging import ( # noqa: PLC0415
+ _init_custom_logger_compatible_class,
+ )
+
+ instance = _init_custom_logger_compatible_class(
+ logging_integration="mavvrik",
+ internal_usage_cache=None,
+ llm_router=None,
+ )
+ if isinstance(instance, MavvrikFocusLogger):
+ loggers = [instance]
if not loggers:
verbose_proxy_logger.debug(
"No MavvrikFocusLogger registered; skipping scheduler"
diff --git a/tests/test_litellm/integrations/focus/test_mavvrik_destination.py b/tests/test_litellm/integrations/focus/test_mavvrik_destination.py
index 797238ae238..3a23dc4ffb2 100644
--- a/tests/test_litellm/integrations/focus/test_mavvrik_destination.py
+++ b/tests/test_litellm/integrations/focus/test_mavvrik_destination.py
@@ -34,6 +34,12 @@ def _dest(**overrides) -> FocusMavvrikDestination:
return FocusMavvrikDestination(prefix="mavvrik_focus_exports", config=config)
+def _patch_resp(status: int = 204) -> MagicMock:
+ r = MagicMock()
+ r.status_code = status
+ return r
+
+
def test_missing_api_key_raises():
with pytest.raises(ValueError, match="MAVVRIK_API_KEY"):
FocusMavvrikDestination(
@@ -127,6 +133,8 @@ async def test_large_content_uploads_in_multiple_chunks():
chunk2_resp = MagicMock()
chunk2_resp.status_code = 200
+ patch_resp = _patch_resp(204)
+
mock_http = MagicMock()
mock_http.client = MagicMock()
mock_http.client.request = AsyncMock(
@@ -136,6 +144,7 @@ async def test_large_content_uploads_in_multiple_chunks():
init_resp,
chunk1_resp,
chunk2_resp,
+ patch_resp,
]
)
dest._http = mock_http
@@ -152,15 +161,20 @@ async def test_large_content_uploads_in_multiple_chunks():
filename="usage.csv",
)
- # register + get_signed_url + init + 2 chunk PUTs = 5 calls
- assert mock_http.client.request.call_count == 5
+ # register + get_signed_url + init + 2 chunk PUTs + PATCH = 6 calls
+ assert mock_http.client.request.call_count == 6
- # Check Content-Range headers
- put_calls = mock_http.client.request.call_args_list[3:]
+ # Check Content-Range headers on the chunk PUTs (calls 3 and 4)
+ put_calls = mock_http.client.request.call_args_list[3:5]
assert "bytes" in put_calls[0].kwargs["headers"]["Content-Range"]
assert "/*" in put_calls[0].kwargs["headers"]["Content-Range"] # intermediate
assert "/*" not in put_calls[1].kwargs["headers"]["Content-Range"] # final
+ # Verify the PATCH call advanced metricsMarker
+ patch_call = mock_http.client.request.call_args_list[5]
+ assert patch_call.kwargs["method"] == "PATCH"
+ assert "metricsMarker" in patch_call.kwargs["json"]
+
@pytest.mark.asyncio
async def test_deliver_calls_register_get_url_and_upload():
@@ -180,12 +194,14 @@ async def test_deliver_calls_register_get_url_and_upload():
upload_resp = MagicMock()
upload_resp.status_code = 200
+ patch_resp = _patch_resp(204)
+
mock_http = MagicMock()
mock_http.client = MagicMock()
- # All 4 calls go through self._http.client.request:
- # 1. register, 2. get_signed_url, 3. GCS session init POST, 4. GCS PUT
+ # All 5 calls go through self._http.client.request:
+ # 1. register, 2. get_signed_url, 3. GCS session init POST, 4. GCS PUT, 5. PATCH marker
mock_http.client.request = AsyncMock(
- side_effect=[register_resp, signed_url_resp, init_resp, upload_resp]
+ side_effect=[register_resp, signed_url_resp, init_resp, upload_resp, patch_resp]
)
dest._http = mock_http
@@ -196,10 +212,14 @@ async def test_deliver_calls_register_get_url_and_upload():
)
assert dest._registered is True
- assert mock_http.client.request.call_count == 4
+ assert mock_http.client.request.call_count == 5
# Verify Content-Range header was set on the PUT
put_call = mock_http.client.request.call_args_list[3]
assert "Content-Range" in put_call.kwargs["headers"]
+ # Verify PATCH was called last with metricsMarker
+ patch_call = mock_http.client.request.call_args_list[4]
+ assert patch_call.kwargs["method"] == "PATCH"
+ assert "metricsMarker" in patch_call.kwargs["json"]
@pytest.mark.asyncio
@@ -224,17 +244,19 @@ async def test_register_called_only_once_across_multiple_deliveries():
mock_http = MagicMock()
mock_http.client = MagicMock()
- # First delivery: register, get_signed_url, GCS init, GCS PUT
- # Second delivery: get_signed_url, GCS init, GCS PUT (register skipped)
+ # First delivery: register, get_signed_url, GCS init, GCS PUT, PATCH
+ # Second delivery: get_signed_url, GCS init, GCS PUT, PATCH (register skipped)
mock_http.client.request = AsyncMock(
side_effect=[
register_resp,
_signed_url_resp(),
init_resp,
upload_resp,
+ _patch_resp(204),
_signed_url_resp(),
init_resp,
upload_resp,
+ _patch_resp(204),
]
)
dest._http = mock_http
@@ -243,8 +265,8 @@ async def test_register_called_only_once_across_multiple_deliveries():
await dest.deliver(content=b"header\nrow1\n", time_window=window, filename="1.csv")
await dest.deliver(content=b"header\nrow2\n", time_window=window, filename="2.csv")
- # 7 total: register(1) + [get_url+init+put](2) × 2 deliveries
- assert mock_http.client.request.call_count == 7
+ # 9 total: register(1) + [get_url+init+put+patch](4) × 2 deliveries
+ assert mock_http.client.request.call_count == 9
# First call was register
first_call = mock_http.client.request.call_args_list[0]
assert first_call.kwargs["method"] == "POST"
@@ -749,3 +771,43 @@ async def test_gcs_session_cancelled_on_chunk_failure():
delete_call = calls[4]
assert delete_call.kwargs["method"] == "DELETE"
assert "storage.googleapis.com/session" in delete_call.kwargs["url"]
+
+
+@pytest.mark.asyncio
+async def test_update_metrics_marker_warns_on_non_410_error():
+ """_update_metrics_marker must log a warning on any >=400 (non-410) status but not raise."""
+ dest = _dest()
+
+ fail_resp = MagicMock()
+ fail_resp.status_code = 500
+ fail_resp.text = "Internal Server Error"
+
+ mock_http = MagicMock()
+ mock_http.client = MagicMock()
+ mock_http.client.request = AsyncMock(return_value=fail_resp)
+ dest._http = mock_http
+
+ # Must not raise — warning only
+ await dest._update_metrics_marker(1234567890)
+ assert mock_http.client.request.call_count == 1
+
+
+@pytest.mark.asyncio
+async def test_update_metrics_marker_raises_on_410():
+ """_update_metrics_marker must raise RuntimeError and reset _registered on 410."""
+ dest = _dest()
+ dest._registered = True
+
+ resp_410 = MagicMock()
+ resp_410.status_code = 410
+ resp_410.text = "Gone"
+
+ mock_http = MagicMock()
+ mock_http.client = MagicMock()
+ mock_http.client.request = AsyncMock(return_value=resp_410)
+ dest._http = mock_http
+
+ with pytest.raises(RuntimeError, match="disconnected"):
+ await dest._update_metrics_marker(1234567890)
+
+ assert dest._registered is False
From e1187c0462a1c456c1eb74f09ac9d907fe0ce64a Mon Sep 17 00:00:00 2001
From: Jim Smith
Date: Wed, 24 Jun 2026 07:15:38 -0400
Subject: [PATCH 26/30] feat: pass through optional `instruction` field in the
rerank API (vLLM/Qwen3-Reranker) (#30757)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
* Add optional `instruction` passthrough to the rerank API
vLLM's /v1/rerank and /v1/score accept an optional top-level `instruction`
field (folded into the model's chat_template_kwargs and consumed by the
chat template — e.g. Qwen3-Reranker). LiteLLM's managed rerank route silently
dropped it: RerankRequest / OptionalRerankParams had no such field, so the
outgoing body was rebuilt without it.
Thread an opt-in `instruction: Optional[str]` through rerank()/arerank(),
get_optional_rerank_params, and the hosted_vllm transformation into the
request body, only when non-None. When callers omit it, model_dump(exclude_none)
drops the field and the outgoing request is byte-for-byte unchanged — fully
backward-compatible. (DeepInfra already forwards `instruction` via
non_default_params; this formalizes the field in the shared types.)
Co-Authored-By: Claude Opus 4.8 (1M context)
* Address review: thread `instruction` as a typed param + cover rerank_utils
Per PR review (greptile P2 + codecov):
- Make `instruction` a typed, named argument on the rerank provider interface
instead of recovering it from the opaque `non_default_params` blob. Adds
`instruction: Optional[str] = None` to `BaseRerankConfig.map_cohere_rerank_params`
and every provider override, and forwards it explicitly from
`get_optional_rerank_params`. hosted_vllm now reads the named param directly.
It is still also surfaced in `non_default_params` so providers that read it
there (e.g. DeepInfra) keep working now that `rerank()` consumes `instruction`
as a named param rather than leaving it in **kwargs.
- Add get_optional_rerank_params unit tests (present + absent) to cover the
previously-uncovered threading line flagged by codecov.
Co-Authored-By: Claude Opus 4.8 (1M context)
* fix: scan rerank `instruction` through request guardrails
The rerank guardrail translation (CohereRerankHandler.process_input_messages)
only scanned `query`, so the newly added `instruction` field reached the
backend model unscanned. Since instruction-aware rerankers (hosted vLLM /
Qwen3-Reranker) fold `instruction` into the prompt, an authenticated caller
could place content there to bypass configured rerank request guardrails.
Generalize the handler to scan every user-controlled text field (`query` and
`instruction`) in one apply_guardrail call and write each sanitized value back
by index. Query-only requests are unchanged (single-element list at index 0);
non-string fields are left untouched. Adds tests covering instruction
scanning, PII masking write-back, and the non-string case.
Addresses the Veria AI security review on PR #30757.
* test: narrow Optional results before len() to satisfy basedpyright budget
The lint gate (basedpyright delta-vs-base budget) flagged one new
reportArgumentType: len(result.results) where results is
List[RerankResponseResult] | None. Assert results is not None first to
narrow the type before len()/indexing.
* fix: read rerank `instruction` from kwargs to satisfy basedpyright budget
The basedpyright delta-vs-base gate flagged one new reportArgumentType: the
Router forwards rerank calls via an untyped `**kwargs` unpack
(`litellm.arerank(**{**data, **kwargs})`), and declaring `instruction` as a
typed named param on the public `rerank`/`arerank` entrypoints made pyright
check that key against `str | None`, adding an error at router.py with no real
safety gain. Read `instruction` from kwargs in `rerank` instead.
It remains fully typed where it matters - threaded as a typed argument through
`get_optional_rerank_params` and each provider's `map_cohere_rerank_params`
(the original Greptile P2 ask). Whole-repo reportArgumentType is back to the
base count (net 0); rerank hosted_vllm + cohere guardrail suites pass; ruff clean.
---------
Co-authored-by: Claude Opus 4.8 (1M context)
---
.../llms/base_llm/rerank/transformation.py | 1 +
.../rerank/guardrail_translation/handler.py | 76 +++++++++++-------
litellm/llms/cohere/rerank/transformation.py | 1 +
.../llms/cohere/rerank_v2/transformation.py | 1 +
.../llms/dashscope/rerank/transformation.py | 1 +
.../llms/deepinfra/rerank/transformation.py | 1 +
.../fireworks_ai/rerank/transformation.py | 1 +
.../llms/hosted_vllm/rerank/transformation.py | 25 ++++--
.../llms/huggingface/rerank/transformation.py | 1 +
litellm/llms/jina_ai/rerank/transformation.py | 1 +
.../llms/nvidia_nim/rerank/transformation.py | 1 +
.../llms/vertex_ai/rerank/transformation.py | 1 +
litellm/llms/voyage/rerank/transformation.py | 1 +
litellm/llms/watsonx/rerank/transformation.py | 1 +
litellm/rerank_api/main.py | 6 ++
litellm/rerank_api/rerank_utils.py | 7 ++
litellm/types/rerank.py | 5 ++
.../rerank/test_rerank_guardrail_handler.py | 70 +++++++++++++++-
.../test_hosted_vllm_rerank_transformation.py | 79 +++++++++++++++++++
19 files changed, 242 insertions(+), 38 deletions(-)
diff --git a/litellm/llms/base_llm/rerank/transformation.py b/litellm/llms/base_llm/rerank/transformation.py
index 166f876ba04..6603c64142b 100644
--- a/litellm/llms/base_llm/rerank/transformation.py
+++ b/litellm/llms/base_llm/rerank/transformation.py
@@ -85,6 +85,7 @@ class BaseRerankConfig(ABC):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
pass
diff --git a/litellm/llms/cohere/rerank/guardrail_translation/handler.py b/litellm/llms/cohere/rerank/guardrail_translation/handler.py
index e9a5823d2b8..0824e1cca41 100644
--- a/litellm/llms/cohere/rerank/guardrail_translation/handler.py
+++ b/litellm/llms/cohere/rerank/guardrail_translation/handler.py
@@ -26,11 +26,18 @@ class CohereRerankHandler(BaseTranslation):
The handler specifically processes:
- The 'query' parameter (string)
+ - The 'instruction' parameter (string), when present
Note: Documents are not processed by guardrails as they are the corpus
being searched, not user input.
"""
+ # User-controlled free-text fields that reach the model and must be
+ # scanned. 'instruction' is folded into the prompt by instruction-aware
+ # rerankers (e.g. hosted vLLM / Qwen3-Reranker), so it is as sensitive as
+ # 'query'; omitting it would let a caller smuggle content past guardrails.
+ _SCANNED_FIELDS = ("query", "instruction")
+
async def process_input_messages(
self,
data: dict,
@@ -38,42 +45,55 @@ class CohereRerankHandler(BaseTranslation):
litellm_logging_obj: Optional[Any] = None,
) -> Any:
"""
- Process input query by applying guardrails.
+ Process input text fields ('query' and 'instruction') by applying
+ guardrails and writing the sanitized values back.
Args:
- data: Request data dictionary containing 'query'
+ data: Request data dictionary containing 'query' and optionally
+ 'instruction'
guardrail_to_apply: The guardrail instance to apply
Returns:
- Modified data with guardrails applied to query only
+ Modified data with guardrails applied to query/instruction only
"""
- # Process query only
- query = data.get("query")
- if query is not None and isinstance(query, str):
- inputs = GenericGuardrailAPIInputs(texts=[query])
- # Include model information if available
- model = data.get("model")
- if model:
- inputs["model"] = model
- guardrailed_inputs = await guardrail_to_apply.apply_guardrail(
- inputs=inputs,
- request_data=data,
- input_type="request",
- logging_obj=litellm_logging_obj,
+ # Collect every scannable text field in a stable order so the
+ # guardrailed results can be written back to the right key by index.
+ fields_to_scan = [
+ (key, data[key])
+ for key in self._SCANNED_FIELDS
+ if isinstance(data.get(key), str)
+ ]
+ if not fields_to_scan:
+ verbose_proxy_logger.debug(
+ "Rerank: No query/instruction to process or not strings"
)
- guardrailed_texts = guardrailed_inputs.get("texts", [])
- data["query"] = guardrailed_texts[0] if guardrailed_texts else query
+ return data
- verbose_proxy_logger.debug(
- "Rerank: Applied guardrail to query. "
- "Original length: %d, New length: %d",
- len(query),
- len(data["query"]),
- )
- else:
- verbose_proxy_logger.debug(
- "Rerank: No query to process or query is not a string"
- )
+ inputs = GenericGuardrailAPIInputs(texts=[value for _, value in fields_to_scan])
+ # Include model information if available
+ model = data.get("model")
+ if model:
+ inputs["model"] = model
+ guardrailed_inputs = await guardrail_to_apply.apply_guardrail(
+ inputs=inputs,
+ request_data=data,
+ input_type="request",
+ logging_obj=litellm_logging_obj,
+ )
+ guardrailed_texts = guardrailed_inputs.get("texts", [])
+
+ for idx, (key, original) in enumerate(fields_to_scan):
+ # Defensive: only write back when the guardrail returned a value for
+ # this index; otherwise keep the original (never forward unscanned).
+ if idx < len(guardrailed_texts):
+ data[key] = guardrailed_texts[idx]
+ verbose_proxy_logger.debug(
+ "Rerank: Applied guardrail to %s. "
+ "Original length: %d, New length: %d",
+ key,
+ len(original),
+ len(data[key]),
+ )
return data
diff --git a/litellm/llms/cohere/rerank/transformation.py b/litellm/llms/cohere/rerank/transformation.py
index 64ae8e8ffa7..d875f420310 100644
--- a/litellm/llms/cohere/rerank/transformation.py
+++ b/litellm/llms/cohere/rerank/transformation.py
@@ -57,6 +57,7 @@ class CohereRerankConfig(BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
"""
Map Cohere rerank params
diff --git a/litellm/llms/cohere/rerank_v2/transformation.py b/litellm/llms/cohere/rerank_v2/transformation.py
index 4c800d6455d..0dcb10d5664 100644
--- a/litellm/llms/cohere/rerank_v2/transformation.py
+++ b/litellm/llms/cohere/rerank_v2/transformation.py
@@ -49,6 +49,7 @@ class CohereRerankV2Config(CohereRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
"""
Map Cohere rerank params
diff --git a/litellm/llms/dashscope/rerank/transformation.py b/litellm/llms/dashscope/rerank/transformation.py
index 629f3cf4af7..745e85de7e3 100644
--- a/litellm/llms/dashscope/rerank/transformation.py
+++ b/litellm/llms/dashscope/rerank/transformation.py
@@ -116,6 +116,7 @@ class DashScopeRerankConfig(BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
# qwen3-rerank accepts query/documents/top_n/return_documents. The
# rest (rank_fields, max_*_per_doc) are silently dropped.
diff --git a/litellm/llms/deepinfra/rerank/transformation.py b/litellm/llms/deepinfra/rerank/transformation.py
index e4bfbcb2513..a5c36ca2e5f 100644
--- a/litellm/llms/deepinfra/rerank/transformation.py
+++ b/litellm/llms/deepinfra/rerank/transformation.py
@@ -104,6 +104,7 @@ class DeepinfraRerankConfig(BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
# Start with the basic parameters
optional_rerank_params = {}
diff --git a/litellm/llms/fireworks_ai/rerank/transformation.py b/litellm/llms/fireworks_ai/rerank/transformation.py
index 4a7b64b9b77..27309780c86 100644
--- a/litellm/llms/fireworks_ai/rerank/transformation.py
+++ b/litellm/llms/fireworks_ai/rerank/transformation.py
@@ -67,6 +67,7 @@ class FireworksAIRerankConfig(FireworksAIMixin, BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict[str, Any]:
"""
Map Cohere rerank params to Fireworks AI rerank params
diff --git a/litellm/llms/hosted_vllm/rerank/transformation.py b/litellm/llms/hosted_vllm/rerank/transformation.py
index 60b6dc7d23d..d0c96f8b420 100644
--- a/litellm/llms/hosted_vllm/rerank/transformation.py
+++ b/litellm/llms/hosted_vllm/rerank/transformation.py
@@ -61,6 +61,7 @@ class HostedVLLMRerankConfig(BaseRerankConfig):
"top_n",
"rank_fields",
"return_documents",
+ "instruction",
]
def map_cohere_rerank_params(
@@ -76,6 +77,7 @@ class HostedVLLMRerankConfig(BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
"""
Map parameters for Hosted VLLM rerank
@@ -83,16 +85,22 @@ class HostedVLLMRerankConfig(BaseRerankConfig):
if max_chunks_per_doc is not None:
raise ValueError("Hosted VLLM does not support max_chunks_per_doc")
- return dict(
- OptionalRerankParams(
- query=query,
- documents=documents,
- top_n=top_n,
- rank_fields=rank_fields,
- return_documents=return_documents,
- )
+ mapped_params = OptionalRerankParams(
+ query=query,
+ documents=documents,
+ top_n=top_n,
+ rank_fields=rank_fields,
+ return_documents=return_documents,
)
+ # `instruction` is a vLLM-supported passthrough (folded into the model's
+ # chat_template_kwargs). Only forward it when explicitly set so omitting
+ # it leaves the request unchanged.
+ if instruction is not None:
+ mapped_params["instruction"] = instruction
+
+ return dict(mapped_params)
+
def validate_environment(
self,
headers: dict,
@@ -135,6 +143,7 @@ class HostedVLLMRerankConfig(BaseRerankConfig):
top_n=optional_rerank_params.get("top_n", None),
rank_fields=optional_rerank_params.get("rank_fields", None),
return_documents=optional_rerank_params.get("return_documents", None),
+ instruction=optional_rerank_params.get("instruction", None),
)
return rerank_request.model_dump(exclude_none=True)
diff --git a/litellm/llms/huggingface/rerank/transformation.py b/litellm/llms/huggingface/rerank/transformation.py
index 2c847b617ef..4e409f31ed2 100644
--- a/litellm/llms/huggingface/rerank/transformation.py
+++ b/litellm/llms/huggingface/rerank/transformation.py
@@ -100,6 +100,7 @@ class HuggingFaceRerankConfig(BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
optional_rerank_params = {}
if non_default_params is not None:
diff --git a/litellm/llms/jina_ai/rerank/transformation.py b/litellm/llms/jina_ai/rerank/transformation.py
index 56be754fc34..0d48ed5edcd 100644
--- a/litellm/llms/jina_ai/rerank/transformation.py
+++ b/litellm/llms/jina_ai/rerank/transformation.py
@@ -45,6 +45,7 @@ class JinaAIRerankConfig(BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
optional_params = {}
supported_params = self.get_supported_cohere_rerank_params(model)
diff --git a/litellm/llms/nvidia_nim/rerank/transformation.py b/litellm/llms/nvidia_nim/rerank/transformation.py
index fc317293acc..8eee188bf46 100644
--- a/litellm/llms/nvidia_nim/rerank/transformation.py
+++ b/litellm/llms/nvidia_nim/rerank/transformation.py
@@ -117,6 +117,7 @@ class NvidiaNimRerankConfig(BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
"""
Map Cohere/OpenAI rerank params to Nvidia NIM format.
diff --git a/litellm/llms/vertex_ai/rerank/transformation.py b/litellm/llms/vertex_ai/rerank/transformation.py
index 3b84972e946..d2041009efb 100644
--- a/litellm/llms/vertex_ai/rerank/transformation.py
+++ b/litellm/llms/vertex_ai/rerank/transformation.py
@@ -242,6 +242,7 @@ class VertexAIRerankConfig(BaseRerankConfig, VertexBase):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
"""
Map Cohere rerank params to Vertex AI format
diff --git a/litellm/llms/voyage/rerank/transformation.py b/litellm/llms/voyage/rerank/transformation.py
index d64450a1211..907e5b7e26b 100644
--- a/litellm/llms/voyage/rerank/transformation.py
+++ b/litellm/llms/voyage/rerank/transformation.py
@@ -39,6 +39,7 @@ class VoyageRerankConfig(BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
# Voyage AI uses 'top_k' instead of 'top_n'
optional_params: Dict[str, Any] = {"query": query, "documents": documents}
diff --git a/litellm/llms/watsonx/rerank/transformation.py b/litellm/llms/watsonx/rerank/transformation.py
index 202760f68a6..a34358a6be3 100644
--- a/litellm/llms/watsonx/rerank/transformation.py
+++ b/litellm/llms/watsonx/rerank/transformation.py
@@ -104,6 +104,7 @@ class IBMWatsonXRerankConfig(IBMWatsonXMixin, BaseRerankConfig):
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
) -> Dict:
"""
Map Cohere rerank params to IBM watsonx.ai rerank params
diff --git a/litellm/rerank_api/main.py b/litellm/rerank_api/main.py
index e40e12e9197..3ef74d596ad 100644
--- a/litellm/rerank_api/main.py
+++ b/litellm/rerank_api/main.py
@@ -103,6 +103,11 @@ def rerank(
"""
Reranks a list of documents based on their relevance to the query
"""
+ # `instruction` is read from kwargs rather than declared as a named param.
+ # The router forwards rerank calls via an untyped `**kwargs` unpack, and a
+ # typed named param there would trip the basedpyright budget gate without
+ # adding real safety; it stays typed downstream via get_optional_rerank_params.
+ instruction: Optional[str] = kwargs.get("instruction", None)
headers: Optional[dict] = kwargs.get("headers") # type: ignore
litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
@@ -155,6 +160,7 @@ def rerank(
return_documents=return_documents,
max_chunks_per_doc=max_chunks_per_doc,
max_tokens_per_doc=max_tokens_per_doc,
+ instruction=instruction,
non_default_params=kwargs,
)
verbose_logger.info(f"optional_rerank_params: {optional_rerank_params}")
diff --git a/litellm/rerank_api/rerank_utils.py b/litellm/rerank_api/rerank_utils.py
index 38e599ef824..a8a665496fc 100644
--- a/litellm/rerank_api/rerank_utils.py
+++ b/litellm/rerank_api/rerank_utils.py
@@ -15,6 +15,7 @@ def get_optional_rerank_params(
return_documents: Optional[bool] = True,
max_chunks_per_doc: Optional[int] = None,
max_tokens_per_doc: Optional[int] = None,
+ instruction: Optional[str] = None,
non_default_params: Optional[dict] = None,
) -> Dict:
all_non_default_params = non_default_params or {}
@@ -30,6 +31,11 @@ def get_optional_rerank_params(
all_non_default_params["max_chunks_per_doc"] = max_chunks_per_doc
if max_tokens_per_doc is not None:
all_non_default_params["max_tokens_per_doc"] = max_tokens_per_doc
+ if instruction is not None:
+ # Also surfaced in non_default_params so providers that read it from
+ # there (e.g. DeepInfra) keep working now that `rerank()` consumes
+ # `instruction` as a named param instead of leaving it in **kwargs.
+ all_non_default_params["instruction"] = instruction
return rerank_provider_config.map_cohere_rerank_params(
model=model,
drop_params=drop_params,
@@ -41,5 +47,6 @@ def get_optional_rerank_params(
return_documents=return_documents,
max_chunks_per_doc=max_chunks_per_doc,
max_tokens_per_doc=max_tokens_per_doc,
+ instruction=instruction,
non_default_params=all_non_default_params,
)
diff --git a/litellm/types/rerank.py b/litellm/types/rerank.py
index d2c252a1e92..376d6f66603 100644
--- a/litellm/types/rerank.py
+++ b/litellm/types/rerank.py
@@ -19,6 +19,10 @@ class RerankRequest(BaseModel):
return_documents: Optional[bool] = None
max_chunks_per_doc: Optional[int] = None
max_tokens_per_doc: Optional[int] = None
+ # Optional task/query instruction passed through to providers that support it
+ # (e.g. hosted vLLM / Qwen3-Reranker, DeepInfra). Omitted from the outgoing
+ # request when None, so this is fully backward-compatible.
+ instruction: Optional[str] = None
class OptionalRerankParams(TypedDict, total=False):
@@ -29,6 +33,7 @@ class OptionalRerankParams(TypedDict, total=False):
return_documents: Optional[bool]
max_chunks_per_doc: Optional[int]
max_tokens_per_doc: Optional[int]
+ instruction: Optional[str]
class RerankBilledUnits(TypedDict, total=False):
diff --git a/tests/test_litellm/llms/cohere/rerank/test_rerank_guardrail_handler.py b/tests/test_litellm/llms/cohere/rerank/test_rerank_guardrail_handler.py
index 88072cd7760..46c37e6af6c 100644
--- a/tests/test_litellm/llms/cohere/rerank/test_rerank_guardrail_handler.py
+++ b/tests/test_litellm/llms/cohere/rerank/test_rerank_guardrail_handler.py
@@ -2,10 +2,8 @@
Unit tests for Cohere Rerank Guardrail Translation Handler
"""
-import asyncio
import os
import sys
-from typing import List, Optional, Tuple
import pytest
@@ -94,6 +92,74 @@ class TestInputProcessing:
"id": "doc2",
}
+ @pytest.mark.asyncio
+ async def test_process_query_and_instruction(self):
+ """Both query and instruction are guardrailed; documents untouched"""
+ handler = CohereRerankHandler()
+ guardrail = MockGuardrail(guardrail_name="test")
+
+ data = {
+ "model": "qwen3-reranker",
+ "query": "What is machine learning?",
+ "instruction": "Rank by relevance to ML research",
+ "documents": ["Doc 1", "Doc 2"],
+ }
+
+ result = await handler.process_input_messages(data, guardrail)
+
+ # Both user-controlled text fields are scanned and written back
+ assert result["query"] == "What is machine learning? [GUARDRAILED]"
+ assert result["instruction"] == "Rank by relevance to ML research [GUARDRAILED]"
+ # Documents unchanged
+ assert result["documents"] == ["Doc 1", "Doc 2"]
+
+ @pytest.mark.asyncio
+ async def test_instruction_masked_with_pii(self):
+ """A masking guardrail rewrites instruction, not just query"""
+
+ class PIIMaskingGuardrail(CustomGuardrail):
+ async def apply_guardrail(
+ self, inputs: dict, request_data: dict, input_type: str, **kwargs
+ ) -> dict:
+ texts = inputs.get("texts", [])
+ return {"texts": [t.replace("John Doe", "[NAME_REDACTED]") for t in texts]}
+
+ handler = CohereRerankHandler()
+ guardrail = PIIMaskingGuardrail(guardrail_name="mask_pii")
+
+ data = {
+ "model": "qwen3-reranker",
+ "query": "find records",
+ "instruction": "prioritize anything authored by John Doe",
+ "documents": ["Doc 1"],
+ }
+
+ result = await handler.process_input_messages(data, guardrail)
+
+ # The sensitive value in instruction is sanitized before forwarding
+ assert "John Doe" not in result["instruction"]
+ assert "[NAME_REDACTED]" in result["instruction"]
+ assert result["documents"] == ["Doc 1"]
+
+ @pytest.mark.asyncio
+ async def test_non_string_instruction_not_scanned(self):
+ """A non-string instruction is left as-is (only strings are scanned)"""
+ handler = CohereRerankHandler()
+ guardrail = MockGuardrail(guardrail_name="test")
+
+ data = {
+ "model": "qwen3-reranker",
+ "query": "hello",
+ "instruction": 12345, # invalid type; backend will reject it
+ "documents": ["Doc 1"],
+ }
+
+ result = await handler.process_input_messages(data, guardrail)
+
+ # Query still guardrailed; non-string instruction untouched
+ assert result["query"] == "hello [GUARDRAILED]"
+ assert result["instruction"] == 12345
+
@pytest.mark.asyncio
async def test_process_no_query(self):
"""Test processing when query is missing"""
diff --git a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py
index 9e6fa608c50..6425e815db0 100644
--- a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py
+++ b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py
@@ -4,6 +4,7 @@ import sys
import pytest
from litellm.llms.hosted_vllm.rerank.transformation import HostedVLLMRerankConfig
+from litellm.rerank_api.rerank_utils import get_optional_rerank_params
from litellm.types.rerank import (
OptionalRerankParams,
RerankBilledUnits,
@@ -37,6 +38,54 @@ class TestHostedVLLMRerankTransform:
assert params["rank_fields"] == ["field1"]
assert params["return_documents"] is True
+ def test_map_cohere_rerank_params_omits_instruction_when_absent(self):
+ # Backward-compat: when no instruction is supplied, it must not appear
+ # in the mapped params (and therefore not in the outgoing request body).
+ params = self.config.map_cohere_rerank_params(
+ non_default_params=None,
+ model=self.model,
+ drop_params=False,
+ query="test query",
+ documents=["doc1", "doc2"],
+ )
+ assert "instruction" not in params
+
+ def test_map_cohere_rerank_params_passes_instruction_when_set(self):
+ params = self.config.map_cohere_rerank_params(
+ non_default_params=None,
+ model=self.model,
+ drop_params=False,
+ query="test query",
+ documents=["doc1", "doc2"],
+ instruction="Rank by relevance to genomics",
+ )
+ assert params["instruction"] == "Rank by relevance to genomics"
+
+ def test_transform_request_includes_instruction_when_set(self):
+ body = self.config.transform_rerank_request(
+ model=self.model,
+ optional_rerank_params={
+ "query": "test query",
+ "documents": ["doc1", "doc2"],
+ "instruction": "Rank by relevance to genomics",
+ },
+ headers={},
+ )
+ assert body["instruction"] == "Rank by relevance to genomics"
+
+ def test_transform_request_omits_instruction_when_absent(self):
+ # exclude_none must drop the field entirely so the body matches the
+ # pre-existing (instruction-less) shape exactly.
+ body = self.config.transform_rerank_request(
+ model=self.model,
+ optional_rerank_params={
+ "query": "test query",
+ "documents": ["doc1", "doc2"],
+ },
+ headers={},
+ )
+ assert "instruction" not in body
+
def test_map_cohere_rerank_params_raises_on_max_chunks_per_doc(self):
with pytest.raises(
ValueError, match="Hosted VLLM does not support max_chunks_per_doc"
@@ -74,6 +123,7 @@ class TestHostedVLLMRerankTransform:
}
result = self.config._transform_response(response_dict)
assert result.id == "abc123"
+ assert result.results is not None
assert len(result.results) == 2
assert result.results[0]["index"] == 0
assert result.results[0]["relevance_score"] == 0.9
@@ -94,3 +144,32 @@ class TestHostedVLLMRerankTransform:
}
with pytest.raises(ValueError, match="Missing required fields in the result="):
self.config._transform_response(response_dict)
+
+
+class TestGetOptionalRerankParamsInstruction:
+ """`instruction` is threaded through get_optional_rerank_params only when set."""
+
+ def setup_method(self):
+ self.config = HostedVLLMRerankConfig()
+ self.model = "hosted-vllm-model"
+
+ def test_instruction_threaded_when_set(self):
+ params = get_optional_rerank_params(
+ rerank_provider_config=self.config,
+ model=self.model,
+ drop_params=False,
+ query="test query",
+ documents=["doc1", "doc2"],
+ instruction="Rank by relevance to genomics",
+ )
+ assert params["instruction"] == "Rank by relevance to genomics"
+
+ def test_instruction_absent_when_not_set(self):
+ params = get_optional_rerank_params(
+ rerank_provider_config=self.config,
+ model=self.model,
+ drop_params=False,
+ query="test query",
+ documents=["doc1", "doc2"],
+ )
+ assert "instruction" not in params
From b302c4204a005efdd705633e0a762fea693d436a Mon Sep 17 00:00:00 2001
From: "David J. M. Karlsen"
Date: Wed, 24 Jun 2026 13:19:38 +0200
Subject: [PATCH 27/30] fix(github_copilot): synthesize empty choices at the
provider seam (#30929)
Newer Copilot Claude models (opus-4.7, opus-4.8) return responses with
choices=[], either carrying Anthropic-native content blocks or, for the
max_tokens=1 probe Claude Code sends, no content at all. github_copilot
is dispatched through the OpenAI SDK handler, which calls
convert_to_model_response_object directly and never invokes
GithubCopilotConfig.transform_response, so the empty-choices guard there
surfaced as a 500
Instead of synthesizing choices inside the shared
convert_to_model_response_object (which would silently turn empty choices
into a fabricated success for every provider), add a no-op
transform_parsed_response_dict hook on BaseConfig. GithubCopilotConfig
overrides it to synthesize choices from Anthropic-native content, reusing
its existing parsing, and the OpenAI SDK handler routes its parsed
response through the hook before generic conversion. The core utility
keeps treating empty choices as an error for all other providers
Fixes: https://github.com/BerriAI/litellm/issues/30927
Signed-off-by: David J. M. Karlsen
---
litellm/llms/base_llm/chat/transformation.py | 11 ++
.../github_copilot/chat/transformation.py | 153 ++++++++++--------
litellm/llms/openai/openai.py | 10 +-
.../test_convert_dict_to_chat_completion.py | 20 ++-
.../test_github_copilot_transformation.py | 104 ++++++++++++
5 files changed, 224 insertions(+), 74 deletions(-)
diff --git a/litellm/llms/base_llm/chat/transformation.py b/litellm/llms/base_llm/chat/transformation.py
index 8f9d5cad7c4..4f7e98af780 100644
--- a/litellm/llms/base_llm/chat/transformation.py
+++ b/litellm/llms/base_llm/chat/transformation.py
@@ -377,6 +377,17 @@ class BaseConfig(ABC):
) -> "ModelResponse":
pass
+ def transform_parsed_response_dict(self, parsed_response: dict) -> dict:
+ """
+ Repair a parsed OpenAI-format response dict before generic conversion.
+
+ Providers routed through the OpenAI SDK handler bypass transform_response,
+ which calls convert_to_model_response_object directly on the SDK's parsed
+ output. Override this to normalize a malformed response (e.g. github_copilot
+ returning empty choices for Anthropic-native Claude responses).
+ """
+ return parsed_response
+
@abstractmethod
def get_error_class(
self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]
diff --git a/litellm/llms/github_copilot/chat/transformation.py b/litellm/llms/github_copilot/chat/transformation.py
index 72dacb59f8a..9880ab1eb6e 100644
--- a/litellm/llms/github_copilot/chat/transformation.py
+++ b/litellm/llms/github_copilot/chat/transformation.py
@@ -194,6 +194,88 @@ class GithubCopilotConfig(OpenAIConfig):
)
return text_content, tool_calls, thinking_blocks
+ @staticmethod
+ def _normalize_anthropic_usage(usage: dict) -> dict:
+ normalized = dict(usage)
+ if "input_tokens" in usage and "prompt_tokens" not in usage:
+ normalized["prompt_tokens"] = usage["input_tokens"]
+ if "output_tokens" in usage and "completion_tokens" not in usage:
+ normalized["completion_tokens"] = usage["output_tokens"]
+ if "total_tokens" not in normalized:
+ normalized["total_tokens"] = normalized.get(
+ "prompt_tokens", 0
+ ) + normalized.get("completion_tokens", 0)
+ return normalized
+
+ @classmethod
+ def _synthesize_choices_for_anthropic_native(cls, response_json: dict) -> dict:
+ """
+ Synthesize a `choices` array from an Anthropic-native Copilot response.
+
+ Newer Copilot Claude models (e.g. opus-4.7, opus-4.8) return content
+ blocks and `stop_reason` without an OpenAI-style `choices` array, and the
+ max_tokens=1 probe returns no content at all. Returns the response
+ unchanged when it already carries choices.
+
+ See: https://github.com/BerriAI/litellm/issues/29391
+ """
+ if response_json.get("choices"):
+ return response_json
+
+ content = ""
+ tool_calls: List[ChatCompletionToolCallChunk] = []
+ thinking_blocks: Optional[List[Any]] = None
+ raw_content = response_json.get("content")
+ if isinstance(raw_content, list):
+ content, tool_calls, thinking_blocks = cls._parse_anthropic_native_content(
+ raw_content
+ )
+ elif isinstance(raw_content, str):
+ content = raw_content
+
+ stop_reason = response_json.get("stop_reason")
+ finish_reason_map = {
+ "end_turn": "stop",
+ "max_tokens": "length",
+ "stop_sequence": "stop",
+ "tool_use": "tool_calls",
+ }
+ if tool_calls:
+ finish_reason = "tool_calls"
+ elif stop_reason in finish_reason_map:
+ finish_reason = finish_reason_map[stop_reason]
+ elif content:
+ finish_reason = "stop"
+ else:
+ finish_reason = "length"
+
+ message: dict = {
+ "role": "assistant",
+ "content": content if content or not tool_calls else None,
+ }
+ if tool_calls:
+ message["tool_calls"] = tool_calls
+ if thinking_blocks:
+ message["thinking_blocks"] = thinking_blocks
+
+ synthesized = {
+ **response_json,
+ "choices": [
+ {"index": 0, "message": message, "finish_reason": finish_reason}
+ ],
+ }
+ usage = response_json.get("usage")
+ if isinstance(usage, dict):
+ synthesized["usage"] = cls._normalize_anthropic_usage(usage)
+ return synthesized
+
+ def transform_parsed_response_dict(self, parsed_response: dict) -> dict:
+ """
+ Repair the OpenAI-SDK-parsed response on the handler path that bypasses
+ transform_response. See: https://github.com/BerriAI/litellm/issues/30927
+ """
+ return self._synthesize_choices_for_anthropic_native(parsed_response)
+
def transform_response(
self,
model: str,
@@ -208,15 +290,6 @@ class GithubCopilotConfig(OpenAIConfig):
api_key: Optional[str] = None,
json_mode: Optional[bool] = None,
) -> "ModelResponse":
- """
- Handle newer Copilot models (e.g. claude-opus-4.7, claude-opus-4.8) that
- return Anthropic-native format responses without a `choices` array.
-
- Synthesizes the missing `choices` from Anthropic-native fields, then
- delegates to the parent so all standard post-processing applies.
-
- See: https://github.com/BerriAI/litellm/issues/29391
- """
try:
response_json = raw_response.json()
except Exception:
@@ -235,70 +308,12 @@ class GithubCopilotConfig(OpenAIConfig):
)
if not response_json.get("choices"):
- content = ""
- tool_calls: List[ChatCompletionToolCallChunk] = []
- thinking_blocks: Optional[List[Any]] = None
- if "content" in response_json and isinstance(
- response_json["content"], list
- ):
- content, tool_calls, thinking_blocks = (
- self._parse_anthropic_native_content(response_json["content"])
- )
- elif isinstance(response_json.get("content"), str):
- content = response_json["content"]
-
- stop_reason = response_json.get("stop_reason")
- finish_reason_map = {
- "end_turn": "stop",
- "max_tokens": "length",
- "stop_sequence": "stop",
- "tool_use": "tool_calls",
- }
- # Prefer tool_calls when blocks were extracted; otherwise map stop_reason.
- if tool_calls:
- finish_reason = "tool_calls"
- elif stop_reason in finish_reason_map:
- finish_reason = finish_reason_map[stop_reason]
- elif content:
- finish_reason = "stop"
- else:
- finish_reason = "length"
-
- message: dict = {
- "role": "assistant",
- "content": content if content or not tool_calls else None,
- }
- if tool_calls:
- message["tool_calls"] = tool_calls
- if thinking_blocks:
- message["thinking_blocks"] = thinking_blocks
-
- response_json["choices"] = [
- {
- "index": 0,
- "message": message,
- "finish_reason": finish_reason,
- }
- ]
-
- if "usage" in response_json:
- usage = response_json["usage"]
- if "input_tokens" in usage and "prompt_tokens" not in usage:
- usage["prompt_tokens"] = usage["input_tokens"]
- if "output_tokens" in usage and "completion_tokens" not in usage:
- usage["completion_tokens"] = usage["output_tokens"]
- if "total_tokens" not in usage:
- usage["total_tokens"] = usage.get("prompt_tokens", 0) + usage.get(
- "completion_tokens", 0
- )
-
- # Build a patched response so super() sees valid JSON with choices
- patched = httpx.Response(
+ response_json = self._synthesize_choices_for_anthropic_native(response_json)
+ raw_response = httpx.Response(
status_code=raw_response.status_code,
headers=raw_response.headers,
content=json.dumps(response_json).encode(),
)
- raw_response = patched
return super().transform_response(
model=model,
diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py
index ea905d8ebca..8237aaa010a 100644
--- a/litellm/llms/openai/openai.py
+++ b/litellm/llms/openai/openai.py
@@ -785,7 +785,11 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM):
)
logging_obj.model_call_details["response_headers"] = headers
- stringified_response = response.model_dump()
+ stringified_response = (
+ provider_config.transform_parsed_response_dict(
+ response.model_dump()
+ )
+ )
logging_obj.post_call(
input=messages,
api_key=api_key,
@@ -933,7 +937,9 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM):
timeout=timeout,
logging_obj=logging_obj,
)
- stringified_response = response.model_dump()
+ stringified_response = provider_config.transform_parsed_response_dict(
+ response.model_dump()
+ )
logging_obj.post_call(
input=data["messages"],
api_key=api_key,
diff --git a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py
index 9a69f513069..d46436f209b 100644
--- a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py
+++ b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py
@@ -1627,7 +1627,12 @@ class TestMissingChoicesGuard:
assert "no 'choices'" in exc_info.value.message
def test_convert_to_model_response_object_empty_choices_raises_api_error(self):
- """Empty choices list raises APIError."""
+ """Empty choices list raises APIError, same as missing/null choices.
+
+ Provider-specific repair (e.g. github_copilot synthesizing choices for
+ Anthropic-native responses) happens before this guard, in the provider
+ config; the core utility keeps treating empty choices as an error.
+ """
from litellm.exceptions import APIError
response_object = {
@@ -1683,7 +1688,9 @@ class TestMissingChoicesGuard:
assert "no 'choices'" in exc_info.value.message
- def test_convert_to_model_response_object_stream_true_no_choices_raises_api_error(self):
+ def test_convert_to_model_response_object_stream_true_no_choices_raises_api_error(
+ self,
+ ):
"""Missing choices via stream=True path raises APIError when generator is consumed."""
from litellm.exceptions import APIError
@@ -2471,6 +2478,13 @@ class TestConvertToModelResponseObjectCompletion:
def test_model_response_none_raises(self):
with pytest.raises(Exception):
convert_to_model_response_object(
- response_object={"choices": [{"message": {"content": "hi", "role": "assistant"}, "finish_reason": "stop"}]},
+ response_object={
+ "choices": [
+ {
+ "message": {"content": "hi", "role": "assistant"},
+ "finish_reason": "stop",
+ }
+ ]
+ },
model_response_object=None,
)
diff --git a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py
index 5673ad81551..f69ba7df938 100644
--- a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py
+++ b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py
@@ -878,3 +878,107 @@ class TestGithubCopilotTransformResponse:
litellm_params={},
encoding=None,
)
+
+
+class TestGithubCopilotTransformParsedResponseDict:
+ """
+ Tests for GithubCopilotConfig.transform_parsed_response_dict, the hook the
+ OpenAI SDK handler calls on its parsed response. That handler bypasses
+ transform_response, so this is the seam that repairs empty-choices responses
+ from newer Copilot Claude models on the live completion path.
+
+ See: https://github.com/BerriAI/litellm/issues/30927
+ """
+
+ def test_synthesizes_choices_from_anthropic_content(self):
+ config = GithubCopilotConfig()
+
+ parsed = {
+ "id": "msg_vrtx_01",
+ "model": "claude-opus-4.8",
+ "object": "chat.completion",
+ "choices": [],
+ "content": [{"type": "text", "text": "Hello!"}],
+ "stop_reason": "end_turn",
+ "usage": {"input_tokens": 10, "output_tokens": 5},
+ }
+
+ repaired = config.transform_parsed_response_dict(parsed)
+
+ assert len(repaired["choices"]) == 1
+ choice = repaired["choices"][0]
+ assert choice["message"]["content"] == "Hello!"
+ assert choice["finish_reason"] == "stop"
+ assert repaired["usage"]["prompt_tokens"] == 10
+ assert repaired["usage"]["completion_tokens"] == 5
+ assert repaired["usage"]["total_tokens"] == 15
+
+ def test_passthrough_when_choices_present(self):
+ config = GithubCopilotConfig()
+
+ parsed = {
+ "id": "chatcmpl-1",
+ "choices": [
+ {
+ "index": 0,
+ "message": {"role": "assistant", "content": "ok"},
+ "finish_reason": "stop",
+ }
+ ],
+ }
+
+ assert config.transform_parsed_response_dict(parsed) is parsed
+
+
+@patch("litellm.llms.openai.openai.OpenAIChatCompletion._get_openai_client")
+@patch(
+ "litellm.llms.openai.openai.OpenAIChatCompletion.make_sync_openai_chat_completion_request"
+)
+def test_openai_handler_repairs_github_copilot_empty_choices(
+ mock_request, mock_get_client
+):
+ """
+ The OpenAI SDK handler calls convert_to_model_response_object directly on the
+ SDK's parsed output, bypassing transform_response. convert raises APIError on
+ empty choices, so the handler must route github_copilot responses through
+ transform_parsed_response_dict first. Removing that wiring (or resolving a
+ config without the override) fails this test with APIError.
+
+ See: https://github.com/BerriAI/litellm/issues/30927
+ """
+ from litellm.llms.openai.openai import OpenAIChatCompletion
+
+ mock_get_client.return_value = MagicMock()
+
+ class _FakeSDKResponse:
+ def model_dump(self):
+ return {
+ "id": "msg_vrtx_01",
+ "model": "claude-opus-4.8",
+ "object": "chat.completion",
+ "choices": [],
+ "content": [{"type": "text", "text": "Hi there"}],
+ "stop_reason": "end_turn",
+ "usage": {"input_tokens": 12, "output_tokens": 3},
+ }
+
+ mock_request.return_value = ({}, _FakeSDKResponse())
+
+ result = OpenAIChatCompletion().completion(
+ model="claude-opus-4.8",
+ messages=[{"role": "user", "content": "Hi"}],
+ model_response=ModelResponse(),
+ timeout=60.0,
+ optional_params={},
+ litellm_params={},
+ logging_obj=MagicMock(),
+ custom_llm_provider="github_copilot",
+ client=MagicMock(),
+ api_key="gh.test-key-123456789",
+ acompletion=False,
+ )
+
+ assert isinstance(result, ModelResponse)
+ assert result.choices[0].message.content == "Hi there"
+ assert result.choices[0].finish_reason == "stop"
+ mock_request.assert_called_once()
From 2f6fd18842bde8bc3cccb15373b0fe1fdc6b358b Mon Sep 17 00:00:00 2001
From: Vedant Agarwal <43557509+Vedant-Agarwal@users.noreply.github.com>
Date: Wed, 24 Jun 2026 19:20:59 +0800
Subject: [PATCH 28/30] fix(router): stop fallback lookups from mutating the
router fallbacks config (#30624)
* fix: correct amazon.titan-embed-text-v2 input price to $0.02/1M tokens (#29693)
* fix: correct amazon.titan-embed-text-v2 input price to $0.02/1M tokens
* test: scope local cost map env var with monkeypatch to avoid test pollution
* fix(sensitive_data_masker): fully mask secrets at or below the reveal threshold (#30764)
* fix(sensitive_data_masker): fully mask secrets at or below the reveal threshold
_mask_value did partial reveal by showing the first visible_prefix and last
visible_suffix characters, but for a value whose length was at or below
visible_prefix + visible_suffix (8 by default) it returned the value verbatim.
A value of exactly 8 chars fell through the length guard and computed
masked_length == 0, reconstructing the original string with no mask characters;
anything shorter hit the early return. Either way short credentials were emitted
in plaintext.
mask_dict routes real secrets through this path, so an 8-char-or-shorter redis
password, api key, or token could be written to logs and the UI unmasked. The
sibling helper mask_sensitive_keys already guards this case; _mask_value now does
the same by fully masking any value at or below the threshold.
* fix(sensitive_data_masker): add mask_short_values opt-out for truncation callers
Fully masking short values is the right default for secret masking, but
CooldownCache reuses the masker purely to truncate exception messages to the
first 50 characters, and it relies on short messages being returned readable.
Masking those blanked out short exception text and broke its tests.
Add a mask_short_values flag (default True, secure) and have CooldownCache pass
False so it keeps the truncation behavior, while every secret-masking caller
still gets short values fully masked.
* fix(mcp_debug): opt out of short-value masking to keep diagnostic token preview
MCPDebug uses the masker to preview auth tokens in debug headers and documents
that values of 10 chars or fewer are shown unchanged so token types stay
distinguishable. Pass mask_short_values=False so that diagnostic behavior is
preserved while secret maskers keep masking short values.
* fix(mcp_debug): mask short auth values in debug headers instead of echoing them
Earlier this masker opted out of short-value masking to keep a token preview, but
that echoes short authorization and token values verbatim in debug response
headers, which is the same leak this change is meant to close. Auth material
should never be emitted in full, so mask short values here too; the first/last
character preview still applies to longer tokens. Only CooldownCache keeps the
opt-out, since it truncates exception text rather than masking secrets.
* test(mcp_debug): assert masked short value preserves length
* refactor(fireworks_ai): remove deprecated audio transcriptions endpoint (#30917)
Fireworks AI deprecated audio inference on 2026-06-10
(https://docs.fireworks.ai/updates/changelog#audio-inference-and-image-generation-deprecation).
Live API testing confirms the endpoint is already non-functional: a valid
Fireworks API key receives HTTP 401 "Unauthorized" from
api.fireworks.ai/inference/v1/audio/transcriptions for every request,
regardless of payload. The audio-prod.api.fireworks.ai host referenced in
the test suite returns 401 for every path; the entire host is decommissioned.
Remove the dead FireworksAIAudioTranscriptionConfig class and every
reference to it across the codebase:
- Delete litellm/llms/fireworks_ai/audio_transcription/ directory (17-line
config class that inherited from OpenAIWhisperAudioTranscriptionConfig)
- Remove the Fireworks branch from
ProviderConfigManager.get_provider_audio_transcription_config() in
litellm/utils.py; update the stale comment in
get_optional_params_transcription that referenced fireworks ai
- Remove the FireworksAIAudioTranscriptionConfig entries from
LLM_CONFIG_NAMES and _LLM_CONFIGS_IMPORT_MAP in
litellm/_lazy_imports_registry.py
- Remove the TYPE_CHECKING re-export in litellm/__init__.py
- Remove the transcription branch in the fireworks_ai case of
get_supported_openai_params() in
litellm/litellm_core_utils/get_supported_openai_params.py
- Remove the whisper-v3 and whisper-v3-turbo entries from
model_prices_and_context_window.json and
litellm/model_prices_and_context_window_backup.json (both had
mode: audio_transcription and zero-cost pricing)
- Remove the TestFireworksAIAudioTranscription test class and its
imports from tests/llm_translation/test_fireworks_ai_translation.py
No other provider is affected. The openai_compatible_providers list,
FireworksAIMixin, and the OpenAI Whisper transcription handler all stay
because they are shared with other Fireworks endpoints and other
providers. The provider_endpoints_support.json registry already had
audio_transcriptions set to false for fireworks_ai.
* feat: add darkbloom provider (#30876)
* feat: add darkbloom provider
* fix: document darkbloom provider endpoints
* fix: address darkbloom review feedback
* fix: update darkbloom tool metadata
* fix: fail fast for non-Postgres database URLs (#30883)
* fix(proxy): fail fast on non-PostgreSQL DATABASE_URL instead of hanging on startup
LiteLLM's Prisma datasource is pinned to provider = 'postgresql', so a sqlite:// or mysql:// DATABASE_URL can never connect.
Today that surfaces as an opaque startup stall where the port never binds, and a separate 'DB not connected' 500 on /key/generate when no DATABASE_URL is set at all leaves operators guessing what to configure.
Validate the DATABASE_URL / DIRECT_URL scheme in run_server before any Prisma call and exit with an actionable message naming the unsupported scheme.
Also reword CommonProxyErrors.db_not_connected_error to tell the operator to set DATABASE_URL to a postgresql:// connection string.
Add regression tests covering postgres acceptance and sqlite/mysql/mssql rejection.
* fix: resolve CI failures and proxy DB URL typing issue
* fix(proxy): fail fast on non-PostgreSQL DATABASE_URLs with clear startup errors instead of hanging
* Validate DIRECT_URL alongside DATABASE_URL startup guards
* fix(bedrock): surface modeled HTTP status for mid-stream error events so 5xx is retryable (#24608) (#30946)
* fix(bedrock): surface modeled HTTP status for mid-stream error events (#24608)
* test(bedrock): mid-stream server errors trigger streaming fallback (#24608)
* style(bedrock): black-format stream-error helper (#24608)
* fix(mcp): re-land native tool preservation with typed annotations (#30645)
* fix(mcp): preserve native tools in semantic filter hook with typed annotations
* fix(mcp): tighten _is_mcp_tool Chat Completions shape check
* fix(sambanova): return embeddings supported params instead of dropping them (#30937)
* fix(router): send fallback metadata when streaming (#30914)
When a streaming request triggers a fallback, there was previously no way to
know it happened. This commit addresses this in a few ways:
1. The response now correctly populates the fallback headers
(`x-litellm-attempted-fallbacks`) so callers know a fallback happened.
2. The correct model ID is passed in the streaming chunks.
3. A streaming chunk with the fallback error can be optionally sent back
to the client (opt-in) by passing `include_fallback_errors: true` in
the request.
The format of the fallback errors while streaming is intentionally OpenAI
compatible to not break existing libraries that parse these events. It was
tested with Vercel's AI SDK (ai-sdk.dev). It is also opt-in, so it is not
delieved unexpectedly to callers by default.
* fix(mistral): drop output-only reasoning fields from input messages (#30884)
LiteLLM attaches reasoning_content and thinking_blocks to assistant
responses. Replaying those assistant turns verbatim forwarded the fields
back to Mistral, whose input schema forbids unknown keys, so the whole
request failed with a 422 extra_forbidden and reasoning models became
unusable across multiple turns.
Strip both fields from assistant messages before the request is built, in
a spot that runs ahead of the image/file branch so it applies on every
path. Fixes #30835
Co-authored-by: Cursor
* fix(perplexity): bill search queries at the per-request price, not 1/1000 of it (#30652)
* fix(perplexity): bill search queries at the per-request price, not 1/1000
The fallback cost calculator divided search_context_cost_per_query by
1000, but that field stores the per-request price in USD: sonar is
{low: 0.005, medium: 0.008, high: 0.012}, matching Perplexity's published
$5/$8/$12 per 1,000 requests expressed per request. The gemini cost
calculator reads the same field per request with no division (its
docstring calls it "the per-request cost").
The division understated search cost by 1000x on every Perplexity call
that falls back to manual calculation (i.e. when the API does not return
a pre-computed usage.cost). Use the value directly.
Update the tests that had encoded the /1000 factor in their expectations,
and drop an unused import flagged by ruff in the touched test file.
* test(perplexity): update integration test search-cost expectations to per-request
The integration tests still encoded the old /1000 search-cost factor, so
they failed once the fallback calculator was corrected to bill
search_context_cost_per_query per request. Update the four expected-cost
computations (and the high-volume dollar-value comments) to match.
* test(perplexity): drop unused mock imports flagged by ruff
* fix: include model_access_groups when expanding all-team-models in get_team_models (#30622)
* fix(fireworks_ai): return None for transcription in get_supported_openai_params
Fireworks AI deprecated audio inference on 2026-06-10; the endpoint is
decommissioned. Without an explicit transcription branch, requests with
request_type='transcription' fell through to the else and returned
FireworksAIConfig chat-completion params. Return None instead to signal
the provider does not support transcription.
* fix(proxy): gate include_fallback_errors behind expose_fallback_errors_to_caller setting
Without an operator gate, any authenticated caller could set include_fallback_errors=True,
trigger a fallback, and read raw upstream exception messages from the
x-litellm-fallback-errors header and the litellm-fallback-metadata SSE event.
Strip include_fallback_errors from request data in common_processing_pre_call_logic
when expose_fallback_errors_to_caller is not set, so the router never builds the
error list. Also gate _should_include_fallback_errors on the same setting as a
secondary check for the streaming SSE injection path.
* test(proxy): opt in to expose_fallback_errors_to_caller in streaming SSE test
The operator gate added in e7ff3e1 means include_fallback_errors is only
honoured when general_settings.expose_fallback_errors_to_caller is True.
Set that flag via monkeypatch in the test that exercises the emit path.
* test(prompt_templates): make test_convert_url hermetic instead of hitting picsum.photos
test_convert_url called convert_url_to_base64 against a live picsum.photos
URL and asserted nothing, so it added no real signal and broke CI whenever
the host was unreachable (it was returning 522 and blocking this branch).
Replace the live call with a mocked HTTP client and assert the produced
base64 data URL, so the conversion path is exercised deterministically with
no network dependency. This suite runs under VCR, which is why a transport
level mock (respx) does not reliably intercept; mocking the client object
itself is robust regardless.
* fix(interactions): drop role from Interaction response to match Google spec
Google removed the output-only role field from the Interaction schema (it
now lives only on Turn), so the live OpenAPI compliance canary started
failing with 'role' not in spec. Reconcile our generated types by removing
role from Interaction, CreateModelInteractionParams, CreateAgentInteractionParams
and from the LiteLLM InteractionsAPIResponse/InteractionsAPIStreamingResponse,
stop stamping role=model in the responses-to-interactions transformation, and
update the compliance and integration tests accordingly. Turn.role is kept
since the spec still defines it.
* fix: align all-team-models sentinel access
* fix(router): forward include_fallback_errors through multi-hop fallbacks
run_async_fallback received include_fallback_errors as an explicit named
parameter, so it was bound out of **kwargs and never reached the nested
async_function_with_fallbacks call. Multi-hop fallback chains (a fallback
group that itself fails over) therefore stopped collecting fallback errors
beyond the first hop when a caller opted in. Re-inject the flag into kwargs
before the nested call so inner hops keep accumulating errors, which
add_fallback_headers_to_response already merges across levels.
* fix(router): stop fallback lookups from mutating the router fallbacks config
get_fallback_model_group resolved a bare-string fallback by popping it out
of the fallbacks list it was handed. That list is frequently the live
router.fallbacks config, so a single lookup permanently removed the entry and
the configured fallback stopped applying to later requests until restart. The
pop also ran inside enumerate(), shifting indices and skipping an adjacent
string fallback. Read the item instead of popping it, and add a regression
test that fails on the old mutating behavior
---------
Co-authored-by: Srivatsa Kamballa
Co-authored-by: Ahmad Shahzad <107808273+shzdehmd@users.noreply.github.com>
Co-authored-by: Jeremy Chapeau <113923302+jychp@users.noreply.github.com>
Co-authored-by: KRISH SONI <67964054+krishvsoni@users.noreply.github.com>
Co-authored-by: Kent <72616338+kingdoooo@users.noreply.github.com>
Co-authored-by: Ayush Shekhar <106994833+ayushh0110@users.noreply.github.com>
Co-authored-by: dav nguyxn
Co-authored-by: Tal Marian
Co-authored-by: Hemant K <51333870+hemant1026@users.noreply.github.com>
Co-authored-by: Cursor
Co-authored-by: Yash Raj Pandey <55940078+devYRPauli@users.noreply.github.com>
Co-authored-by: Zang Peiyu <166481866+factnn@users.noreply.github.com>
Co-authored-by: Sameer Kankute
Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
---
.../router_utils/fallback_event_handlers.py | 2 +-
.../test_fallback_event_handlers.py | 18 +++++++++++++++++-
2 files changed, 18 insertions(+), 2 deletions(-)
diff --git a/litellm/router_utils/fallback_event_handlers.py b/litellm/router_utils/fallback_event_handlers.py
index f0edc7fc9db..891d80d785a 100644
--- a/litellm/router_utils/fallback_event_handlers.py
+++ b/litellm/router_utils/fallback_event_handlers.py
@@ -72,7 +72,7 @@ def get_fallback_model_group(
elif list(item.keys())[0] == "*": # check generic fallback
generic_fallback_idx = idx
elif isinstance(item, str):
- fallback_model_group = [fallbacks.pop(idx)] # returns single-item list
+ fallback_model_group = [item]
## if none, check for generic fallback
if fallback_model_group is None:
if stripped_model_fallback is not None:
diff --git a/tests/test_litellm/router_utils/test_fallback_event_handlers.py b/tests/test_litellm/router_utils/test_fallback_event_handlers.py
index ca647bdce55..98a34de295c 100644
--- a/tests/test_litellm/router_utils/test_fallback_event_handlers.py
+++ b/tests/test_litellm/router_utils/test_fallback_event_handlers.py
@@ -2,7 +2,10 @@ import json
import pytest
-from litellm.router_utils.fallback_event_handlers import run_async_fallback
+from litellm.router_utils.fallback_event_handlers import (
+ get_fallback_model_group,
+ run_async_fallback,
+)
class StreamingWrapper:
@@ -137,3 +140,16 @@ async def test_run_async_fallback_skips_original_model_group():
)
assert response._hidden_params["additional_headers"]["x-litellm-attempted-fallbacks"] == 1
+
+
+def test_get_fallback_model_group_does_not_mutate_fallbacks():
+ """A string fallback must be resolved without mutating the caller's
+ fallbacks list, which is the live router config shared across requests."""
+ fallbacks = [{"gpt-3.5-turbo": ["claude-3-haiku"]}, "gpt-4o-mini"]
+
+ fallback_model_group, _ = get_fallback_model_group(
+ fallbacks=fallbacks, model_group="unmatched-model"
+ )
+
+ assert fallback_model_group == ["gpt-4o-mini"]
+ assert fallbacks == [{"gpt-3.5-turbo": ["claude-3-haiku"]}, "gpt-4o-mini"]
From af7b0af52fa1383ceb5804b6c92a8e724b52947c Mon Sep 17 00:00:00 2001
From: bhumikadangayach <139267865+bhumikadangayach@users.noreply.github.com>
Date: Wed, 24 Jun 2026 16:57:41 +0530
Subject: [PATCH 29/30] fix(sambanova): update pricing, deprecate retired
models, and add missing models (#30016)
* feat(bedrock): add amazon.titan-embed-g1-text-02 embedding model support
- Add model to provider routing allowlist in embedding.py
- Add request transformation using AmazonTitanG1Config
- Add response transformation using AmazonTitanG1Config
- Add pricing metadata to model_prices_and_context_window.json
- Add unit tests for embedding and model info
Fixes missing cost tracking reported in #29786
Related to VANDRANKI/litellm PR #29790
* style: fix syntax error, trailing whitespace and missing newline
* style: apply black formatting to embedding.py
* style: apply black formatting to test_bedrock_embedding.py
* fix(sambanova): update pricing, fix context windows, add deprecation dates, and add missing models
* fix(sambanova): sync model_prices_and_context_window_backup.json with primary
* fix(sambanova): fix indentation on Meta-Llama-3.2-1B-Instruct deprecation_date
* fix(bedrock): add amazon.titan-embed-g1-text-02 to unmapped model error message
* style: apply black formatting to embedding.py
* fix(sambanova): correct indentation on DeepSeek-V3.2 entry
* fix(sambanova): replace gemma-3-12b-it with gemma-4-31B-it (verified pricing)
---
litellm/llms/bedrock/embed/embedding.py | 10 +
...odel_prices_and_context_window_backup.json | 987 +++++++++++++-----
model_prices_and_context_window.json | 57 +-
.../llm_translation/test_bedrock_embedding.py | 15 +
4 files changed, 819 insertions(+), 250 deletions(-)
diff --git a/litellm/llms/bedrock/embed/embedding.py b/litellm/llms/bedrock/embed/embedding.py
index b6aa99842d7..98c2d87bfdb 100644
--- a/litellm/llms/bedrock/embed/embedding.py
+++ b/litellm/llms/bedrock/embed/embedding.py
@@ -224,6 +224,10 @@ class BedrockEmbedding(BaseAWSLLM):
returned_response = AmazonTitanV2Config()._transform_response(
response_list=response_list, model=model
)
+ elif model == "amazon.titan-embed-g1-text-02":
+ returned_response = AmazonTitanG1Config()._transform_response(
+ response_list=response_list, model=model
+ )
elif provider == "twelvelabs":
returned_response = (
TwelveLabsMarengoEmbeddingConfig()._transform_response(
@@ -447,6 +451,7 @@ class BedrockEmbedding(BaseAWSLLM):
"amazon.titan-embed-image-v1",
"amazon.titan-embed-text-v1",
"amazon.titan-embed-text-v2:0",
+ "amazon.titan-embed-g1-text-02",
]:
batch_data = []
for i in input:
@@ -464,6 +469,10 @@ class BedrockEmbedding(BaseAWSLLM):
transformed_request = AmazonTitanV2Config()._transform_request(
input=i, inference_params=inference_params
)
+ elif model == "amazon.titan-embed-g1-text-02":
+ transformed_request = AmazonTitanG1Config()._transform_request(
+ input=i, inference_params=inference_params
+ )
else:
raise Exception(
"Unmapped model. Received={}. Expected={}".format(
@@ -472,6 +481,7 @@ class BedrockEmbedding(BaseAWSLLM):
"amazon.titan-embed-image-v1",
"amazon.titan-embed-text-v1",
"amazon.titan-embed-text-v2:0",
+ "amazon.titan-embed-g1-text-02",
],
)
)
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index 6ebac7efc8d..452a33be695 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -570,6 +570,15 @@
"output_cost_per_token": 0.0,
"output_vector_size": 1536
},
+ "amazon.titan-embed-g1-text-02": {
+ "input_cost_per_token": 1e-07,
+ "litellm_provider": "bedrock",
+ "max_input_tokens": 8192,
+ "max_tokens": 8192,
+ "mode": "embedding",
+ "output_cost_per_token": 0.0,
+ "output_vector_size": 1536
+ },
"amazon.titan-embed-text-v2:0": {
"input_cost_per_token": 2e-08,
"litellm_provider": "bedrock",
@@ -14143,6 +14152,14 @@
"notes": "APISerpent deep search (/api/search), multi-engine (Google, Bing, Yahoo, DuckDuckGo). Pricing: $0.60/1k searches."
}
},
+ "tinyfish/search": {
+ "input_cost_per_query": 0.0,
+ "litellm_provider": "tinyfish",
+ "mode": "search",
+ "metadata": {
+ "notes": "TinyFish Search API"
+ }
+ },
"elevenlabs/scribe_v1": {
"input_cost_per_second": 6.11e-05,
"litellm_provider": "elevenlabs",
@@ -31357,13 +31374,13 @@
"output_cost_per_token": 0.0
},
"sambanova/MiniMax-M2.7": {
- "input_cost_per_token": 3e-07,
+ "input_cost_per_token": 6e-07,
"litellm_provider": "sambanova",
- "max_input_tokens": 204800,
+ "max_input_tokens": 196608,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
- "output_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 2.4e-06,
"source": "https://cloud.sambanova.ai/plans/pricing",
"supports_function_calling": true,
"supports_reasoning": true,
@@ -31380,6 +31397,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/DeepSeek-R1-Distill-Llama-70B": {
+ "deprecation_date": "2026-03-20",
"input_cost_per_token": 7e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 131072,
@@ -31390,6 +31408,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/DeepSeek-V3-0324": {
+ "deprecation_date": "2026-04-14",
"input_cost_per_token": 3e-06,
"litellm_provider": "sambanova",
"max_input_tokens": 32768,
@@ -31420,6 +31439,7 @@
"supports_vision": true
},
"sambanova/Llama-4-Scout-17B-16E-Instruct": {
+ "deprecation_date": "2025-06-19",
"input_cost_per_token": 4e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 8192,
@@ -31436,6 +31456,7 @@
"supports_tool_choice": true
},
"sambanova/Meta-Llama-3.1-405B-Instruct": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 5e-06,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31449,6 +31470,7 @@
"supports_tool_choice": true
},
"sambanova/Meta-Llama-3.1-8B-Instruct": {
+ "deprecation_date": "2026-04-14",
"input_cost_per_token": 1e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31462,6 +31484,7 @@
"supports_tool_choice": true
},
"sambanova/Meta-Llama-3.2-1B-Instruct": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 4e-08,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31472,6 +31495,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/Meta-Llama-3.2-3B-Instruct": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 8e-08,
"litellm_provider": "sambanova",
"max_input_tokens": 4096,
@@ -31495,6 +31519,7 @@
"supports_tool_choice": true
},
"sambanova/Meta-Llama-Guard-3-8B": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 3e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31505,6 +31530,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/QwQ-32B": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 5e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31515,6 +31541,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/Qwen2-Audio-7B-Instruct": {
+ "deprecation_date": "2025-06-19",
"input_cost_per_token": 5e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 4096,
@@ -31526,6 +31553,7 @@
"supports_audio_input": true
},
"sambanova/Qwen3-32B": {
+ "deprecation_date": "2026-04-06",
"input_cost_per_token": 4e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 8192,
@@ -31539,9 +31567,9 @@
"supports_tool_choice": true
},
"sambanova/DeepSeek-V3.1": {
- "max_tokens": 32768,
- "max_input_tokens": 32768,
- "max_output_tokens": 32768,
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
"input_cost_per_token": 3e-06,
"output_cost_per_token": 4.5e-06,
"litellm_provider": "sambanova",
@@ -31555,8 +31583,8 @@
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
- "input_cost_per_token": 3e-06,
- "output_cost_per_token": 4.5e-06,
+ "input_cost_per_token": 2.2e-07,
+ "output_cost_per_token": 5.9e-07,
"litellm_provider": "sambanova",
"mode": "chat",
"supports_function_calling": true,
@@ -31564,21 +31592,55 @@
"supports_reasoning": true,
"source": "https://cloud.sambanova.ai/plans/pricing"
},
- "snowflake/claude-3-5-sonnet": {
- "litellm_provider": "snowflake",
- "max_input_tokens": 18000,
- "max_output_tokens": 8192,
- "max_tokens": 8192,
- "mode": "chat",
- "supports_computer_use": true
- },
- "snowflake/deepseek-r1": {
- "litellm_provider": "snowflake",
+ "sambanova/DeepSeek-V3.2": {
+ "max_tokens": 32768,
"max_input_tokens": 32768,
- "max_output_tokens": 8192,
- "max_tokens": 8192,
+ "max_output_tokens": 32768,
+ "input_cost_per_token": 3e-06,
+ "output_cost_per_token": 4.5e-06,
+ "litellm_provider": "sambanova",
"mode": "chat",
- "supports_reasoning": true
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "source": "https://cloud.sambanova.ai/plans/pricing"
+ },
+ "sambanova/gemma-4-31B-it": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 3.8e-07,
+ "output_cost_per_token": 1.15e-06,
+ "litellm_provider": "sambanova",
+ "mode": "chat",
+ "supports_vision": true,
+ "source": "https://cloud.sambanova.ai/plans/pricing"
+ },
+ "snowflake/claude-3-5-sonnet": {
+ "litellm_provider": "snowflake",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "input_cost_per_token": 0.000003,
+ "output_cost_per_token": 0.000015,
+ "cache_read_input_token_cost": 0.0000003,
+ "supports_computer_use": true,
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_response_schema": true
+ },
+ "snowflake/deepseek-r1": {
+ "litellm_provider": "snowflake",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "input_cost_per_token": 0.00000135,
+ "output_cost_per_token": 0.0000054,
+ "supports_reasoning": true,
+ "supports_system_messages": true
},
"snowflake/gemma-7b": {
"litellm_provider": "snowflake",
@@ -31632,23 +31694,34 @@
"snowflake/llama3.1-405b": {
"litellm_provider": "snowflake",
"max_input_tokens": 128000,
- "max_output_tokens": 8192,
- "max_tokens": 8192,
- "mode": "chat"
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "input_cost_per_token": 0.0000012,
+ "output_cost_per_token": 0.0000012,
+ "supports_function_calling": true,
+ "supports_system_messages": true
},
"snowflake/llama3.1-70b": {
"litellm_provider": "snowflake",
"max_input_tokens": 128000,
- "max_output_tokens": 8192,
- "max_tokens": 8192,
- "mode": "chat"
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "input_cost_per_token": 0.00000072,
+ "output_cost_per_token": 0.00000072,
+ "supports_function_calling": true,
+ "supports_system_messages": true
},
"snowflake/llama3.1-8b": {
"litellm_provider": "snowflake",
"max_input_tokens": 128000,
- "max_output_tokens": 8192,
- "max_tokens": 8192,
- "mode": "chat"
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "input_cost_per_token": 0.00000024,
+ "output_cost_per_token": 0.00000024,
+ "supports_system_messages": true
},
"snowflake/llama3.2-1b": {
"litellm_provider": "snowflake",
@@ -31664,13 +31737,17 @@
"max_tokens": 8192,
"mode": "chat"
},
- "snowflake/llama3.3-70b": {
- "litellm_provider": "snowflake",
+ "snowflake/llama3.3-70b": {
+ "max_tokens": 16384,
"max_input_tokens": 128000,
- "max_output_tokens": 8192,
- "max_tokens": 8192,
- "mode": "chat"
- },
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.00000072,
+ "output_cost_per_token": 0.00000072,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_system_messages": true
+ },
"snowflake/mistral-7b": {
"litellm_provider": "snowflake",
"max_input_tokens": 32000,
@@ -31685,12 +31762,17 @@
"max_tokens": 8192,
"mode": "chat"
},
- "snowflake/mistral-large2": {
+ "snowflake/mistral-large2": {
"litellm_provider": "snowflake",
"max_input_tokens": 128000,
- "max_output_tokens": 8192,
- "max_tokens": 8192,
- "mode": "chat"
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "input_cost_per_token": 0.000002,
+ "output_cost_per_token": 0.000006,
+ "supports_function_calling": true,
+ "supports_system_messages": true,
+ "supports_response_schema": true
},
"snowflake/mixtral-8x7b": {
"litellm_provider": "snowflake",
@@ -31727,13 +31809,17 @@
"max_tokens": 8192,
"mode": "chat"
},
- "snowflake/snowflake-llama-3.3-70b": {
+ "snowflake/snowflake-llama-3.3-70b": {
+ "max_tokens": 16384,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.00000072,
+ "output_cost_per_token": 0.00000072,
"litellm_provider": "snowflake",
- "max_input_tokens": 8000,
- "max_output_tokens": 8192,
- "max_tokens": 8192,
- "mode": "chat"
- },
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_system_messages": true
+ },
"stability/sd3": {
"litellm_provider": "stability",
"mode": "image_generation",
@@ -32076,6 +32162,11 @@
"litellm_provider": "tavily",
"mode": "search"
},
+ "you_com/search": {
+ "input_cost_per_query": 0.0,
+ "litellm_provider": "you_com",
+ "mode": "search"
+ },
"text-completion-codestral/codestral-2405": {
"input_cost_per_token": 0.0,
"litellm_provider": "text-completion-codestral",
@@ -36595,17 +36686,7 @@
"max_input_tokens": 32000,
"max_tokens": 32000,
"mode": "embedding",
- "output_cost_per_token": 0.0,
- "supports_vision": true
- },
- "voyage/voyage-multimodal-3.5": {
- "input_cost_per_token": 1.2e-07,
- "litellm_provider": "voyage",
- "max_input_tokens": 32000,
- "max_tokens": 32000,
- "mode": "embedding",
- "output_cost_per_token": 0.0,
- "supports_vision": true
+ "output_cost_per_token": 0.0
},
"wandb/openai/gpt-oss-120b": {
"max_tokens": 131072,
@@ -40270,6 +40351,178 @@
"supports_tool_choice": true,
"supports_vision": true
},
+ "scaleway/qwen/qwen3.5-397b-a17b": {
+ "input_cost_per_token": 6e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 256000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "output_cost_per_token": 3.6e-06,
+ "supports_function_calling": true,
+ "supports_reasoning": true,
+ "supports_vision": true
+ },
+ "scaleway/qwen/qwen3.6-35b-a3b": {
+ "input_cost_per_token": 2.5e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 256000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "output_cost_per_token": 1.5e-06,
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_reasoning": true
+ },
+ "scaleway/qwen/qwen3-235b-a22b-instruct-2507": {
+ "input_cost_per_token": 7.5e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 256000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "output_cost_per_token": 2.25e-06,
+ "supports_function_calling": true
+ },
+ "scaleway/qwen/qwen3-embedding-8b": {
+ "input_cost_per_token": 1e-07,
+ "litellm_provider": "scaleway",
+ "mode": "embedding",
+ "output_cost_per_token": 0.0
+ },
+ "scaleway/qwen/qwen3-coder-30b-a3b-instruct": {
+ "input_cost_per_token": 2e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 32768,
+ "max_tokens": 32768,
+ "mode": "chat",
+ "output_cost_per_token": 8e-07,
+ "supports_function_calling": true
+ },
+ "scaleway/openai/gpt-oss-120b": {
+ "input_cost_per_token": 1.5e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 32768,
+ "max_tokens": 32768,
+ "mode": "chat",
+ "output_cost_per_token": 6e-07,
+ "supports_function_calling": true
+ },
+ "scaleway/openai/whisper-large-v3": {
+ "input_cost_per_audio_token": 0.0,
+ "litellm_provider": "scaleway",
+ "mode": "audio_transcription",
+ "output_cost_per_token": 0.0
+ },
+ "scaleway/google/gemma-4-26b-a4b-it": {
+ "input_cost_per_token": 2.5e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 256000,
+ "max_output_tokens": 32768,
+ "max_tokens": 32768,
+ "mode": "chat",
+ "output_cost_per_token": 5e-07,
+ "supports_function_calling": true,
+ "supports_reasoning": true,
+ "supports_vision": true
+ },
+ "scaleway/google/gemma-3-27b-it": {
+ "input_cost_per_token": 2.5e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 40000,
+ "max_output_tokens": 8192,
+ "max_tokens": 8192,
+ "mode": "chat",
+ "output_cost_per_token": 5e-07,
+ "supports_function_calling": true,
+ "supports_vision": true
+ },
+ "scaleway/hcompany/holo2-30b-a3b": {
+ "input_cost_per_token": 3e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 22000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "output_cost_per_token": 7e-07,
+ "supports_reasoning": true,
+ "supports_vision": true
+ },
+ "scaleway/mistralai/mistral-medium-3.5-128b": {
+ "input_cost_per_token": 1.5e-06,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 256000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "output_cost_per_token": 7.5e-06,
+ "supports_reasoning": true,
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_tool_choice": true
+ },
+ "scaleway/mistralai/devstral-2-123b-instruct-2512": {
+ "input_cost_per_token": 4e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 200000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "output_cost_per_token": 2e-06,
+ "supports_function_calling": true
+ },
+ "scaleway/mistralai/voxtral-small-24b-2507": {
+ "input_cost_per_audio_token": 1.5e-07,
+ "input_cost_per_token": 1.5e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 32000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "output_cost_per_token": 3.5e-07,
+ "supports_audio_input": true
+ },
+ "scaleway/mistralai/mistral-small-3.2-24b-instruct-2506": {
+ "input_cost_per_token": 1.5e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 32768,
+ "max_tokens": 32768,
+ "mode": "chat",
+ "output_cost_per_token": 3.5e-07,
+ "supports_function_calling": true,
+ "supports_vision": true
+ },
+ "scaleway/mistralai/pixtral-12b-2409": {
+ "input_cost_per_token": 2e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 4096,
+ "max_tokens": 4096,
+ "mode": "chat",
+ "output_cost_per_token": 2e-07,
+ "supports_vision": true,
+ "supports_function_calling": true
+ },
+ "scaleway/BAAI/bge-multilingual-gemma2": {
+ "input_cost_per_token": 1e-07,
+ "litellm_provider": "scaleway",
+ "mode": "embedding",
+ "output_cost_per_token": 0.0
+ },
+ "scaleway/meta/llama-3.3-70b-instruct": {
+ "input_cost_per_token": 9e-07,
+ "litellm_provider": "scaleway",
+ "max_input_tokens": 128000,
+ "max_output_tokens": 16384,
+ "max_tokens": 16384,
+ "mode": "chat",
+ "output_cost_per_token": 9e-07,
+ "supports_function_calling": true
+ },
"novita/deepseek/deepseek-v3.2": {
"litellm_provider": "novita",
"mode": "chat",
@@ -43051,192 +43304,362 @@
"supports_native_structured_output": true,
"supports_pdf_input": true
},
- "soniox/stt-async-v4": {
- "litellm_provider": "soniox",
- "max_output_tokens": 8000,
- "max_tokens": 8000,
- "input_cost_per_second": 0.0,
- "output_cost_per_second": 0.0000277778,
- "mode": "audio_transcription",
- "source": "https://soniox.com/pricing",
- "supported_endpoints": [
- "/v1/audio/transcriptions"
- ],
- "supports_audio_input": true
- },
- "soniox/stt-async-v5": {
- "litellm_provider": "soniox",
- "max_output_tokens": 8000,
- "max_tokens": 8000,
- "input_cost_per_second": 0.0,
- "output_cost_per_second": 0.0000277778,
- "mode": "audio_transcription",
- "source": "https://soniox.com/pricing",
- "supported_endpoints": [
- "/v1/audio/transcriptions"
- ],
- "supports_audio_input": true
- },
- "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 6e-07,
- "output_cost_per_token": 3.6e-06,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 262144,
- "max_output_tokens": 262144,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "supports_reasoning": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- },
- "tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 4.5e-07,
- "output_cost_per_token": 1.8e-06,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 262144,
- "max_output_tokens": 262144,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- },
- "tensormesh/Qwen/Qwen3.6-27B-FP8": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 3.2e-07,
- "output_cost_per_token": 3.2e-06,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 262144,
- "max_output_tokens": 262144,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "supports_reasoning": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- },
- "tensormesh/lukealonso/GLM-5.1-NVFP4-MTP": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 1.4e-06,
- "output_cost_per_token": 4.4e-06,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 202752,
- "max_output_tokens": 202752,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "supports_reasoning": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- },
- "tensormesh/deepseek-ai/DeepSeek-V4-Flash": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 1.4e-07,
- "output_cost_per_token": 2.8e-07,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 32768,
- "max_output_tokens": 32768,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "supports_reasoning": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- },
- "tensormesh/moonshotai/Kimi-K2.6": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 9.6e-07,
- "output_cost_per_token": 4e-06,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 32768,
- "max_output_tokens": 32768,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "supports_reasoning": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- },
- "tensormesh/MiniMaxAI/MiniMax-M2.5": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 3e-07,
- "output_cost_per_token": 1.2e-06,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 196608,
- "max_output_tokens": 196608,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "supports_reasoning": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- },
- "tensormesh/google/gemma-4-31B-it": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 1.4e-07,
- "output_cost_per_token": 5.6e-07,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 32768,
- "max_output_tokens": 32768,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "supports_reasoning": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- },
- "tensormesh/openai/gpt-oss-120b": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 1.5e-07,
- "output_cost_per_token": 6e-07,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 131072,
- "max_output_tokens": 131072,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "supports_reasoning": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- },
- "tensormesh/openai/gpt-oss-20b": {
- "litellm_provider": "tensormesh",
- "mode": "chat",
- "input_cost_per_token": 7e-08,
- "output_cost_per_token": 2.8e-07,
- "cache_read_input_token_cost": 0,
- "max_input_tokens": 131072,
- "max_output_tokens": 131072,
- "supports_function_calling": true,
- "supports_tool_choice": true,
- "supports_response_schema": true,
- "supports_prompt_caching": true,
- "supports_system_messages": true,
- "supports_reasoning": true,
- "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
- }
-,
+ "snowflake/claude-sonnet-4-5": {
+ "max_tokens": 16384,
+ "max_input_tokens": 200000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.000003,
+ "output_cost_per_token": 0.000015,
+ "cache_read_input_token_cost": 0.0000003,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_response_schema": true
+ },
+ "snowflake/claude-sonnet-4-6": {
+ "max_tokens": 16384,
+ "max_input_tokens": 200000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.000003,
+ "output_cost_per_token": 0.000015,
+ "cache_read_input_token_cost": 0.0000003,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_response_schema": true
+ },
+ "snowflake/claude-4-sonnet": {
+ "max_tokens": 16384,
+ "max_input_tokens": 200000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.000003,
+ "output_cost_per_token": 0.000015,
+ "cache_read_input_token_cost": 0.0000003,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_response_schema": true
+ },
+ "snowflake/claude-4-opus": {
+ "max_tokens": 16384,
+ "max_input_tokens": 200000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.000005,
+ "output_cost_per_token": 0.000025,
+ "cache_read_input_token_cost": 0.0000005,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true
+ },
+ "snowflake/claude-haiku-4-5": {
+ "max_tokens": 16384,
+ "max_input_tokens": 200000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.000001,
+ "output_cost_per_token": 0.000005,
+ "cache_read_input_token_cost": 0.0000001,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_response_schema": true
+ },
+ "snowflake/claude-3-7-sonnet": {
+ "max_tokens": 16384,
+ "max_input_tokens": 200000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.000003,
+ "output_cost_per_token": 0.000015,
+ "cache_read_input_token_cost": 0.0000003,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true
+ },
+ "snowflake/openai-gpt-4.1": {
+ "max_tokens": 16384,
+ "max_input_tokens": 300000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.000002,
+ "output_cost_per_token": 0.000008,
+ "cache_read_input_token_cost": 0.0000005,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_response_schema": true
+ },
+ "snowflake/openai-gpt-5": {
+ "max_tokens": 16384,
+ "max_input_tokens": 300000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.00000125,
+ "output_cost_per_token": 0.00001,
+ "cache_read_input_token_cost": 0.000000125,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "supports_response_schema": true
+ },
+ "snowflake/openai-gpt-5-mini": {
+ "max_tokens": 16384,
+ "max_input_tokens": 1000000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.0000003,
+ "output_cost_per_token": 0.0000012,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_system_messages": true,
+ "supports_response_schema": true
+ },
+ "snowflake/openai-gpt-5-nano": {
+ "max_tokens": 16384,
+ "max_input_tokens": 5000000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.00000015,
+ "output_cost_per_token": 0.0000006,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_system_messages": true,
+ "supports_response_schema": true
+ },
+ "snowflake/llama4-maverick": {
+ "max_tokens": 16384,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 16384,
+ "input_cost_per_token": 0.00000024,
+ "output_cost_per_token": 0.00000097,
+ "litellm_provider": "snowflake",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_system_messages": true
+ },
+ "snowflake/snowflake-arctic-embed-l-v2.0": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "input_cost_per_token": 0.00000007,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "snowflake",
+ "mode": "embedding"
+ },
+ "snowflake/snowflake-arctic-embed-m-v2.0": {
+ "max_tokens": 8192,
+ "max_input_tokens": 8192,
+ "input_cost_per_token": 0.00000007,
+ "output_cost_per_token": 0.0,
+ "litellm_provider": "snowflake",
+ "mode": "embedding"
+ },
+ "soniox/stt-async-v4": {
+ "litellm_provider": "soniox",
+ "max_output_tokens": 8000,
+ "max_tokens": 8000,
+ "input_cost_per_second": 0.0,
+ "output_cost_per_second": 0.0000277778,
+ "mode": "audio_transcription",
+ "source": "https://soniox.com/pricing",
+ "supported_endpoints": ["/v1/audio/transcriptions"],
+ "supports_audio_input": true
+ },
+ "soniox/stt-async-v5": {
+ "litellm_provider": "soniox",
+ "max_output_tokens": 8000,
+ "max_tokens": 8000,
+ "input_cost_per_second": 0.0,
+ "output_cost_per_second": 0.0000277778,
+ "mode": "audio_transcription",
+ "source": "https://soniox.com/pricing",
+ "supported_endpoints": ["/v1/audio/transcriptions"],
+ "supports_audio_input": true
+ },
+ "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 6e-07,
+ "output_cost_per_token": 3.6e-06,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ },
+ "tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 4.5e-07,
+ "output_cost_per_token": 1.8e-06,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ },
+ "tensormesh/Qwen/Qwen3.6-27B-FP8": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 3.2e-07,
+ "output_cost_per_token": 3.2e-06,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 262144,
+ "max_output_tokens": 262144,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ },
+ "tensormesh/lukealonso/GLM-5.1-NVFP4-MTP": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 1.4e-06,
+ "output_cost_per_token": 4.4e-06,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 202752,
+ "max_output_tokens": 202752,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ },
+ "tensormesh/deepseek-ai/DeepSeek-V4-Flash": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 1.4e-07,
+ "output_cost_per_token": 2.8e-07,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ },
+ "tensormesh/moonshotai/Kimi-K2.6": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 9.6e-07,
+ "output_cost_per_token": 4e-06,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ },
+ "tensormesh/MiniMaxAI/MiniMax-M2.5": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 3e-07,
+ "output_cost_per_token": 1.2e-06,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 196608,
+ "max_output_tokens": 196608,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ },
+ "tensormesh/google/gemma-4-31B-it": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 1.4e-07,
+ "output_cost_per_token": 5.6e-07,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ },
+ "tensormesh/openai/gpt-oss-120b": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 1.5e-07,
+ "output_cost_per_token": 6e-07,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ },
+ "tensormesh/openai/gpt-oss-20b": {
+ "litellm_provider": "tensormesh",
+ "mode": "chat",
+ "input_cost_per_token": 7e-08,
+ "output_cost_per_token": 2.8e-07,
+ "cache_read_input_token_cost": 0,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_response_schema": true,
+ "supports_prompt_caching": true,
+ "supports_system_messages": true,
+ "supports_reasoning": true,
+ "source": "https://serverless.tensormesh.ai/v1/models/openrouter"
+ }
+ ,
"deepseek-v4-flash": {
"cache_creation_input_token_cost": 0.0,
"cache_read_input_token_cost": 2.8e-09,
@@ -43370,5 +43793,83 @@
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": false
+ },
+ "pinstripes/ps/glm-4.5-air": {
+ "max_tokens": 128000,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 128000,
+ "input_cost_per_token": 0.000000125,
+ "output_cost_per_token": 0.00000045,
+ "litellm_provider": "pinstripes",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_assistant_prefill": true,
+ "supports_reasoning": true,
+ "source": "https://pinstripes.io/pricing"
+ },
+ "pinstripes/ps/qwen3.6-35b-a3b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 0.00000014,
+ "output_cost_per_token": 0.00000045,
+ "litellm_provider": "pinstripes",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_assistant_prefill": true,
+ "supports_reasoning": true,
+ "source": "https://pinstripes.io/pricing"
+ },
+ "pinstripes/ps/qwen3-30b-a3b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 0.00000009,
+ "output_cost_per_token": 0.0000002,
+ "litellm_provider": "pinstripes",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_assistant_prefill": true,
+ "supports_reasoning": true,
+ "source": "https://pinstripes.io/pricing"
+ },
+ "pinstripes/ps/qwen3-coder-30b-a3b": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 0.0000003,
+ "output_cost_per_token": 0.0000006,
+ "litellm_provider": "pinstripes",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_assistant_prefill": true,
+ "supports_reasoning": false,
+ "source": "https://pinstripes.io/pricing"
+ },
+ "pinstripes/ps/deepseek-v4-flash": {
+ "max_tokens": 163840,
+ "max_input_tokens": 163840,
+ "max_output_tokens": 163840,
+ "input_cost_per_token": 0.0000001,
+ "output_cost_per_token": 0.0000002,
+ "litellm_provider": "pinstripes",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_assistant_prefill": true,
+ "supports_reasoning": true,
+ "source": "https://pinstripes.io/pricing"
+ },
+ "pinstripes/ps/minimax-m2.7": {
+ "max_tokens": 1000192,
+ "max_input_tokens": 1000192,
+ "max_output_tokens": 1000192,
+ "input_cost_per_token": 0.000000255,
+ "output_cost_per_token": 0.00000055,
+ "litellm_provider": "pinstripes",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_assistant_prefill": true,
+ "supports_reasoning": false,
+ "source": "https://pinstripes.io/pricing"
}
}
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index d7017a40993..cffb00c6b22 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -570,6 +570,15 @@
"output_cost_per_token": 0.0,
"output_vector_size": 1536
},
+ "amazon.titan-embed-g1-text-02": {
+ "input_cost_per_token": 1e-07,
+ "litellm_provider": "bedrock",
+ "max_input_tokens": 8192,
+ "max_tokens": 8192,
+ "mode": "embedding",
+ "output_cost_per_token": 0.0,
+ "output_vector_size": 1536
+ },
"amazon.titan-embed-text-v2:0": {
"input_cost_per_token": 2e-08,
"litellm_provider": "bedrock",
@@ -31381,13 +31390,13 @@
"output_cost_per_token": 0.0
},
"sambanova/MiniMax-M2.7": {
- "input_cost_per_token": 3e-07,
+ "input_cost_per_token": 6e-07,
"litellm_provider": "sambanova",
- "max_input_tokens": 204800,
+ "max_input_tokens": 196608,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
- "output_cost_per_token": 1.2e-06,
+ "output_cost_per_token": 2.4e-06,
"source": "https://cloud.sambanova.ai/plans/pricing",
"supports_function_calling": true,
"supports_reasoning": true,
@@ -31404,6 +31413,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/DeepSeek-R1-Distill-Llama-70B": {
+ "deprecation_date": "2026-03-20",
"input_cost_per_token": 7e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 131072,
@@ -31414,6 +31424,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/DeepSeek-V3-0324": {
+ "deprecation_date": "2026-04-14",
"input_cost_per_token": 3e-06,
"litellm_provider": "sambanova",
"max_input_tokens": 32768,
@@ -31444,6 +31455,7 @@
"supports_vision": true
},
"sambanova/Llama-4-Scout-17B-16E-Instruct": {
+ "deprecation_date": "2025-06-19",
"input_cost_per_token": 4e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 8192,
@@ -31460,6 +31472,7 @@
"supports_tool_choice": true
},
"sambanova/Meta-Llama-3.1-405B-Instruct": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 5e-06,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31473,6 +31486,7 @@
"supports_tool_choice": true
},
"sambanova/Meta-Llama-3.1-8B-Instruct": {
+ "deprecation_date": "2026-04-14",
"input_cost_per_token": 1e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31486,6 +31500,7 @@
"supports_tool_choice": true
},
"sambanova/Meta-Llama-3.2-1B-Instruct": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 4e-08,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31496,6 +31511,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/Meta-Llama-3.2-3B-Instruct": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 8e-08,
"litellm_provider": "sambanova",
"max_input_tokens": 4096,
@@ -31519,6 +31535,7 @@
"supports_tool_choice": true
},
"sambanova/Meta-Llama-Guard-3-8B": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 3e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31529,6 +31546,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/QwQ-32B": {
+ "deprecation_date": "2025-06-25",
"input_cost_per_token": 5e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 16384,
@@ -31539,6 +31557,7 @@
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"sambanova/Qwen2-Audio-7B-Instruct": {
+ "deprecation_date": "2025-06-19",
"input_cost_per_token": 5e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 4096,
@@ -31550,6 +31569,7 @@
"supports_audio_input": true
},
"sambanova/Qwen3-32B": {
+ "deprecation_date": "2026-04-06",
"input_cost_per_token": 4e-07,
"litellm_provider": "sambanova",
"max_input_tokens": 8192,
@@ -31563,9 +31583,9 @@
"supports_tool_choice": true
},
"sambanova/DeepSeek-V3.1": {
- "max_tokens": 32768,
- "max_input_tokens": 32768,
- "max_output_tokens": 32768,
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
"input_cost_per_token": 3e-06,
"output_cost_per_token": 4.5e-06,
"litellm_provider": "sambanova",
@@ -31579,13 +31599,36 @@
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
+ "input_cost_per_token": 2.2e-07,
+ "output_cost_per_token": 5.9e-07,
+ "litellm_provider": "sambanova",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_tool_choice": true,
+ "supports_reasoning": true,
+ "source": "https://cloud.sambanova.ai/plans/pricing"
+ },
+ "sambanova/DeepSeek-V3.2": {
+ "max_tokens": 32768,
+ "max_input_tokens": 32768,
+ "max_output_tokens": 32768,
"input_cost_per_token": 3e-06,
"output_cost_per_token": 4.5e-06,
"litellm_provider": "sambanova",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
- "supports_reasoning": true,
+ "source": "https://cloud.sambanova.ai/plans/pricing"
+ },
+ "sambanova/gemma-4-31B-it": {
+ "max_tokens": 131072,
+ "max_input_tokens": 131072,
+ "max_output_tokens": 131072,
+ "input_cost_per_token": 3.8e-07,
+ "output_cost_per_token": 1.15e-06,
+ "litellm_provider": "sambanova",
+ "mode": "chat",
+ "supports_vision": true,
"source": "https://cloud.sambanova.ai/plans/pricing"
},
"snowflake/claude-3-5-sonnet": {
diff --git a/tests/llm_translation/test_bedrock_embedding.py b/tests/llm_translation/test_bedrock_embedding.py
index 92c22f582d9..2bc4192833b 100644
--- a/tests/llm_translation/test_bedrock_embedding.py
+++ b/tests/llm_translation/test_bedrock_embedding.py
@@ -34,6 +34,11 @@ img_base_64 = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAGQAAABkBAMAAACCzIh
"text",
titan_embedding_response,
), # V2 text model
+ (
+ "bedrock/amazon.titan-embed-g1-text-02",
+ "text",
+ titan_embedding_response,
+ ), # G1 text model
(
"bedrock/amazon.titan-embed-image-v1",
"image",
@@ -459,3 +464,13 @@ def test_bedrock_embedding_region_bug_reproduction():
os.environ["AWS_REGION_NAME"] = original_region_name
else:
os.environ.pop("AWS_REGION_NAME", None)
+
+
+def test_bedrock_titan_g1_text_02_model_info():
+ """Test that amazon.titan-embed-g1-text-02 has correct pricing metadata"""
+ model_info = litellm.get_model_info("amazon.titan-embed-g1-text-02")
+ assert model_info is not None, "Model info should not be None"
+ assert model_info["litellm_provider"] == "bedrock"
+ assert model_info["mode"] == "embedding"
+ assert model_info["input_cost_per_token"] == 1e-07
+ assert model_info["max_input_tokens"] == 8192
From 6bf3b9b6db64d80bd91c1a59c8b2bb334d7d2f87 Mon Sep 17 00:00:00 2001
From: Ewertonslv
Date: Wed, 24 Jun 2026 08:29:25 -0300
Subject: [PATCH 30/30] fix(utils): preserve arbitrary above-threshold tiered
pricing keys in get_model_info (#30880)
* fix(utils): preserve arbitrary above-threshold tiered pricing keys in get_model_info
get_model_info rebuilt ModelInfo by copying a fixed allow-list of
input/output_cost_per_token_above__tokens keys (128k/200k/272k/512k), so any other
threshold a user registered was dropped before reaching _get_token_base_cost, which already
reads an arbitrary threshold out of the key name. Custom tiers such as above_500k_tokens were
silently ignored and billing fell back to the base per-token rate. Carry over any
_above__tokens cost key present on the source cost-map entry that the fixed fields miss
Fixes #30344
* test(cost): keep suite hermetic by popping the temp tiered-pricing model
Wrap the regression body in try/finally so litellm.model_cost no longer
leaks the litellm-test-non-standard-tier entry into later tests that
iterate or reset the global cost map. Addresses Greptile review thread.
---
litellm/utils.py | 12 +++++-
.../llm_cost_calc/test_llm_cost_calc_utils.py | 38 +++++++++++++++++++
2 files changed, 49 insertions(+), 1 deletion(-)
diff --git a/litellm/utils.py b/litellm/utils.py
index 5c3ab3e1490..0a7dd1a1b8f 100644
--- a/litellm/utils.py
+++ b/litellm/utils.py
@@ -5844,6 +5844,9 @@ def _is_potential_model_name_in_model_cost(
)
+_ABOVE_THRESHOLD_COST_KEY = re.compile(r"_above_\d+k?_tokens$")
+
+
def _get_model_info_helper(
model: str,
custom_llm_provider: Optional[str] = None,
@@ -6021,7 +6024,7 @@ def _get_model_info_helper(
)
_output_cost_per_token = 0
- return ModelInfoBase(
+ returned_model_info = ModelInfoBase(
key=key,
max_tokens=_model_info.get("max_tokens", None),
max_input_tokens=_model_info.get("max_input_tokens", None),
@@ -6238,6 +6241,13 @@ def _get_model_info_helper(
uses_embed_content=_model_info.get("uses_embed_content", None),
supports_image_size=_model_info.get("supports_image_size", None),
)
+ for cost_key, cost_value in _model_info.items():
+ if (
+ cost_key not in returned_model_info
+ and _ABOVE_THRESHOLD_COST_KEY.search(cost_key) is not None
+ ):
+ returned_model_info[cost_key] = cost_value # type: ignore[literal-required]
+ return returned_model_info
except Exception as e:
verbose_logger.debug(f"Error getting model info: {e}")
raise Exception(
diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py
index 7f3d5a959a1..d47558d302d 100644
--- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py
+++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py
@@ -384,6 +384,44 @@ def test_generic_cost_per_token_minimax_m3_above_512k_tokens():
assert round(completion_cost, 10) == round(expected_completion, 10)
+def test_generic_cost_per_token_honors_non_standard_above_threshold():
+ """Regression for #30344: get_model_info must keep arbitrary
+ input/output_cost_per_token_above__tokens thresholds, not only the hard-coded
+ 128k/200k/272k/512k set, so a custom tier boundary is applied past its limit."""
+ model = "litellm-test-non-standard-tier"
+ custom_llm_provider = "openai"
+ litellm.register_model(
+ {
+ model: {
+ "litellm_provider": custom_llm_provider,
+ "mode": "chat",
+ "input_cost_per_token": 1e-6,
+ "output_cost_per_token": 2e-6,
+ "input_cost_per_token_above_500k_tokens": 9e-6,
+ "output_cost_per_token_above_500k_tokens": 18e-6,
+ }
+ }
+ )
+
+ try:
+ prompt_tokens = 600000
+ completion_tokens = 1000
+ usage = Usage(
+ prompt_tokens=prompt_tokens,
+ completion_tokens=completion_tokens,
+ total_tokens=prompt_tokens + completion_tokens,
+ )
+ prompt_cost, completion_cost = generic_cost_per_token(
+ model=model,
+ usage=usage,
+ custom_llm_provider=custom_llm_provider,
+ )
+ assert round(prompt_cost, 10) == round(9e-6 * prompt_tokens, 10)
+ assert round(completion_cost, 10) == round(18e-6 * completion_tokens, 10)
+ finally:
+ litellm.model_cost.pop(model, None)
+
+
def test_generic_cost_per_token_gpt55():
"""gpt-5.5: base pricing — $5/1M input, $30/1M output, $0.50/1M cached input."""
model = "gpt-5.5"