Merge remote-tracking branch 'origin/litellm_internal_staging' into litellm_health_test_connection_health_check_params

# Conflicts:
#	tests/test_litellm/proxy/test_health_check_max_tokens.py
This commit is contained in:
mateo-berri 2026-08-24 11:30:04 -07:00
commit 531aafdf75
76 changed files with 4393 additions and 304 deletions

View file

@ -4,6 +4,7 @@ Custom A2A Card Resolver for LiteLLM.
Extends the A2A SDK's card resolver to support multiple well-known paths.
"""
from types import MappingProxyType
from typing import TYPE_CHECKING, Any, Final
from litellm._logging import verbose_logger
@ -48,6 +49,43 @@ def is_localhost_or_internal_url(url: str | None) -> bool:
return any(pattern in url_lower for pattern in LOCALHOST_URL_PATTERNS)
_CANONICAL_PROTOCOL_BINDINGS: Final = MappingProxyType(
{
"jsonrpc": "JSONRPC",
"http+json": "HTTP+JSON",
"grpc": "GRPC",
}
)
_LEGACY_PROTOCOL_VERSION: Final = "0.3"
def normalize_agent_card_interfaces(agent_card: "AgentCard") -> "AgentCard":
"""
Canonicalize the supported interfaces of spec-adjacent agent cards.
Some A2A servers (e.g. LangGraph Platform) serve agent cards with lowercase
bindings like "jsonrpc", but a2a-sdk's ClientFactory matches bindings
case-sensitively against its uppercase TransportProtocol constants and fails
with "no compatible transports found." for spec-adjacent casings.
The same servers also speak the A2A 0.3 JSON dialect ("kind"-discriminated
payloads) while declaring protocolVersion "1.0", which a2a-sdk's strict v1
proto parsing rejects. A mis-cased binding fingerprints such a server, so its
declared version is downgraded to 0.3 to route the SDK's ClientFactory onto
its v0.3 compat transport, which speaks that dialect.
"""
normalized: Final = type(agent_card)()
normalized.CopyFrom(agent_card)
for interface in normalized.supported_interfaces:
canonical: str | None = _CANONICAL_PROTOCOL_BINDINGS.get(interface.protocol_binding.lower())
if canonical is None or canonical == interface.protocol_binding:
continue
interface.protocol_binding = canonical
interface.protocol_version = _LEGACY_PROTOCOL_VERSION
return normalized
def get_agent_card_url(agent_card: "AgentCard") -> str | None:
"""Return the agent endpoint URL from the resolved SDK card."""
url: Final = getattr(agent_card, "url", None)

View file

@ -73,6 +73,7 @@ except ImportError:
from litellm.a2a_protocol.card_resolver import (
LiteLLMA2ACardResolver,
get_agent_card_url,
normalize_agent_card_interfaces,
)
from litellm.a2a_protocol.exception_mapping_utils import (
handle_a2a_localhost_retry,
@ -782,13 +783,17 @@ async def create_a2a_client(
if extra_headers:
verbose_proxy_logger.debug("A2A client created with extra_headers=%s", list(extra_headers.keys()))
resolver: Final = A2ACardResolver(httpx_client=httpx_client, base_url=base_url)
agent_card: Final = normalize_agent_card_interfaces(
await resolver.get_agent_card(http_kwargs={"headers": extra_headers} if extra_headers else None)
)
a2a_client: Final = await create_client( # pyright: ignore[reportOptionalCall]
base_url,
agent_card,
client_config=ClientConfig( # pyright: ignore[reportOptionalCall]
httpx_client=httpx_client,
streaming=streaming,
),
resolver_http_kwargs={"headers": extra_headers} if extra_headers else None,
)
# Stash LiteLLM-owned handles on the client so the localhost-retry path can reuse
# the configured httpx client and this agent's headers without excavating
@ -799,9 +804,7 @@ async def create_a2a_client(
if extra_headers
else None
)
agent_card: Final = getattr(a2a_client, "_card", None)
if agent_card is not None:
a2a_client._litellm_agent_card = agent_card
a2a_client._litellm_agent_card = agent_card
verbose_logger.info("A2A client created for %s", base_url)

View file

@ -21,6 +21,9 @@ from pydantic import BaseModel
import litellm
from litellm import ModelResponse
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.prompt_templates.common_utils import (
responses_reasoning_item_from_thinking_blocks,
)
from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator
from litellm.llms.base_llm.bridges.completion_transformation import (
CompletionTransformationBridge,
@ -85,6 +88,22 @@ def _get_reasoning_items(
return []
def _reasoning_input_items(msg: "AllMessageValues") -> list[dict[str, object]]: # mutable-ok: API message payload
"""Reasoning input items for an assistant message.
Stored reasoning items win because they carry an id the Responses API minted; thinking
blocks are the fallback for turns that arrived over another API surface.
"""
items: Final = _get_reasoning_items(msg)
stored: Final = [_reasoning_item_to_response_input(item) for item in items] # mutable-ok: API message payload
if stored:
return stored
raw_blocks: Final = msg.get("thinking_blocks") or ()
blocks: Final = cast("Iterable[ChatCompletionThinkingBlock]", raw_blocks) # cast-ok: untyped client json
from_thinking: Final = responses_reasoning_item_from_thinking_blocks(blocks)
return [] if from_thinking is None else [dict(from_thinking)] # mutable-ok: API message payload
def _build_reasoning_item(
item_id: str,
encrypted_content: str | None,
@ -372,8 +391,15 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
)
)
elif role == "assistant" and tool_calls and isinstance(tool_calls, list):
for r_item in _get_reasoning_items(msg):
input_items.append(_reasoning_item_to_response_input(r_item))
input_items.extend(_reasoning_input_items(msg))
if content:
input_items.append(
{ # mutable-ok: API message payload
"type": "message",
"role": "assistant",
"content": self._convert_content_to_responses_format(content, "assistant"),
}
)
for tool_call in tool_calls:
function = tool_call.get("function")
custom = tool_call.get("custom")
@ -400,15 +426,16 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
raise ValueError(f"tool call not supported: {tool_call}")
elif content is not None:
if role == "assistant":
for r_item in _get_reasoning_items(msg):
input_items.append(_reasoning_item_to_response_input(r_item))
input_items.extend(_reasoning_input_items(msg))
input_items.append(
{
{ # mutable-ok: API message payload
"type": "message",
"role": role,
"content": self._convert_content_to_responses_format(content, cast(str, role)),
}
)
elif role == "assistant":
input_items.extend(_reasoning_input_items(msg))
return input_items, instructions

View file

@ -1563,6 +1563,19 @@ STALE_OBJECT_CLEANUP_BATCH_SIZE: Final = max(1, int(os.getenv("STALE_OBJECT_CLEA
# installations with large numbers of stale managed objects).
_batch_polling_env: Final = os.getenv("PROXY_BATCH_POLLING_ENABLED", "true").lower()
PROXY_BATCH_POLLING_ENABLED: Final = _batch_polling_env == "true"
BACKGROUND_INTERACTION_COST_POLL_INITIAL_INTERVAL_SECONDS: Final = float(
os.getenv("BACKGROUND_INTERACTION_COST_POLL_INITIAL_INTERVAL_SECONDS", "5")
)
BACKGROUND_INTERACTION_COST_POLL_MAX_INTERVAL_SECONDS: Final = float(
os.getenv("BACKGROUND_INTERACTION_COST_POLL_MAX_INTERVAL_SECONDS", "60")
)
BACKGROUND_INTERACTION_COST_POLL_TIMEOUT_SECONDS: Final = float(
os.getenv("BACKGROUND_INTERACTION_COST_POLL_TIMEOUT_SECONDS", "3600")
)
_background_interaction_cost_polling_env: Final = os.getenv(
"BACKGROUND_INTERACTION_COST_POLLING_ENABLED", "true"
).lower()
BACKGROUND_INTERACTION_COST_POLLING_ENABLED: Final = _background_interaction_cost_polling_env == "true"
PROXY_BUDGET_RESCHEDULER_MAX_TIME: Final = int(os.getenv("PROXY_BUDGET_RESCHEDULER_MAX_TIME", 605))
PROXY_BATCH_WRITE_AT: Final = int(os.getenv("PROXY_BATCH_WRITE_AT", 10)) # in seconds, increased from 10
PROXY_CONFIG_RELOAD_INTERVAL_SECONDS: Final = get_env_int("PROXY_CONFIG_RELOAD_INTERVAL_SECONDS", 30)

View file

@ -19,6 +19,7 @@ from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
StandardBuiltInToolCostTracking,
)
from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import (
InteractionsUsageObjectTransformation,
TranscriptionUsageObjectTransformation,
)
from litellm.litellm_core_utils.llm_cost_calc.utils import (
@ -912,6 +913,8 @@ def _get_usage_object(
usage_obj,
)
)
elif isinstance(usage_obj, dict) and InteractionsUsageObjectTransformation.is_interactions_usage_object(usage_obj):
return InteractionsUsageObjectTransformation.transform_interactions_usage_object(usage_obj)
elif isinstance(usage_obj, dict):
return Usage(**usage_obj)
elif isinstance(usage_obj, BaseModel):
@ -1288,6 +1291,10 @@ def completion_cost(
)
if tr_usage is not None:
_usage = tr_usage.model_dump()
elif InteractionsUsageObjectTransformation.is_interactions_usage_object(_usage):
_usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
_usage
).model_dump()
else:
_usage = _usage

View file

@ -0,0 +1,313 @@
"""
Cost tracking for background interactions.
A create request with ``background=true`` returns ``in_progress`` with no
usage block, and GET polls are deliberately never billed (billing them would
double-charge every poll; the GET response also does not echo ``background``,
so a poll cannot be told apart from a re-fetch of an already-billed
interaction). The create call is therefore the only place that can own
billing: it schedules a poll task that fetches the interaction until it
reaches a terminal status and logs the final usage as a single success event
attributed to the original request.
``requires_action`` is terminal for the interaction it names. The API has no
operation that resumes one: a caller answers a tool request by creating a new
interaction whose ``previous_interaction_id`` points at it, and that new
interaction bills itself. The paused interaction keeps the tokens it already
spent producing the tool request, so it is billed and settled where it stops
rather than polled until the timeout, which would both lose that usage and
hold its budget reservation open for the whole timeout window.
Deleting an interaction makes every subsequent poll fail, which would let a
caller retrieve the completed output themselves and then delete it before the
poll task settles, leaving the work unbilled and the budget reservation
refunded at the poll timeout. ``adelete`` therefore settles any pending poll
for the interaction before dispatching the delete: it fetches the current
state with the create's credentials, bills it if it is terminal with usage,
and releases the reservation otherwise. A settlement gate on the create's
logging object makes the poll task and the delete path mutually exclusive, so
the interaction is billed exactly once no matter who settles first.
"""
import asyncio
from collections.abc import Awaitable, Callable, Iterator, Mapping
from dataclasses import dataclass
from typing import TYPE_CHECKING, Final, TypeAlias
from litellm._logging import verbose_logger
from litellm.constants import (
BACKGROUND_INTERACTION_COST_POLL_INITIAL_INTERVAL_SECONDS,
BACKGROUND_INTERACTION_COST_POLL_MAX_INTERVAL_SECONDS,
BACKGROUND_INTERACTION_COST_POLL_TIMEOUT_SECONDS,
BACKGROUND_INTERACTION_COST_POLLING_ENABLED,
)
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs
from litellm.types.interactions import InteractionsAPIResponse
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
_TERMINAL_STATUSES: Final = frozenset(
{"completed", "failed", "cancelled", "incomplete", "budget_exceeded", "requires_action"}
)
_POLLABLE_STATUSES: Final = frozenset({"in_progress", "queued"})
_STATUSES_THAT_PRODUCED_OUTPUT: Final = frozenset({"completed", "requires_action"})
@dataclass(frozen=True, slots=True)
class BackgroundInteractionPollContext:
interaction_id: str
custom_llm_provider: str
logging_obj: "LiteLLMLoggingObj"
api_key: str | None = None
api_base: str | None = None
initial_interval_seconds: float = BACKGROUND_INTERACTION_COST_POLL_INITIAL_INTERVAL_SECONDS
max_interval_seconds: float = BACKGROUND_INTERACTION_COST_POLL_MAX_INTERVAL_SECONDS
timeout_seconds: float = BACKGROUND_INTERACTION_COST_POLL_TIMEOUT_SECONDS
FetchInteraction: TypeAlias = Callable[[BackgroundInteractionPollContext], Awaitable[InteractionsAPIResponse]]
async def _fetch_interaction(context: BackgroundInteractionPollContext) -> InteractionsAPIResponse:
from litellm.interactions import aget
return await aget(
interaction_id=context.interaction_id,
custom_llm_provider=context.custom_llm_provider,
api_key=context.api_key,
api_base=context.api_base,
**{
"no-log": True
}, # mutable-ok: "no-log" is not a valid identifier, so it can only be passed through a mapping
)
def _poll_intervals(initial: float, maximum: float, timeout: float) -> Iterator[float]:
elapsed = 0.0
interval = initial
while interval > 0 and elapsed + interval <= timeout:
yield interval
elapsed += interval
interval = min(interval * 2, maximum)
_SETTLED_KEY = "background_interaction_settled"
def _is_settled(logging_obj: "LiteLLMLoggingObj") -> bool:
return logging_obj.model_call_details.get(_SETTLED_KEY) is True
def _claim_settlement(logging_obj: "LiteLLMLoggingObj") -> bool:
"""
Exactly-once gate between the poll task and the delete-time settlement:
both run on the same event loop and neither awaits between reading and
setting the flag, so whichever claims first owns billing or release.
"""
if _is_settled(logging_obj):
return False
logging_obj.model_call_details[_SETTLED_KEY] = True # rebind-ok: both settlers must see the same settlement flag
return True
async def poll_and_log_background_interaction_cost(
context: BackgroundInteractionPollContext,
fetch_interaction: FetchInteraction = _fetch_interaction,
) -> None:
last_seen_status: str | None = None
for interval in _poll_intervals(
initial=context.initial_interval_seconds,
maximum=context.max_interval_seconds,
timeout=context.timeout_seconds,
):
await asyncio.sleep(interval)
if _is_settled(context.logging_obj):
return
try:
response = await fetch_interaction(context)
except Exception as e: # noqa: BLE001 # any fetch error must not kill the billing poll loop
verbose_logger.debug(
"Background interaction cost poll for %s failed, will retry: %s",
context.interaction_id,
e,
)
continue
last_seen_status = response.status
if response.status not in _TERMINAL_STATUSES:
continue
if not _claim_settlement(context.logging_obj):
return
if response.usage is not None:
await _bill_settled_interaction(logging_obj=context.logging_obj, response=response)
else:
await _release_open_budget_reservation(logging_obj=context.logging_obj)
return
if not _claim_settlement(context.logging_obj):
return
if last_seen_status is not None and last_seen_status not in _POLLABLE_STATUSES:
verbose_logger.error(
"Gave up cost polling for background interaction %s after %ss: its last status %r is in neither "
"the pollable nor the terminal set, so this proxy never learned how to settle it and its usage "
"will not be tracked",
context.interaction_id,
context.timeout_seconds,
last_seen_status,
)
else:
verbose_logger.warning(
"Gave up cost polling for background interaction %s after %ss; its usage will not be tracked",
context.interaction_id,
context.timeout_seconds,
)
await _release_open_budget_reservation(logging_obj=context.logging_obj)
async def _release_open_budget_reservation(logging_obj: "LiteLLMLoggingObj") -> None:
"""
The proxy keeps the pre-call budget reservation open for an in-progress
background interaction so concurrent creates cannot stack past the budget.
The completion success event reconciles it to the actual cost; when the
interaction terminates without billable usage (or polling gives up, or it
is deleted before settling), no such event fires, so whoever claims the
settlement must release the reservation here or the spend counters stay
pinned at the estimated cost.
"""
metadata = get_litellm_metadata_from_kwargs(kwargs=logging_obj.model_call_details)
budget_reservation = metadata.get("user_api_key_budget_reservation")
if not isinstance(budget_reservation, dict):
return
from litellm.proxy.spend_tracking.budget_reservation import release_budget_reservation
try:
await release_budget_reservation(budget_reservation=budget_reservation)
except Exception: # noqa: BLE001 # a failed release must not crash the poll task; counters expire via TTL
verbose_logger.exception("Failed to release budget reservation for an unbilled background interaction")
async def _bill_settled_interaction(logging_obj: "LiteLLMLoggingObj", response: InteractionsAPIResponse) -> None:
"""
Claiming the settlement makes the claimer solely responsible for the
reservation, and no one retries a claim that is already set. A billing
failure here must therefore release the reservation on its way out, or it
stays pinned at the estimated cost until the whole poll times out.
"""
try:
await logging_obj.async_log_background_interaction_completion(result=response)
except Exception:
await _release_open_budget_reservation(logging_obj=logging_obj)
raise
def is_pollable_background_interaction(response: InteractionsAPIResponse) -> bool:
"""
The single gate deciding whether a create's response gets a poll task.
The proxy's success callback defers releasing the budget reservation for
exactly these responses, on the promise that a poll task will settle them,
so a response one site accepts and the other refuses strands its
reservation on the spend counters with nothing left to reconcile it.
``queued`` belongs here alongside ``in_progress``. It is the API's
not-started-yet state, so it reaches a terminal status the same way and
needs polling for the same reason: nothing else in the proxy ever bills a
create that came back without usage, so a status missing from both this
set and ``_TERMINAL_STATUSES`` is billed nowhere and alerts nobody.
"""
return response.status in _POLLABLE_STATUSES and bool(response.id)
def missing_usage_is_expected(response: InteractionsAPIResponse) -> bool:
"""
Whether a response arriving with no usage block is a normal outcome rather
than lost billing data. An interaction that is still running, or that
stopped at ``failed``, ``cancelled``, ``incomplete`` or ``budget_exceeded``,
has nothing to charge for and should not raise a cost-tracking alarm.
``completed`` and ``requires_action`` both mean the model produced output,
so a usage block is always expected with them. If one arrives without it
the charge for real work has been lost, which is precisely what the
proxy's cost-tracking alert exists to surface.
"""
return response.status not in _STATUSES_THAT_PRODUCED_OUTPUT
@dataclass(frozen=True, slots=True)
class _ActiveBackgroundPoll:
task: "asyncio.Task[None]"
context: BackgroundInteractionPollContext
_ACTIVE_POLLS: dict[str, _ActiveBackgroundPoll] = {} # mutable-ok: asyncio needs strong refs to running poll tasks
def _discard_poll(interaction_id: str, task: "asyncio.Task[None]") -> None:
entry = _ACTIVE_POLLS.get(interaction_id)
if entry is not None and entry.task is task:
del _ACTIVE_POLLS[interaction_id]
def maybe_schedule_background_interaction_cost_polling(
response: object,
create_kwargs: Mapping[str, object],
custom_llm_provider: str,
) -> "asyncio.Task[None] | None":
from litellm.litellm_core_utils.litellm_logging import Logging
if not BACKGROUND_INTERACTION_COST_POLLING_ENABLED:
return None
if not isinstance(response, InteractionsAPIResponse):
return None
if not is_pollable_background_interaction(response):
return None
logging_obj = create_kwargs.get("litellm_logging_obj")
if not isinstance(logging_obj, Logging):
return None
try:
asyncio.get_running_loop()
except RuntimeError:
return None
api_key = create_kwargs.get("api_key")
api_base = create_kwargs.get("api_base")
context = BackgroundInteractionPollContext(
interaction_id=response.id,
custom_llm_provider=custom_llm_provider,
logging_obj=logging_obj,
api_key=api_key if isinstance(api_key, str) else None,
api_base=api_base if isinstance(api_base, str) else None,
)
task = asyncio.create_task(poll_and_log_background_interaction_cost(context))
_ACTIVE_POLLS[context.interaction_id] = _ActiveBackgroundPoll(task=task, context=context)
task.add_done_callback(
lambda finished, interaction_id=context.interaction_id: _discard_poll(interaction_id, finished)
)
return task
async def maybe_settle_background_interaction_before_delete(
interaction_id: str,
fetch_interaction: FetchInteraction = _fetch_interaction,
) -> None:
entry = _ACTIVE_POLLS.get(interaction_id)
if entry is None:
return
context = entry.context
try:
response = await fetch_interaction(context)
except Exception as e: # noqa: BLE001 # unfetchable pre-delete state settles by releasing the reservation
verbose_logger.debug(
"Could not fetch background interaction %s before delete, releasing its reservation: %s",
interaction_id,
e,
)
if _claim_settlement(context.logging_obj):
await _release_open_budget_reservation(logging_obj=context.logging_obj)
return
if not _claim_settlement(context.logging_obj):
return
if response.status in _TERMINAL_STATUSES and response.usage is not None:
await _bill_settled_interaction(logging_obj=context.logging_obj, response=response)
return
await _release_open_budget_reservation(logging_obj=context.logging_obj)

View file

@ -40,6 +40,10 @@ from typing import Any, Final
import httpx
import litellm
from litellm.interactions.background_cost_polling import (
maybe_schedule_background_interaction_cost_polling,
maybe_settle_background_interaction_before_delete,
)
from litellm.interactions.http_handler import interactions_http_handler
from litellm.interactions.utils import (
InteractionsAPIRequestUtils,
@ -171,6 +175,12 @@ async def acreate(
else:
response = init_response
maybe_schedule_background_interaction_cost_polling(
response=response,
create_kwargs=kwargs,
custom_llm_provider=custom_llm_provider,
)
return response
except Exception as e:
raise litellm.exception_type(
@ -462,6 +472,8 @@ async def adelete(
loop: Final = asyncio.get_event_loop()
kwargs["adelete_interaction"] = True
await maybe_settle_background_interaction_before_delete(interaction_id=interaction_id)
func: Final = partial(
delete,
interaction_id=interaction_id,

View file

@ -71,6 +71,9 @@ from litellm.litellm_core_utils.llm_cost_calc.guardrail_cost import (
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
StandardBuiltInToolCostTracking,
)
from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import (
InteractionsUsageObjectTransformation,
)
from litellm.litellm_core_utils.logging_utils import truncate_base64_in_messages
from litellm.litellm_core_utils.model_param_helper import ModelParamHelper
from litellm.litellm_core_utils.redact_messages import (
@ -83,6 +86,10 @@ from litellm.llms.base_llm.search.transformation import SearchResponse
from litellm.responses.utils import ResponseAPILoggingUtils
from litellm.types.agents import LiteLLMSendMessageResponse
from litellm.types.containers.main import ContainerObject
from litellm.types.interactions import (
InteractionsAPIResponse,
InteractionsAPIStreamingResponse,
)
from litellm.types.llms.openai import (
AllMessageValues,
Batch,
@ -2145,6 +2152,11 @@ class Logging(LiteLLMLoggingBaseClass):
or isinstance(logging_result, OpenAIModerationResponse)
or isinstance(logging_result, OCRResponse) # OCR
or isinstance(logging_result, SearchResponse) # Search API
or (
isinstance(logging_result, InteractionsAPIResponse)
and logging_result.usage is not None
and self._is_interactions_create_call_type()
)
or isinstance(logging_result, dict)
and logging_result.get("object") == "vector_store.search_results.page"
or isinstance(logging_result, dict)
@ -2157,6 +2169,87 @@ class Logging(LiteLLMLoggingBaseClass):
return True
return False
def _is_interactions_create_call_type(self) -> bool:
"""
Only interaction creation is billable. GET polls, deletes, and cancels
also return an ``InteractionsAPIResponse`` (with usage once completed),
so recognizing those would write spend on every poll of a background
interaction. The proxy sets ``call_type`` from its route_type
(``create_interaction``/``acreate_interaction``); the SDK sets it from
the decorated function name (``create``/``acreate``).
Recognition additionally requires a usage block (checked at the call
site): a ``background=true`` create returns ``in_progress`` without
usage, and billing it would write a $0 spend log under the interaction
id that collides with the row the background poll task writes once the
interaction completes (see
``litellm.interactions.background_cost_polling``).
"""
return self.call_type in (
CallTypes.create_interaction.value,
CallTypes.acreate_interaction.value,
"create",
"acreate",
)
async def async_log_background_interaction_completion(
self,
result: InteractionsAPIResponse,
) -> None:
"""
Log the terminal result of a background interaction as a fresh success
event. The create request already ran success logging for its
``in_progress`` response (no usage, so no cost was tracked); clearing
the dedup flags lets the completed result flow through cost calculation
and spend tracking exactly once, spanning create to completion.
The poll fetched this body through its own client call, which priced it
against a throwaway logging object holding none of this request's
deployment context: no ``model_info``, no router ``model_id``, no
deployment ``litellm_params``. Keeping that price would bill a
custom-priced deployment at the wrong rate, and it would also satisfy
the "already calculated" shortcut and skip repricing here, leaving the
cost breakdown at the zeros the usage-less create stamped and writing
those zeros to the spend log. Dropping it makes this event price the
settled body itself, against the deployment that served the create.
The same throwaway call stamped the deployment identity that travels
with the price, so ``model_id`` and ``litellm_model_name`` go with it.
Left in place they overwrite the create's real deployment with the
poll's empty one in the payload every logging integration reads.
"""
settled_hidden_params: Final = getattr(result, "_hidden_params", None)
if isinstance(settled_hidden_params, dict):
for poll_scoped_key in ("response_cost", "model_id", "litellm_model_name"):
settled_hidden_params.pop(poll_scoped_key, None)
self._reset_success_emission_dedupe()
await self.async_success_handler(result=result)
def _reset_success_emission_dedupe(self) -> None:
"""
Success callbacks dedupe per request, because the sync and async
handlers both fire on some paths and would otherwise report one call
twice. A settled background interaction is a genuinely second success
event on the same request, so every such marker has to be cleared or
the completion, the only event that carries usage and cost, is
discarded as a duplicate of the in-progress create.
"""
self.model_call_details.pop("has_logged_async_success", None)
litellm_params = self.model_call_details.get("litellm_params")
if not isinstance(litellm_params, dict):
return
metadata = litellm_params.get("metadata")
if not isinstance(metadata, dict):
return
otel_internal = metadata.get("_otel_internal")
if not isinstance(otel_internal, dict):
return
spans_logged = otel_internal.get("spans_logged")
if not isinstance(spans_logged, dict):
return
for scope in [key for key in spans_logged if isinstance(key, tuple) and key[-1:] == ("success",)]:
del spans_logged[scope]
def _flush_passthrough_collected_chunks_helper(
self,
raw_bytes: list[bytes],
@ -2282,7 +2375,9 @@ class Logging(LiteLLMLoggingBaseClass):
is_sync_request: Final = self._is_sync_litellm_request(litellm_params)
try:
## BUILD COMPLETE STREAMED RESPONSE
complete_streaming_response: ModelResponse | TextCompletionResponse | ResponsesAPIResponse | None = None
complete_streaming_response: (
ModelResponse | TextCompletionResponse | ResponsesAPIResponse | InteractionsAPIResponse | None
) = None
if "complete_streaming_response" in self.model_call_details:
return # break out of this.
complete_streaming_response = self._get_assembled_streaming_response(
@ -2768,14 +2863,14 @@ class Logging(LiteLLMLoggingBaseClass):
## BUILD COMPLETE STREAMED RESPONSE
if "async_complete_streaming_response" in self.model_call_details:
return # break out of this.
complete_streaming_response: Final[ModelResponse | TextCompletionResponse | ResponsesAPIResponse | None] = (
self._get_assembled_streaming_response(
result=result,
start_time=start_time,
end_time=end_time,
is_async=True,
streaming_chunks=self.streaming_chunks,
)
complete_streaming_response: Final[
ModelResponse | TextCompletionResponse | ResponsesAPIResponse | InteractionsAPIResponse | None
] = self._get_assembled_streaming_response(
result=result,
start_time=start_time,
end_time=end_time,
is_async=True,
streaming_chunks=self.streaming_chunks,
)
if complete_streaming_response is not None:
@ -3558,7 +3653,7 @@ class Logging(LiteLLMLoggingBaseClass):
end_time: datetime.datetime,
is_async: bool,
streaming_chunks: list[object],
) -> ModelResponse | TextCompletionResponse | ResponsesAPIResponse | None:
) -> ModelResponse | TextCompletionResponse | ResponsesAPIResponse | InteractionsAPIResponse | None:
if self.stream is not True:
return None
if isinstance(result, ModelResponse) or isinstance(result, TextCompletionResponse):
@ -3583,9 +3678,40 @@ class Logging(LiteLLMLoggingBaseClass):
),
)
return result.response
elif isinstance(result, InteractionsAPIStreamingResponse):
return self._assemble_completed_interaction_response(result)
else:
return None
@staticmethod
def _assemble_completed_interaction_response(
result: InteractionsAPIStreamingResponse,
) -> InteractionsAPIResponse | None:
"""
The Interactions API streaming iterator hands the terminal event to the
success handlers: the new schema (Api-Revision: 2026-05-20) emits
``interaction.completed`` carrying the full interaction object, the
legacy schema (2026-05-07) emits a chunk with ``status="completed"``
and usage on the chunk itself. Build the equivalent non-streaming
response so cost calculation and spend tracking see one shape.
"""
if result.event_type == "interaction.completed" and result.interaction is not None:
return InteractionsAPIResponse(**result.interaction)
if result.status == "completed":
return InteractionsAPIResponse(
**result.model_dump(
exclude={ # mutable-ok: pydantic types exclude as set[str], which a frozenset does not satisfy
"event_type",
"delta",
"index",
"step",
"interaction_id",
"interaction",
}
)
)
return None
def _handle_anthropic_messages_response_logging(self, result: Any) -> ModelResponse:
"""
Handles logging for Anthropic messages responses.
@ -5092,6 +5218,8 @@ class StandardLoggingPayloadSetup:
elif isinstance(usage, dict):
if ResponseAPILoggingUtils._is_response_api_usage(usage):
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(usage)
if InteractionsUsageObjectTransformation.is_interactions_usage_object(usage):
return InteractionsUsageObjectTransformation.transform_interactions_usage_object(usage)
return Usage(**usage)
raise ValueError(f"usage is required, got={usage} of type {type(usage)}")
@ -5118,6 +5246,8 @@ class StandardLoggingPayloadSetup:
if isinstance(_raw, dict):
if ResponseAPILoggingUtils._is_response_api_usage(_raw):
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_raw).model_dump()
if InteractionsUsageObjectTransformation.is_interactions_usage_object(_raw):
return InteractionsUsageObjectTransformation.transform_interactions_usage_object(_raw).model_dump()
return _raw
if isinstance(_raw, Usage):
return _raw.model_dump()

View file

@ -1,6 +1,9 @@
from collections.abc import Mapping, Sequence
from types import MappingProxyType
from typing import Any
from litellm.types.utils import (
CompletionTokensDetailsWrapper,
PromptTokensDetailsWrapper,
TranscriptionUsageDurationObject,
TranscriptionUsageTokensObject,
@ -34,3 +37,127 @@ class TranscriptionUsageObjectTransformation:
),
)
return None
_INTERACTIONS_MODALITY_FIELDS: Mapping[str, str] = MappingProxyType(
{
"text": "text_tokens",
"audio": "audio_tokens",
"image": "image_tokens",
"video": "video_tokens",
"document": "text_tokens",
}
)
def _modality_field(entry: Mapping[str, Any]) -> str | None:
return _INTERACTIONS_MODALITY_FIELDS.get(str(entry.get("modality", "")).lower())
def _token_count(value: object) -> int:
return value if isinstance(value, int) else 0
def _modality_token_sums(entries: Sequence[Mapping[str, Any]]) -> Mapping[str, int]:
fields = frozenset(field for entry in entries if (field := _modality_field(entry)) is not None)
return MappingProxyType(
{
field: sum(_token_count(entry.get("tokens")) for entry in entries if _modality_field(entry) == field)
for field in fields
}
)
def _google_search_query_count(usage_object: Mapping[str, Any]) -> int:
return sum(
_token_count(entry.get("count"))
for entry in tuple(usage_object.get("grounding_tool_count") or ())
if isinstance(entry, Mapping) and entry.get("type") == "google_search" # pyright: ignore[reportUnnecessaryIsInstance] # provider JSON, not the empty tuple inferred from `or ()`
)
def _subtract_cached_from_input(
input_sums: Mapping[str, int],
cached_sums: Mapping[str, int],
total_cached_tokens: int,
) -> Mapping[str, int]:
if cached_sums:
return MappingProxyType(
{field: max(0, tokens - cached_sums.get(field, 0)) for field, tokens in input_sums.items()}
)
if total_cached_tokens and "text_tokens" in input_sums:
return MappingProxyType(
{
**input_sums,
"text_tokens": max(0, input_sums["text_tokens"] - total_cached_tokens),
}
)
return input_sums
class InteractionsUsageObjectTransformation:
"""
Maps the Google Interactions API usage block (total_input_tokens,
output_tokens_by_modality, ...) into LiteLLM's chat-format ``Usage`` so the
generic cost calculator and spend tracking can bill it.
"""
@staticmethod
def is_interactions_usage_object(usage_object: object) -> bool:
if not isinstance(usage_object, dict):
return False
if "prompt_tokens" in usage_object or "input_tokens" in usage_object:
return False
return "total_input_tokens" in usage_object or "total_output_tokens" in usage_object
@staticmethod
def transform_interactions_usage_object(usage_object: Mapping[str, Any]) -> Usage:
input_entries = tuple(usage_object.get("input_tokens_by_modality") or ()) + tuple(
usage_object.get("tool_use_tokens_by_modality") or ()
)
cached_sums = _modality_token_sums(tuple(usage_object.get("cached_tokens_by_modality") or ()))
output_sums = _modality_token_sums(tuple(usage_object.get("output_tokens_by_modality") or ()))
total_cached_tokens = _token_count(usage_object.get("total_cached_tokens"))
input_sums = _subtract_cached_from_input(
input_sums=_modality_token_sums(input_entries),
cached_sums=cached_sums,
total_cached_tokens=total_cached_tokens,
)
reasoning_tokens = _token_count(usage_object.get("total_reasoning_tokens")) or _token_count(
usage_object.get("total_thought_tokens")
)
prompt_tokens = _token_count(usage_object.get("total_input_tokens")) + _token_count(
usage_object.get("total_tool_use_tokens")
)
completion_tokens = _token_count(usage_object.get("total_output_tokens")) + reasoning_tokens
total_tokens = _token_count(usage_object.get("total_tokens")) or (prompt_tokens + completion_tokens)
web_search_requests = _google_search_query_count(usage_object)
prompt_tokens_details = (
PromptTokensDetailsWrapper(
cached_tokens=total_cached_tokens or None,
web_search_requests=web_search_requests or None,
**input_sums,
)
if input_sums or total_cached_tokens or web_search_requests
else None
)
completion_tokens_details = (
CompletionTokensDetailsWrapper(
reasoning_tokens=reasoning_tokens or None,
**output_sums,
)
if output_sums or reasoning_tokens
else None
)
return Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=total_tokens,
prompt_tokens_details=prompt_tokens_details,
completion_tokens_details=completion_tokens_details,
cache_read_input_tokens=total_cached_tokens or None,
)

View file

@ -28,8 +28,12 @@ from litellm.types.llms.openai import (
ChatCompletionAssistantMessage,
ChatCompletionFileObject,
ChatCompletionImageObject,
ChatCompletionReasoningItem,
ChatCompletionReasoningSummaryTextBlock,
ChatCompletionRedactedThinkingBlock,
ChatCompletionResponseMessage,
ChatCompletionTextObject,
ChatCompletionThinkingBlock,
ChatCompletionToolParam,
ChatCompletionUserMessage,
)
@ -1549,6 +1553,44 @@ def _extract_reasoning_content(message: dict) -> tuple[str | None, str | None]:
return None, message_content
def _readable_thinking_text(
block: ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock,
) -> str:
"""The text a chat model can read back, empty for redacted blocks and malformed ones."""
if block.get("type") != "thinking":
return ""
thinking: Final = cast(ChatCompletionThinkingBlock, block).get("thinking") # cast-ok: narrowed by the type tag
return str(thinking or "")
def reasoning_content_from_thinking_blocks(
thinking_blocks: Iterable[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock],
) -> str:
"""Flatten Anthropic thinking blocks into the `reasoning_content` string chat models expect.
Redacted blocks carry no readable text, so they contribute nothing.
"""
return "\n".join(text for block in thinking_blocks if (text := _readable_thinking_text(block)))
def responses_reasoning_item_from_thinking_blocks(
thinking_blocks: Iterable[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock],
) -> ChatCompletionReasoningItem | None:
"""Build a Responses API `reasoning` input item from Anthropic thinking blocks.
The item carries no `id`: the Responses API rejects an empty one and 404s on any id it
did not mint itself, while an item without an id is always accepted.
"""
summary: Final[list[ChatCompletionReasoningSummaryTextBlock]] = [ # mutable-ok: API message payload
ChatCompletionReasoningSummaryTextBlock(type="summary_text", text=text)
for block in thinking_blocks
if (text := _readable_thinking_text(block))
]
if not summary:
return None
return ChatCompletionReasoningItem(type="reasoning", summary=summary)
def _parse_content_for_reasoning(
message_text: str | None,
) -> tuple[str | None, str | None]:

View file

@ -64,6 +64,7 @@ from openai.types.chat.chat_completion_chunk import Choice as OpenAIStreamingCho
from litellm.litellm_core_utils.prompt_templates.common_utils import (
parse_tool_call_arguments,
reasoning_content_from_thinking_blocks,
with_prompt_cache_breakpoint,
)
from litellm.litellm_core_utils.prompt_templates.factory import (
@ -592,6 +593,9 @@ class LiteLLMAnthropicMessagesAdapter:
assistant_message["tool_calls"] = tool_calls
if len(thinking_blocks) > 0:
assistant_message["thinking_blocks"] = thinking_blocks
reasoning_content = reasoning_content_from_thinking_blocks(thinking_blocks)
if reasoning_content:
assistant_message["reasoning_content"] = reasoning_content
new_messages.append(assistant_message)
return new_messages

View file

@ -152,7 +152,10 @@ class AnthropicResponsesStreamWrapper:
if block_idx < 0:
if not delta:
return
block_idx = self._open_block(item_id, {"type": "thinking", "thinking": ""})
block_idx = self._open_block(
item_id,
{"type": "thinking", "thinking": "", "signature": ""}, # mutable-ok: API message payload
)
self._chunk_queue.append(
{
"type": "content_block_delta",

View file

@ -6,12 +6,14 @@ path used for OpenAI and Azure models.
"""
import json
from collections.abc import Iterable
from collections.abc import Iterable, Mapping
from itertools import groupby
from typing import Any, Final, cast
from litellm.litellm_core_utils.prompt_templates.common_utils import (
TOOL_RESULT_IMAGE_BOUNDARY,
TOOL_RESULT_IMAGE_PLACEHOLDER,
responses_reasoning_item_from_thinking_blocks,
with_prompt_cache_breakpoint,
)
from litellm.litellm_core_utils.reasoning_effort_utils import (
@ -36,7 +38,11 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
AnthropicUsage,
)
from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse
from litellm.types.llms.openai import (
ChatCompletionThinkingBlock,
ResponseAPIUsage,
ResponsesAPIResponse,
)
class LiteLLMAnthropicToResponsesAPIAdapter:
@ -100,6 +106,58 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
if isinstance(block, dict) and block.get("type") == "text" and (text := block.get("text")) # pyright: ignore[reportUnnecessaryIsInstance] # untrusted client payload
]
@staticmethod
def _summary_part_text(part: object) -> str:
if isinstance(part, Mapping):
mapping: Final = cast(Mapping[str, Any], part) # cast-ok: summary parts are untyped provider json
return str(mapping.get("text") or "")
return str(getattr(part, "text", None) or "")
@classmethod
def _thinking_blocks_from_reasoning_item(
cls,
summary: Iterable[object],
) -> tuple[dict[str, Any], ...]: # mutable-ok: API message payload
"""Anthropic thinking blocks for one Responses reasoning item.
The signature stays empty: only Anthropic can sign a thinking block, and a stand-in
value would be replayed as a real one and rejected by every backend that verifies it.
"""
return tuple(
AnthropicResponseContentBlockThinking(
type="thinking",
thinking=text,
signature=None,
).model_dump()
for part in summary
if (text := cls._summary_part_text(part))
)
@staticmethod
def _assistant_block_group_key(indexed_block: tuple[int, Mapping[str, Any]]) -> str:
"""Group a run of consecutive thinking blocks together; keep every other block alone."""
index, block = indexed_block
return "thinking" if block.get("type") == "thinking" else f"block:{index}"
@classmethod
def _assistant_group_to_input_item(
cls, group: tuple[Mapping[str, Any], ...]
) -> dict[str, Any] | None: # mutable-ok: API message payload
first: Final = group[0]
btype: Final = first.get("type")
if btype == "thinking":
blocks: Final = cast(tuple[ChatCompletionThinkingBlock, ...], group) # cast-ok: untrusted client payload
reasoning_item: Final = responses_reasoning_item_from_thinking_blocks(blocks)
return None if reasoning_item is None else dict(reasoning_item) # mutable-ok: API message payload
if btype == "tool_use":
return { # mutable-ok: API message payload
"type": "function_call",
"call_id": first.get("id", ""),
"name": first.get("name", ""),
"arguments": json.dumps(first.get("input", {})), # mutable-ok: API message payload
}
return None
def translate_messages_to_responses_input(
self,
messages: list[AllAnthropicPassThroughMessageValues],
@ -113,6 +171,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
user image -> message(role=user, input_image)
user tool_result -> function_call_output
assistant text -> message(role=assistant, output_text)
assistant thinking -> reasoning
assistant tool_use -> function_call
"""
input_items: Final[list[dict[str, Any]]] = []
@ -233,27 +292,17 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
}
)
elif isinstance(content, list):
asst_parts: list[dict[str, Any]] = []
for block in content:
if not isinstance(block, dict):
continue
btype = block.get("type")
if btype == "text":
asst_parts.append({"type": "output_text", "text": block.get("text", "")})
elif btype == "tool_use":
# tool_use becomes a top-level function_call item
input_items.append(
{
"type": "function_call",
"call_id": block.get("id", ""),
"name": block.get("name", ""),
"arguments": json.dumps(block.get("input", {})),
}
)
elif btype == "thinking":
thinking_text = block.get("thinking", "")
if thinking_text:
asst_parts.append({"type": "output_text", "text": thinking_text})
blocks = tuple(block for block in content if isinstance(block, dict))
input_items.extend(
item
for _, group in groupby(enumerate(blocks), key=self._assistant_block_group_key)
if (item := self._assistant_group_to_input_item(tuple(block for _, block in group))) is not None
)
asst_parts: list[dict[str, Any]] = [ # mutable-ok: API message payload
{"type": "output_text", "text": block.get("text", "")} # mutable-ok: API message payload
for block in blocks
if block.get("type") == "text"
]
if asst_parts:
input_items.append(
{
@ -514,16 +563,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
for item in response.output:
if isinstance(item, ResponseReasoningItem):
for summary in item.summary:
text = getattr(summary, "text", "")
if text:
content.append(
AnthropicResponseContentBlockThinking(
type="thinking",
thinking=text,
signature=None,
).model_dump()
)
content.extend(self._thinking_blocks_from_reasoning_item(item.summary))
elif isinstance(item, ResponseOutputMessage):
for part in item.content:
@ -555,6 +595,12 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
content.append(
AnthropicResponseContentBlockText(type="text", text=part.get("text", "")).model_dump()
)
elif item_type == "reasoning":
content.extend(
self._thinking_blocks_from_reasoning_item(
cast(Iterable[object], item.get("summary") or ()), # cast-ok: untyped provider json
)
)
elif item_type == "function_call":
try:
input_data = json.loads(item.get("arguments", "{}"))

View file

@ -30,7 +30,12 @@ class AzureFoundryErrorStrings(str, enum.Enum):
SET_EXTRA_PARAMETERS_TO_PASS_THROUGH = "Set extra-parameters to 'pass-through'"
NON_OPENAI_SPEC_MESSAGE_FIELDS: Final = ("thinking_blocks", "provider_specific_fields", "cache_control")
NON_OPENAI_SPEC_MESSAGE_FIELDS: Final = (
"thinking_blocks",
"reasoning_content",
"provider_specific_fields",
"cache_control",
)
class AzureAIStudioConfig(OpenAIConfig):
@ -173,7 +178,8 @@ class AzureAIStudioConfig(OpenAIConfig):
"""
- Azure AI Studio doesn't support content as a list. This handles:
1. Strips message fields that are not part of the OpenAI chat-completions
schema (thinking_blocks, provider_specific_fields, cache_control).
schema (thinking_blocks, reasoning_content, provider_specific_fields,
cache_control).
Azure AI Foundry backends set additionalProperties=false and reject
these with "Extra inputs are not permitted", which breaks multi-turn
Anthropic-format clients that echo thinking blocks back as history.

View file

@ -3,10 +3,31 @@ Helper util for handling databricks-specific cost calculation
- e.g.: handling 'dbrx-instruct-*'
"""
from types import MappingProxyType
from typing import Final
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
from litellm.types.utils import Usage
from litellm.utils import get_model_info
_LEGACY_ENDPOINT_NAMES: Final = MappingProxyType(
{
"dbrx-instruct": "databricks-dbrx-instruct",
"meta-llama-3.1-70b-instruct": "databricks-meta-llama-3-1-70b-instruct",
"meta-llama-3.1-405b-instruct": "databricks-meta-llama-3-1-405b-instruct",
"mixtral-8x7b-instruct-v0.1": "databricks-mixtral-8x7b-instruct",
"bge-large-en": "databricks-bge-large-en",
"gte-large-en": "databricks-gte-large-en",
"llama-2-70b-chat": "databricks-llama-2-70b-chat",
}
)
def _registry_key(model: str) -> str:
name: Final = model.removeprefix("databricks/")
return next(
(key for prefix, key in _LEGACY_ENDPOINT_NAMES.items() if name.startswith(prefix)),
name,
)
def cost_per_token(model: str, usage: Usage) -> tuple[float, float]:
@ -20,36 +41,8 @@ def cost_per_token(model: str, usage: Usage) -> tuple[float, float]:
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
"""
base_model = model
if model.startswith("databricks/dbrx-instruct") or model.startswith("dbrx-instruct"):
base_model = "databricks-dbrx-instruct"
elif model.startswith("databricks/meta-llama-3.1-70b-instruct") or model.startswith("meta-llama-3.1-70b-instruct"):
base_model = "databricks-meta-llama-3-1-70b-instruct"
elif model.startswith("databricks/meta-llama-3.1-405b-instruct") or model.startswith(
"meta-llama-3.1-405b-instruct"
):
base_model = "databricks-meta-llama-3-1-405b-instruct"
elif (
model.startswith("databricks/mixtral-8x7b-instruct-v0.1")
or model.startswith("mixtral-8x7b-instruct-v0.1")
or model.startswith("databricks/mixtral-8x7b-instruct-v0.1")
or model.startswith("mixtral-8x7b-instruct-v0.1")
):
base_model = "databricks-mixtral-8x7b-instruct"
elif model.startswith("databricks/bge-large-en") or model.startswith("bge-large-en"):
base_model = "databricks-bge-large-en"
elif model.startswith("databricks/gte-large-en") or model.startswith("gte-large-en"):
base_model = "databricks-gte-large-en"
elif model.startswith("databricks/llama-2-70b-chat") or model.startswith("llama-2-70b-chat"):
base_model = "databricks-llama-2-70b-chat"
## GET MODEL INFO
model_info: Final = get_model_info(model=base_model, custom_llm_provider="databricks")
## CALCULATE INPUT COST
prompt_cost: Final[float] = usage["prompt_tokens"] * model_info["input_cost_per_token"]
## CALCULATE OUTPUT COST
completion_cost: Final = usage["completion_tokens"] * model_info["output_cost_per_token"]
return prompt_cost, completion_cost
return generic_cost_per_token(
model=_registry_key(model),
usage=usage,
custom_llm_provider="databricks",
)

View file

@ -504,6 +504,7 @@ class FireworksAIConfig(FireworksAIMixin, OpenAIGPTConfig):
m = cast(dict, message)
m.pop("provider_specific_fields", None)
m.pop("thinking_blocks", None)
m.pop("reasoning_content", None)
return messages

View file

@ -164,12 +164,13 @@ class HostedVLLMChatConfig(OpenAIGPTConfig):
"""
Support translating:
- video files from file_id or file_data to video_url
- thinking_blocks on assistant messages are removed, and content lists
are converted to strings for vLLM compatibility
- thinking_blocks and reasoning_content on assistant messages are removed,
and content lists are converted to strings for vLLM compatibility
"""
for message in messages:
if message["role"] == "assistant":
message.pop("thinking_blocks", None)
message.pop("reasoning_content", None)
existing_content = message.get("content")
if isinstance(existing_content, list):
text_parts = []

View file

@ -14551,6 +14551,8 @@
]
},
"databricks/databricks-bge-large-en": {
"cache_creation_input_token_cost": 1.0003e-07,
"cache_read_input_token_cost": 1.0003e-07,
"input_cost_per_token": 1.0003e-07,
"input_dbu_cost_per_token": 1.429e-06,
"litellm_provider": "databricks",
@ -14566,6 +14568,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-claude-3-7-sonnet": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14581,10 +14585,41 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-fable-5": {
"cache_creation_input_token_cost": 1.250004e-05,
"cache_read_input_token_cost": 1.00002e-06,
"input_cost_per_token": 1.000006e-05,
"input_dbu_cost_per_token": 0.000142858,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
},
"mode": "chat",
"output_cost_per_token": 5.000002e-05,
"output_dbu_cost_per_token": 0.000714286,
"prompt_cache_min_tokens": 512,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_mid_conversation_system": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": false,
"thinking_always_on": true
},
"databricks/databricks-claude-haiku-4-5": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.0003e-07,
"input_cost_per_token": 1.00002e-06,
"input_dbu_cost_per_token": 1.4286e-05,
"litellm_provider": "databricks",
@ -14600,10 +14635,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-opus-4": {
"cache_creation_input_token_cost": 1.874999e-05,
"cache_read_input_token_cost": 1.50003e-06,
"input_cost_per_token": 1.5000020000000002e-05,
"input_dbu_cost_per_token": 0.000214286,
"litellm_provider": "databricks",
@ -14619,10 +14657,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-opus-4-1": {
"cache_creation_input_token_cost": 1.874999e-05,
"cache_read_input_token_cost": 1.50003e-06,
"input_cost_per_token": 1.5000020000000002e-05,
"input_dbu_cost_per_token": 0.000214286,
"litellm_provider": "databricks",
@ -14638,10 +14679,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-opus-4-5": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
@ -14657,11 +14701,14 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"supports_output_config": true
},
"databricks/databricks-claude-opus-4-6": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
@ -14677,10 +14724,93 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
},
"mode": "chat",
"output_cost_per_token": 2.500001e-05,
"output_dbu_cost_per_token": 0.000357143,
"prompt_cache_min_tokens": 2048,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": true
},
"databricks/databricks-claude-opus-4-8": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
},
"mode": "chat",
"output_cost_per_token": 2.500001e-05,
"output_dbu_cost_per_token": 0.000357143,
"prompt_cache_min_tokens": 1024,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_mid_conversation_system": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": true
},
"databricks/databricks-claude-opus-5": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
},
"mode": "chat",
"output_cost_per_token": 2.500001e-05,
"output_dbu_cost_per_token": 0.000357143,
"prompt_cache_min_tokens": 512,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_mid_conversation_system": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": true
},
"databricks/databricks-claude-sonnet-4": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14696,10 +14826,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-4-1": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14715,10 +14848,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-4-5": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14734,10 +14870,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14753,10 +14892,40 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-5": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.99999e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields. Introductory launch rates of 28.571 input / 142.857 output / 35.714 cache write / 2.857 cache read DBU run through 2026-08-31; the standard rates are listed here because entries carry no expiry date."
},
"mode": "chat",
"output_cost_per_token": 1.500002e-05,
"output_dbu_cost_per_token": 0.000214286,
"prompt_cache_min_tokens": 1024,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_mid_conversation_system": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": true
},
"databricks/databricks-gemini-2-5-flash": {
"cache_creation_input_token_cost": 3.0002e-07,
"cache_read_input_token_cost": 3.0002e-08,
"input_cost_per_token": 3.0001999999999996e-07,
"input_dbu_cost_per_token": 4.285999999999999e-06,
"litellm_provider": "databricks",
@ -14771,9 +14940,12 @@
"output_dbu_cost_per_token": 3.5714e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-2-5-pro": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.24999e-07,
"input_cost_per_token": 1.24999e-06,
"input_dbu_cost_per_token": 1.7857e-05,
"litellm_provider": "databricks",
@ -14788,9 +14960,12 @@
"output_dbu_cost_per_token": 0.000142857,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-3-1-flash-lite": {
"cache_creation_input_token_cost": 3.1248e-07,
"cache_read_input_token_cost": 3.122e-08,
"input_cost_per_token": 3.1248e-07,
"input_dbu_cost_per_token": 4.464e-06,
"litellm_provider": "databricks",
@ -14805,9 +14980,12 @@
"output_dbu_cost_per_token": 2.6786e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-3-1-pro": {
"cache_creation_input_token_cost": 2.49998e-06,
"cache_read_input_token_cost": 2.4997e-07,
"input_cost_per_token": 2.49998e-06,
"input_dbu_cost_per_token": 3.5714e-05,
"litellm_provider": "databricks",
@ -14822,9 +15000,12 @@
"output_dbu_cost_per_token": 0.000214286,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-3-flash": {
"cache_creation_input_token_cost": 6.2503e-07,
"cache_read_input_token_cost": 6.251e-08,
"input_cost_per_token": 6.2503e-07,
"input_dbu_cost_per_token": 8.929e-06,
"litellm_provider": "databricks",
@ -14839,9 +15020,12 @@
"output_dbu_cost_per_token": 5.3571e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-3-pro": {
"cache_creation_input_token_cost": 2.49998e-06,
"cache_read_input_token_cost": 2.4997e-07,
"input_cost_per_token": 2.49998e-06,
"input_dbu_cost_per_token": 3.5714e-05,
"litellm_provider": "databricks",
@ -14856,9 +15040,12 @@
"output_dbu_cost_per_token": 0.000214286,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemma-3-12b": {
"cache_creation_input_token_cost": 1.5001e-07,
"cache_read_input_token_cost": 1.5001e-07,
"input_cost_per_token": 1.5000999999999998e-07,
"input_dbu_cost_per_token": 2.1429999999999996e-06,
"litellm_provider": "databricks",
@ -14874,6 +15061,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-gpt-5": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.2502e-07,
"input_cost_per_token": 1.24999e-06,
"input_dbu_cost_per_token": 1.7857e-05,
"litellm_provider": "databricks",
@ -14886,9 +15075,12 @@
"mode": "chat",
"output_cost_per_token": 9.999990000000002e-06,
"output_dbu_cost_per_token": 0.000142857,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-1": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.2502e-07,
"input_cost_per_token": 1.24999e-06,
"input_dbu_cost_per_token": 1.7857e-05,
"litellm_provider": "databricks",
@ -14901,9 +15093,12 @@
"mode": "chat",
"output_cost_per_token": 9.999990000000002e-06,
"output_dbu_cost_per_token": 0.000142857,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-1-codex-max": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.2502e-07,
"input_cost_per_token": 1.24999e-06,
"input_dbu_cost_per_token": 1.7857e-05,
"litellm_provider": "databricks",
@ -14916,9 +15111,12 @@
"mode": "chat",
"output_cost_per_token": 9.999990000000002e-06,
"output_dbu_cost_per_token": 0.000142857,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-1-codex-mini": {
"cache_creation_input_token_cost": 2.4997e-07,
"cache_read_input_token_cost": 2.499e-08,
"input_cost_per_token": 2.4997e-07,
"input_dbu_cost_per_token": 3.571e-06,
"litellm_provider": "databricks",
@ -14931,9 +15129,12 @@
"mode": "chat",
"output_cost_per_token": 1.99997e-06,
"output_dbu_cost_per_token": 2.8571e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-2": {
"cache_creation_input_token_cost": 1.75e-06,
"cache_read_input_token_cost": 1.75e-07,
"input_cost_per_token": 1.75e-06,
"input_dbu_cost_per_token": 2.5e-05,
"litellm_provider": "databricks",
@ -14946,9 +15147,12 @@
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_dbu_cost_per_token": 0.0002,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-2-codex": {
"cache_creation_input_token_cost": 1.75e-06,
"cache_read_input_token_cost": 1.75e-07,
"input_cost_per_token": 1.75e-06,
"input_dbu_cost_per_token": 2.5e-05,
"litellm_provider": "databricks",
@ -14961,9 +15165,12 @@
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_dbu_cost_per_token": 0.0002,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-3-codex": {
"cache_creation_input_token_cost": 1.75e-06,
"cache_read_input_token_cost": 1.75e-07,
"input_cost_per_token": 1.75e-06,
"input_dbu_cost_per_token": 2.5e-05,
"litellm_provider": "databricks",
@ -14976,9 +15183,12 @@
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_dbu_cost_per_token": 0.0002,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-4": {
"cache_creation_input_token_cost": 2.49998e-06,
"cache_read_input_token_cost": 2.4997e-07,
"input_cost_per_token": 2.49998e-06,
"input_dbu_cost_per_token": 3.5714e-05,
"litellm_provider": "databricks",
@ -14991,9 +15201,12 @@
"mode": "chat",
"output_cost_per_token": 1.5000020000000002e-05,
"output_dbu_cost_per_token": 0.000214286,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-4-mini": {
"cache_creation_input_token_cost": 7.4998e-07,
"cache_read_input_token_cost": 7.497e-08,
"input_cost_per_token": 7.4998e-07,
"input_dbu_cost_per_token": 1.0714e-05,
"litellm_provider": "databricks",
@ -15006,9 +15219,12 @@
"mode": "chat",
"output_cost_per_token": 4.50002e-06,
"output_dbu_cost_per_token": 6.4286e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-4-nano": {
"cache_creation_input_token_cost": 1.9999e-07,
"cache_read_input_token_cost": 2.002e-08,
"input_cost_per_token": 1.9999e-07,
"input_dbu_cost_per_token": 2.857e-06,
"litellm_provider": "databricks",
@ -15021,9 +15237,12 @@
"mode": "chat",
"output_cost_per_token": 1.24999e-06,
"output_dbu_cost_per_token": 1.7857e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-mini": {
"cache_creation_input_token_cost": 2.4997e-07,
"cache_read_input_token_cost": 2.499e-08,
"input_cost_per_token": 2.4997000000000006e-07,
"input_dbu_cost_per_token": 3.571e-06,
"litellm_provider": "databricks",
@ -15036,9 +15255,12 @@
"mode": "chat",
"output_cost_per_token": 1.9999700000000004e-06,
"output_dbu_cost_per_token": 2.8571e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-nano": {
"cache_creation_input_token_cost": 4.998e-08,
"cache_read_input_token_cost": 4.97e-09,
"input_cost_per_token": 4.998e-08,
"input_dbu_cost_per_token": 7.14e-07,
"litellm_provider": "databricks",
@ -15051,9 +15273,12 @@
"mode": "chat",
"output_cost_per_token": 3.9998000000000007e-07,
"output_dbu_cost_per_token": 5.714000000000001e-06,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-oss-120b": {
"cache_creation_input_token_cost": 1.5001e-07,
"cache_read_input_token_cost": 1.5001e-07,
"input_cost_per_token": 1.5000999999999998e-07,
"input_dbu_cost_per_token": 2.1429999999999996e-06,
"litellm_provider": "databricks",
@ -15069,6 +15294,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-gpt-oss-20b": {
"cache_creation_input_token_cost": 7e-08,
"cache_read_input_token_cost": 7e-08,
"input_cost_per_token": 7e-08,
"input_dbu_cost_per_token": 1e-06,
"litellm_provider": "databricks",
@ -15084,6 +15311,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-gte-large-en": {
"cache_creation_input_token_cost": 1.2999e-07,
"cache_read_input_token_cost": 1.2999e-07,
"input_cost_per_token": 1.2999000000000001e-07,
"input_dbu_cost_per_token": 1.857e-06,
"litellm_provider": "databricks",
@ -15099,6 +15328,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-llama-2-70b-chat": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",
@ -15115,6 +15346,8 @@
"supports_tool_choice": true
},
"databricks/databricks-llama-4-maverick": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",
@ -15131,6 +15364,8 @@
"supports_tool_choice": true
},
"databricks/databricks-meta-llama-3-1-405b-instruct": {
"cache_creation_input_token_cost": 5.00003e-06,
"cache_read_input_token_cost": 5.00003e-06,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
@ -15147,6 +15382,8 @@
"supports_tool_choice": true
},
"databricks/databricks-meta-llama-3-1-8b-instruct": {
"cache_creation_input_token_cost": 1.5001e-07,
"cache_read_input_token_cost": 1.5001e-07,
"input_cost_per_token": 1.5000999999999998e-07,
"input_dbu_cost_per_token": 2.1429999999999996e-06,
"litellm_provider": "databricks",
@ -15162,6 +15399,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-meta-llama-3-3-70b-instruct": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",
@ -15178,6 +15417,8 @@
"supports_tool_choice": true
},
"databricks/databricks-meta-llama-3-70b-instruct": {
"cache_creation_input_token_cost": 1.00002e-06,
"cache_read_input_token_cost": 1.00002e-06,
"input_cost_per_token": 1.00002e-06,
"input_dbu_cost_per_token": 1.4286e-05,
"litellm_provider": "databricks",
@ -15194,6 +15435,8 @@
"supports_tool_choice": true
},
"databricks/databricks-mixtral-8x7b-instruct": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",
@ -15210,6 +15453,8 @@
"supports_tool_choice": true
},
"databricks/databricks-mpt-30b-instruct": {
"cache_creation_input_token_cost": 1.00002e-06,
"cache_read_input_token_cost": 1.00002e-06,
"input_cost_per_token": 1.00002e-06,
"input_dbu_cost_per_token": 1.4286e-05,
"litellm_provider": "databricks",
@ -15226,6 +15471,8 @@
"supports_tool_choice": true
},
"databricks/databricks-mpt-7b-instruct": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",

View file

@ -6,21 +6,21 @@ Plugins are stored as metadata + git source references in LiteLLM database.
Actual plugin files are hosted on GitHub/GitLab/Bitbucket.
Endpoints:
/claude-code/marketplace.json - GET - List plugins for Claude Code discovery
/claude-code/plugins - POST - Register a new plugin (create-only)
/claude-code/plugins - GET - List plugins (admin)
/claude-code/plugins/{name} - GET - Get plugin details
/claude-code/plugins/{name} - PUT - Update an existing plugin
/claude-code/plugins/{name}/enable - POST - Enable a plugin
/claude-code/plugins/{name}/disable - POST - Disable a plugin
/claude-code/plugins/{name} - DELETE - Delete a plugin
/claude-code/marketplace.json - GET - List plugins for Claude Code discovery (unauthenticated)
/claude-code/plugins - POST - Register a new plugin (create-only, proxy admin only)
/claude-code/plugins - GET - List plugins (any authenticated key)
/claude-code/plugins/{name} - GET - Get plugin details (any authenticated key)
/claude-code/plugins/{name} - PUT - Update an existing plugin (proxy admin only)
/claude-code/plugins/{name}/enable - POST - Enable a plugin (proxy admin only)
/claude-code/plugins/{name}/disable - POST - Disable a plugin (proxy admin only)
/claude-code/plugins/{name} - DELETE - Delete a plugin (proxy admin only)
"""
import json
import re
from collections.abc import Mapping, Sequence
from datetime import datetime, timezone
from typing import Final, Protocol, TypedDict
from typing import Annotated, Final, Protocol, TypedDict
from fastapi import APIRouter, Depends, HTTPException
from fastapi.responses import JSONResponse
@ -28,6 +28,7 @@ from fastapi.responses import JSONResponse
from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.common_utils.resource_ownership import is_proxy_admin
from litellm.repositories.table_repositories import ClaudeCodePluginRepository
from litellm.types.proxy.claude_code_endpoints import (
ListPluginsResponse,
@ -221,6 +222,18 @@ def _name_conflict_error(name: str) -> HTTPException:
)
def _require_proxy_admin(user_api_key_dict: UserAPIKeyAuth) -> None:
"""Catalog mutations are restricted to proxy admins: marketplace.json is served
unauthenticated and any registered/updated entry is immediately installable by
every user, so a non-admin key must never be able to add or overwrite one.
"""
if not is_proxy_admin(user_api_key_dict):
raise HTTPException(
status_code=403,
detail={"error": "Only proxy admins may modify the Claude Code plugin marketplace."},
)
@router.post(
"/claude-code/plugins",
tags=["Claude Code Marketplace"],
@ -242,6 +255,8 @@ async def register_plugin(
the same name already exists it returns 409 Conflict; use
PUT /claude-code/plugins/{plugin_name} to update an existing plugin.
Requires a proxy admin API key.
Parameters:
- name: Plugin name (kebab-case)
- source: Git source reference (github, url, or git-subdir format)
@ -271,6 +286,8 @@ async def register_plugin(
from prisma.errors import UniqueViolationError
try:
_require_proxy_admin(user_api_key_dict)
prisma_client: Final = await _get_prisma_client()
if not re.match(r"^[a-z0-9-]+$", request.name):
@ -468,6 +485,7 @@ async def get_plugin(
async def update_plugin(
plugin_name: str,
request: UpdatePluginRequest,
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
):
"""
Update an existing plugin in the LiteLLM marketplace.
@ -481,6 +499,8 @@ async def update_plugin(
Returns 404 if no plugin with the given name exists; use
POST /claude-code/plugins to create a new plugin.
Requires a proxy admin API key.
Parameters:
- plugin_name: Name of the plugin to update (path parameter)
- source: Git source reference (github, url, or git-subdir format)
@ -509,6 +529,8 @@ async def update_plugin(
from prisma.errors import PrismaError
try:
_require_proxy_admin(user_api_key_dict)
prisma_client: Final = await _get_prisma_client()
_validate_plugin_source(request.source)
@ -566,10 +588,14 @@ async def enable_plugin(
"""
Enable a disabled plugin.
Requires a proxy admin API key.
Parameters:
- plugin_name: The name of the plugin to enable
"""
try:
_require_proxy_admin(user_api_key_dict)
prisma_client: Final = await _get_prisma_client()
plugin: Final[_PluginRecord | None] = await ClaudeCodePluginRepository(prisma_client).table.find_unique(
@ -611,10 +637,14 @@ async def disable_plugin(
"""
Disable a plugin without deleting it.
Requires a proxy admin API key.
Parameters:
- plugin_name: The name of the plugin to disable
"""
try:
_require_proxy_admin(user_api_key_dict)
prisma_client: Final = await _get_prisma_client()
plugin: Final[_PluginRecord | None] = await ClaudeCodePluginRepository(prisma_client).table.find_unique(
@ -656,10 +686,14 @@ async def delete_plugin(
"""
Delete a plugin from the marketplace.
Requires a proxy admin API key.
Parameters:
- plugin_name: The name of the plugin to delete
"""
try:
_require_proxy_admin(user_api_key_dict)
prisma_client: Final = await _get_prisma_client()
plugin: Final[_PluginRecord | None] = await ClaudeCodePluginRepository(prisma_client).table.find_unique(

View file

@ -18,6 +18,7 @@ from litellm.constants import (
DEFAULT_HEALTH_CHECK_PROMPT,
HEALTH_CHECK_TIMEOUT_SECONDS,
)
from litellm.router_utils.auto_router_model_naming import classify_strategy_router_model
ILLEGAL_DISPLAY_PARAMS: Final = [
"messages",
@ -182,30 +183,17 @@ async def run_with_timeout(task, timeout):
return {"error": "Timeout exceeded", "exception": timeout_exception}
def _is_semantic_auto_router_deployment(litellm_params: dict) -> bool:
"""
True for semantic auto_router deployments (auto_router/<name>) that are not
sub-strategies (complexity_router, adaptive_router, quality_router).
These are meta-routers that select among real LLM deployments at request time;
they have no LLM endpoint to health-check.
"""
def _is_strategy_router_deployment(litellm_params: dict) -> bool:
"""True for strategy-router deployments."""
model: Final[object] = litellm_params.get("model", "")
if not isinstance(model, str):
return False
if not model.startswith("auto_router/"):
return False
for sub_strategy in ("complexity_router", "adaptive_router", "quality_router"):
if model.startswith(f"auto_router/{sub_strategy}"):
return False
return True
return isinstance(model, str) and classify_strategy_router_model(model) is not None
async def _run_model_health_check(model: dict):
litellm_params = model["litellm_params"]
model_info: Final = model.get("model_info", {})
if _is_semantic_auto_router_deployment(litellm_params):
if _is_strategy_router_deployment(litellm_params):
return {}
mode: Final = _resolve_health_check_mode(

View file

@ -5,6 +5,7 @@ from typing import Any, Final, cast
import litellm
from litellm._logging import verbose_proxy_logger
from litellm.constants import BACKGROUND_INTERACTION_COST_POLLING_ENABLED
from litellm.integrations.custom_logger import CustomLogger
from litellm.litellm_core_utils.core_helpers import (
_get_parent_otel_span_from_kwargs,
@ -318,6 +319,21 @@ class _ProxyDBLogger(CustomLogger):
elif budget_reservation is not None:
await _release_budget_reservation(budget_reservation=budget_reservation)
else:
if _is_unbilled_interaction_response(completion_response):
if BACKGROUND_INTERACTION_COST_POLLING_ENABLED and _is_unbilled_in_progress_interaction(
completion_response
):
verbose_proxy_logger.debug(
"Cost tracking deferred for in-progress background interaction; "
"the budget reservation stays open until the poll task logs the final usage"
)
return
await _release_budget_reservation(budget_reservation=budget_reservation)
verbose_proxy_logger.debug(
"Released the budget reservation for an interaction create with no usage "
"that no poll task will settle"
)
return
await _release_budget_reservation(budget_reservation=budget_reservation)
# Non-model call types (health checks, afile_delete) have no model or standard_logging_object.
# Use .get() for "stream" to avoid KeyError on health checks.
@ -463,6 +479,24 @@ def _write_spend_metadata_to_kwargs(kwargs: dict, metadata: dict) -> None:
bucket[key] = value
def _is_unbilled_interaction_response(completion_response: object) -> bool:
from litellm.interactions.background_cost_polling import missing_usage_is_expected
from litellm.types.interactions import InteractionsAPIResponse
if not isinstance(completion_response, InteractionsAPIResponse):
return False
return completion_response.usage is None and missing_usage_is_expected(completion_response)
def _is_unbilled_in_progress_interaction(completion_response: object) -> bool:
from litellm.interactions.background_cost_polling import is_pollable_background_interaction
from litellm.types.interactions import InteractionsAPIResponse
if not isinstance(completion_response, InteractionsAPIResponse):
return False
return completion_response.usage is None and is_pollable_background_interaction(completion_response)
def _should_track_cost_callback(
user_api_key: str | None,
user_id: str | None,

View file

@ -440,6 +440,12 @@ class CallTypes(str, Enum):
query = "query"
aquery = "aquery"
#########################################################
# Google Interactions API Call Types
#########################################################
create_interaction = "create_interaction"
acreate_interaction = "acreate_interaction"
#########################################################
# Container Call Types
#########################################################

View file

@ -14551,6 +14551,8 @@
]
},
"databricks/databricks-bge-large-en": {
"cache_creation_input_token_cost": 1.0003e-07,
"cache_read_input_token_cost": 1.0003e-07,
"input_cost_per_token": 1.0003e-07,
"input_dbu_cost_per_token": 1.429e-06,
"litellm_provider": "databricks",
@ -14566,6 +14568,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-claude-3-7-sonnet": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14581,10 +14585,41 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-fable-5": {
"cache_creation_input_token_cost": 1.250004e-05,
"cache_read_input_token_cost": 1.00002e-06,
"input_cost_per_token": 1.000006e-05,
"input_dbu_cost_per_token": 0.000142858,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
},
"mode": "chat",
"output_cost_per_token": 5.000002e-05,
"output_dbu_cost_per_token": 0.000714286,
"prompt_cache_min_tokens": 512,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_mid_conversation_system": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": false,
"thinking_always_on": true
},
"databricks/databricks-claude-haiku-4-5": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.0003e-07,
"input_cost_per_token": 1.00002e-06,
"input_dbu_cost_per_token": 1.4286e-05,
"litellm_provider": "databricks",
@ -14600,10 +14635,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-opus-4": {
"cache_creation_input_token_cost": 1.874999e-05,
"cache_read_input_token_cost": 1.50003e-06,
"input_cost_per_token": 1.5000020000000002e-05,
"input_dbu_cost_per_token": 0.000214286,
"litellm_provider": "databricks",
@ -14619,10 +14657,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-opus-4-1": {
"cache_creation_input_token_cost": 1.874999e-05,
"cache_read_input_token_cost": 1.50003e-06,
"input_cost_per_token": 1.5000020000000002e-05,
"input_dbu_cost_per_token": 0.000214286,
"litellm_provider": "databricks",
@ -14638,10 +14679,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-opus-4-5": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
@ -14657,11 +14701,14 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true,
"supports_output_config": true
},
"databricks/databricks-claude-opus-4-6": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
@ -14677,10 +14724,93 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
},
"mode": "chat",
"output_cost_per_token": 2.500001e-05,
"output_dbu_cost_per_token": 0.000357143,
"prompt_cache_min_tokens": 2048,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": true
},
"databricks/databricks-claude-opus-4-8": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
},
"mode": "chat",
"output_cost_per_token": 2.500001e-05,
"output_dbu_cost_per_token": 0.000357143,
"prompt_cache_min_tokens": 1024,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_mid_conversation_system": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": true
},
"databricks/databricks-claude-opus-5": {
"cache_creation_input_token_cost": 6.25002e-06,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
},
"mode": "chat",
"output_cost_per_token": 2.500001e-05,
"output_dbu_cost_per_token": 0.000357143,
"prompt_cache_min_tokens": 512,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_mid_conversation_system": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": true
},
"databricks/databricks-claude-sonnet-4": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14696,10 +14826,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-4-1": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14715,10 +14848,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-4-5": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14734,10 +14870,13 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.9999900000000002e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
@ -14753,10 +14892,40 @@
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"databricks/databricks-claude-sonnet-5": {
"cache_creation_input_token_cost": 3.74997e-06,
"cache_read_input_token_cost": 3.0002e-07,
"input_cost_per_token": 2.99999e-06,
"input_dbu_cost_per_token": 4.2857e-05,
"litellm_provider": "databricks",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"metadata": {
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields. Introductory launch rates of 28.571 input / 142.857 output / 35.714 cache write / 2.857 cache read DBU run through 2026-08-31; the standard rates are listed here because entries carry no expiry date."
},
"mode": "chat",
"output_cost_per_token": 1.500002e-05,
"output_dbu_cost_per_token": 0.000214286,
"prompt_cache_min_tokens": 1024,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_mid_conversation_system": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_sampling_params": false,
"supports_tool_choice": true,
"supports_vision": true
},
"databricks/databricks-gemini-2-5-flash": {
"cache_creation_input_token_cost": 3.0002e-07,
"cache_read_input_token_cost": 3.0002e-08,
"input_cost_per_token": 3.0001999999999996e-07,
"input_dbu_cost_per_token": 4.285999999999999e-06,
"litellm_provider": "databricks",
@ -14771,9 +14940,12 @@
"output_dbu_cost_per_token": 3.5714e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-2-5-pro": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.24999e-07,
"input_cost_per_token": 1.24999e-06,
"input_dbu_cost_per_token": 1.7857e-05,
"litellm_provider": "databricks",
@ -14788,9 +14960,12 @@
"output_dbu_cost_per_token": 0.000142857,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-3-1-flash-lite": {
"cache_creation_input_token_cost": 3.1248e-07,
"cache_read_input_token_cost": 3.122e-08,
"input_cost_per_token": 3.1248e-07,
"input_dbu_cost_per_token": 4.464e-06,
"litellm_provider": "databricks",
@ -14805,9 +14980,12 @@
"output_dbu_cost_per_token": 2.6786e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-3-1-pro": {
"cache_creation_input_token_cost": 2.49998e-06,
"cache_read_input_token_cost": 2.4997e-07,
"input_cost_per_token": 2.49998e-06,
"input_dbu_cost_per_token": 3.5714e-05,
"litellm_provider": "databricks",
@ -14822,9 +15000,12 @@
"output_dbu_cost_per_token": 0.000214286,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-3-flash": {
"cache_creation_input_token_cost": 6.2503e-07,
"cache_read_input_token_cost": 6.251e-08,
"input_cost_per_token": 6.2503e-07,
"input_dbu_cost_per_token": 8.929e-06,
"litellm_provider": "databricks",
@ -14839,9 +15020,12 @@
"output_dbu_cost_per_token": 5.3571e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemini-3-pro": {
"cache_creation_input_token_cost": 2.49998e-06,
"cache_read_input_token_cost": 2.4997e-07,
"input_cost_per_token": 2.49998e-06,
"input_dbu_cost_per_token": 3.5714e-05,
"litellm_provider": "databricks",
@ -14856,9 +15040,12 @@
"output_dbu_cost_per_token": 0.000214286,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_tool_choice": true
},
"databricks/databricks-gemma-3-12b": {
"cache_creation_input_token_cost": 1.5001e-07,
"cache_read_input_token_cost": 1.5001e-07,
"input_cost_per_token": 1.5000999999999998e-07,
"input_dbu_cost_per_token": 2.1429999999999996e-06,
"litellm_provider": "databricks",
@ -14874,6 +15061,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-gpt-5": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.2502e-07,
"input_cost_per_token": 1.24999e-06,
"input_dbu_cost_per_token": 1.7857e-05,
"litellm_provider": "databricks",
@ -14886,9 +15075,12 @@
"mode": "chat",
"output_cost_per_token": 9.999990000000002e-06,
"output_dbu_cost_per_token": 0.000142857,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-1": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.2502e-07,
"input_cost_per_token": 1.24999e-06,
"input_dbu_cost_per_token": 1.7857e-05,
"litellm_provider": "databricks",
@ -14901,9 +15093,12 @@
"mode": "chat",
"output_cost_per_token": 9.999990000000002e-06,
"output_dbu_cost_per_token": 0.000142857,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-1-codex-max": {
"cache_creation_input_token_cost": 1.24999e-06,
"cache_read_input_token_cost": 1.2502e-07,
"input_cost_per_token": 1.24999e-06,
"input_dbu_cost_per_token": 1.7857e-05,
"litellm_provider": "databricks",
@ -14916,9 +15111,12 @@
"mode": "chat",
"output_cost_per_token": 9.999990000000002e-06,
"output_dbu_cost_per_token": 0.000142857,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-1-codex-mini": {
"cache_creation_input_token_cost": 2.4997e-07,
"cache_read_input_token_cost": 2.499e-08,
"input_cost_per_token": 2.4997e-07,
"input_dbu_cost_per_token": 3.571e-06,
"litellm_provider": "databricks",
@ -14931,9 +15129,12 @@
"mode": "chat",
"output_cost_per_token": 1.99997e-06,
"output_dbu_cost_per_token": 2.8571e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-2": {
"cache_creation_input_token_cost": 1.75e-06,
"cache_read_input_token_cost": 1.75e-07,
"input_cost_per_token": 1.75e-06,
"input_dbu_cost_per_token": 2.5e-05,
"litellm_provider": "databricks",
@ -14946,9 +15147,12 @@
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_dbu_cost_per_token": 0.0002,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-2-codex": {
"cache_creation_input_token_cost": 1.75e-06,
"cache_read_input_token_cost": 1.75e-07,
"input_cost_per_token": 1.75e-06,
"input_dbu_cost_per_token": 2.5e-05,
"litellm_provider": "databricks",
@ -14961,9 +15165,12 @@
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_dbu_cost_per_token": 0.0002,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-3-codex": {
"cache_creation_input_token_cost": 1.75e-06,
"cache_read_input_token_cost": 1.75e-07,
"input_cost_per_token": 1.75e-06,
"input_dbu_cost_per_token": 2.5e-05,
"litellm_provider": "databricks",
@ -14976,9 +15183,12 @@
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_dbu_cost_per_token": 0.0002,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-4": {
"cache_creation_input_token_cost": 2.49998e-06,
"cache_read_input_token_cost": 2.4997e-07,
"input_cost_per_token": 2.49998e-06,
"input_dbu_cost_per_token": 3.5714e-05,
"litellm_provider": "databricks",
@ -14991,9 +15201,12 @@
"mode": "chat",
"output_cost_per_token": 1.5000020000000002e-05,
"output_dbu_cost_per_token": 0.000214286,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-4-mini": {
"cache_creation_input_token_cost": 7.4998e-07,
"cache_read_input_token_cost": 7.497e-08,
"input_cost_per_token": 7.4998e-07,
"input_dbu_cost_per_token": 1.0714e-05,
"litellm_provider": "databricks",
@ -15006,9 +15219,12 @@
"mode": "chat",
"output_cost_per_token": 4.50002e-06,
"output_dbu_cost_per_token": 6.4286e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-4-nano": {
"cache_creation_input_token_cost": 1.9999e-07,
"cache_read_input_token_cost": 2.002e-08,
"input_cost_per_token": 1.9999e-07,
"input_dbu_cost_per_token": 2.857e-06,
"litellm_provider": "databricks",
@ -15021,9 +15237,12 @@
"mode": "chat",
"output_cost_per_token": 1.24999e-06,
"output_dbu_cost_per_token": 1.7857e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-mini": {
"cache_creation_input_token_cost": 2.4997e-07,
"cache_read_input_token_cost": 2.499e-08,
"input_cost_per_token": 2.4997000000000006e-07,
"input_dbu_cost_per_token": 3.571e-06,
"litellm_provider": "databricks",
@ -15036,9 +15255,12 @@
"mode": "chat",
"output_cost_per_token": 1.9999700000000004e-06,
"output_dbu_cost_per_token": 2.8571e-05,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-5-nano": {
"cache_creation_input_token_cost": 4.998e-08,
"cache_read_input_token_cost": 4.97e-09,
"input_cost_per_token": 4.998e-08,
"input_dbu_cost_per_token": 7.14e-07,
"litellm_provider": "databricks",
@ -15051,9 +15273,12 @@
"mode": "chat",
"output_cost_per_token": 3.9998000000000007e-07,
"output_dbu_cost_per_token": 5.714000000000001e-06,
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
"supports_prompt_caching": true
},
"databricks/databricks-gpt-oss-120b": {
"cache_creation_input_token_cost": 1.5001e-07,
"cache_read_input_token_cost": 1.5001e-07,
"input_cost_per_token": 1.5000999999999998e-07,
"input_dbu_cost_per_token": 2.1429999999999996e-06,
"litellm_provider": "databricks",
@ -15069,6 +15294,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-gpt-oss-20b": {
"cache_creation_input_token_cost": 7e-08,
"cache_read_input_token_cost": 7e-08,
"input_cost_per_token": 7e-08,
"input_dbu_cost_per_token": 1e-06,
"litellm_provider": "databricks",
@ -15084,6 +15311,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-gte-large-en": {
"cache_creation_input_token_cost": 1.2999e-07,
"cache_read_input_token_cost": 1.2999e-07,
"input_cost_per_token": 1.2999000000000001e-07,
"input_dbu_cost_per_token": 1.857e-06,
"litellm_provider": "databricks",
@ -15099,6 +15328,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-llama-2-70b-chat": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",
@ -15115,6 +15346,8 @@
"supports_tool_choice": true
},
"databricks/databricks-llama-4-maverick": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",
@ -15131,6 +15364,8 @@
"supports_tool_choice": true
},
"databricks/databricks-meta-llama-3-1-405b-instruct": {
"cache_creation_input_token_cost": 5.00003e-06,
"cache_read_input_token_cost": 5.00003e-06,
"input_cost_per_token": 5.00003e-06,
"input_dbu_cost_per_token": 7.1429e-05,
"litellm_provider": "databricks",
@ -15147,6 +15382,8 @@
"supports_tool_choice": true
},
"databricks/databricks-meta-llama-3-1-8b-instruct": {
"cache_creation_input_token_cost": 1.5001e-07,
"cache_read_input_token_cost": 1.5001e-07,
"input_cost_per_token": 1.5000999999999998e-07,
"input_dbu_cost_per_token": 2.1429999999999996e-06,
"litellm_provider": "databricks",
@ -15162,6 +15399,8 @@
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
},
"databricks/databricks-meta-llama-3-3-70b-instruct": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",
@ -15178,6 +15417,8 @@
"supports_tool_choice": true
},
"databricks/databricks-meta-llama-3-70b-instruct": {
"cache_creation_input_token_cost": 1.00002e-06,
"cache_read_input_token_cost": 1.00002e-06,
"input_cost_per_token": 1.00002e-06,
"input_dbu_cost_per_token": 1.4286e-05,
"litellm_provider": "databricks",
@ -15194,6 +15435,8 @@
"supports_tool_choice": true
},
"databricks/databricks-mixtral-8x7b-instruct": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",
@ -15210,6 +15453,8 @@
"supports_tool_choice": true
},
"databricks/databricks-mpt-30b-instruct": {
"cache_creation_input_token_cost": 1.00002e-06,
"cache_read_input_token_cost": 1.00002e-06,
"input_cost_per_token": 1.00002e-06,
"input_dbu_cost_per_token": 1.4286e-05,
"litellm_provider": "databricks",
@ -15226,6 +15471,8 @@
"supports_tool_choice": true
},
"databricks/databricks-mpt-7b-instruct": {
"cache_creation_input_token_cost": 5.0001e-07,
"cache_read_input_token_cost": 5.0001e-07,
"input_cost_per_token": 5.0001e-07,
"input_dbu_cost_per_token": 7.143e-06,
"litellm_provider": "databricks",

View file

@ -14,6 +14,14 @@ longer signal it.
## [Unreleased]
### Added
- **team**: `soft_budget`, `tags`, and `soft_budget_alerting_emails` attributes on `litellm_team`, matching what `/team/new` and `/team/update` already accept; `soft_budget_alerting_emails` is sent under `metadata`, where the proxy reads it
### Fixed
- **team**: Read now decodes the `team_info` envelope `/team/info` actually returns, so team attributes refresh from the proxy instead of always falling back to the prior state
### Changed
- **Versioning**: the provider is now published at the LiteLLM version, from the same commit as the proxy, on every LiteLLM release (dev, rc, stable). The `0.x` line ends at `0.4.0`; a `~> 0.4` constraint will not receive further releases, so re-pin to the LiteLLM version your proxy runs (for example `~> 1.99.0`). Existing `0.x` versions remain in the registry and keep verifying

View file

@ -24,11 +24,18 @@ resource "litellm_team" "advanced_team" {
# Budget and rate limiting
max_budget = 1000.0
soft_budget = 800.0
budget_duration = "1mo"
tpm_limit = 500000
rpm_limit = 5000
blocked = false
# Who gets paged when spend crosses soft_budget
soft_budget_alerting_emails = ["finops@example.com"]
# Tags for spend tracking and tag-based routing
tags = ["team:ai-research", "environment:production"]
# Team member permissions
team_member_permissions = [
"create_key",
@ -91,7 +98,9 @@ The following arguments are supported:
* `models` - (Optional) List of model names that this team can access.
* `metadata` - (Optional) A map of metadata key-value pairs associated with the team.
* `metadata` - (Optional) A map of string metadata key-value pairs associated with the team. `tags` and `soft_budget_alerting_emails` are stored by the proxy under metadata but are managed through their own attributes below, not this map.
* `tags` - (Optional) List of tags applied to the team, used for [spend tracking](https://docs.litellm.ai/docs/proxy/enterprise#tracking-spend-for-custom-tags) and [tag-based routing](https://docs.litellm.ai/docs/proxy/tag_routing).
* `blocked` - (Optional) Whether the team is blocked from making requests. Default is `false`.
@ -101,6 +110,10 @@ The following arguments are supported:
* `max_budget` - (Optional) Maximum budget allocated to the team.
* `soft_budget` - (Optional) Spend threshold at which the proxy sends a soft budget alert without blocking requests.
* `soft_budget_alerting_emails` - (Optional) List of email addresses notified when the team's spend crosses `soft_budget`.
* `budget_duration` - (Optional) Duration for the budget cycle. Valid values are:
* `daily`
* `weekly`

View file

@ -53,6 +53,11 @@ func ResourceLiteLLMTeam() *schema.Resource {
Type: schema.TypeFloat,
Optional: true,
},
"soft_budget": {
Type: schema.TypeFloat,
Optional: true,
Description: "Spend threshold that triggers a soft budget alert without blocking requests",
},
"budget_duration": {
Type: schema.TypeString,
Optional: true,
@ -72,6 +77,18 @@ func ResourceLiteLLMTeam() *schema.Resource {
Elem: &schema.Schema{Type: schema.TypeString},
Description: "List of permissions granted to team members",
},
"tags": {
Type: schema.TypeList,
Optional: true,
Elem: &schema.Schema{Type: schema.TypeString},
Description: "Tags for spend tracking and tag-based routing",
},
"soft_budget_alerting_emails": {
Type: schema.TypeList,
Optional: true,
Elem: &schema.Schema{Type: schema.TypeString},
Description: "Email addresses alerted when the team crosses soft_budget",
},
},
}
}
@ -117,21 +134,20 @@ func resourceLiteLLMTeamRead(d *schema.ResourceData, m interface{}) error {
return nil
}
var teamResp TeamResponse
if err := json.NewDecoder(resp.Body).Decode(&teamResp); err != nil {
var infoResp TeamInfoResponse
if err := json.NewDecoder(resp.Body).Decode(&infoResp); err != nil {
return fmt.Errorf("error decoding team info response: %w", err)
}
teamResp := infoResp.TeamInfo
// Update the state with values from the response or fall back to the data passed in during creation
d.Set("team_alias", GetStringValue(teamResp.TeamAlias, d.Get("team_alias").(string)))
d.Set("organization_id", GetStringValue(teamResp.OrganizationID, d.Get("organization_id").(string)))
// Handle metadata separately as it's a map
if teamResp.Metadata != nil {
d.Set("metadata", teamResp.Metadata)
} else {
d.Set("metadata", d.Get("metadata"))
}
metadata, tags, alertEmails := splitTeamMetadata(teamResp.Metadata)
d.Set("metadata", metadata)
d.Set("tags", tags)
d.Set("soft_budget_alerting_emails", alertEmails)
if teamResp.TPMLimit != nil {
d.Set("tpm_limit", *teamResp.TPMLimit)
@ -142,6 +158,7 @@ func resourceLiteLLMTeamRead(d *schema.ResourceData, m interface{}) error {
if teamResp.MaxBudget != nil {
d.Set("max_budget", *teamResp.MaxBudget)
}
d.Set("soft_budget", teamResp.SoftBudget)
d.Set("budget_duration", GetStringValue(teamResp.BudgetDuration, d.Get("budget_duration").(string)))
// Handle models separately as it's a list
@ -240,15 +257,77 @@ func buildTeamData(d *schema.ResourceData, teamID string) map[string]interface{}
"team_alias": d.Get("team_alias").(string),
}
for _, key := range []string{"organization_id", "metadata", "tpm_limit", "rpm_limit", "max_budget", "budget_duration", "models", "blocked", "team_member_permissions"} {
for _, key := range []string{"organization_id", "tpm_limit", "rpm_limit", "max_budget", "budget_duration", "models", "blocked", "team_member_permissions"} {
if v, ok := d.GetOk(key); ok {
teamData[key] = v
}
}
if v, ok := d.GetOk("soft_budget"); ok {
teamData["soft_budget"] = v
} else if d.HasChange("soft_budget") {
teamData["soft_budget"] = nil
}
if v, ok := d.GetOk("tags"); ok || d.HasChange("tags") {
teamData["tags"] = v
}
if metadata := buildTeamMetadata(d); metadata != nil {
teamData["metadata"] = metadata
}
return teamData
}
// /team/update replaces metadata wholesale, so the full map must go out whenever either half changed.
func buildTeamMetadata(d *schema.ResourceData) map[string]interface{} {
metadata := map[string]interface{}{}
for k, v := range d.Get("metadata").(map[string]interface{}) {
metadata[k] = v
}
if v, ok := d.GetOk("soft_budget_alerting_emails"); ok {
metadata["soft_budget_alerting_emails"] = v
}
if len(metadata) == 0 && !d.HasChange("metadata") && !d.HasChange("soft_budget_alerting_emails") {
return nil
}
return metadata
}
func splitTeamMetadata(raw map[string]interface{}) (map[string]string, []string, []string) {
metadata := map[string]string{}
var tags, alertEmails []string
for k, v := range raw {
switch k {
case "tags":
tags = toStringSlice(v)
case "soft_budget_alerting_emails":
alertEmails = toStringSlice(v)
case "team_member_budget_id":
default:
if s, ok := v.(string); ok {
metadata[k] = s
}
}
}
return metadata, tags, alertEmails
}
func toStringSlice(v interface{}) []string {
items, ok := v.([]interface{})
if !ok {
return nil
}
out := make([]string, 0, len(items))
for _, item := range items {
if s, ok := item.(string); ok {
out = append(out, s)
}
}
return out
}
func handleResponse(resp *http.Response, action string) error {
if resp.StatusCode != http.StatusOK {
body, _ := io.ReadAll(resp.Body)

View file

@ -0,0 +1,184 @@
package litellm
import (
"context"
"encoding/json"
"io"
"net/http"
"net/http/httptest"
"reflect"
"testing"
"github.com/hashicorp/terraform-plugin-sdk/v2/helper/schema"
"github.com/hashicorp/terraform-plugin-sdk/v2/terraform"
)
func newTeamTestServer(t *testing.T, captured *map[string]interface{}, infoBody string) *httptest.Server {
t.Helper()
return httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/json")
switch r.URL.Path {
case endpointTeamNew, endpointTeamUpdate:
body, _ := io.ReadAll(r.Body)
json.Unmarshal(body, captured)
w.Write([]byte(`{}`))
case endpointTeamInfo:
w.Write([]byte(infoBody))
case endpointTeamPermissionsList:
w.Write([]byte(`{"team_id":"team-1","team_member_permissions":[],"all_available_permissions":[]}`))
default:
w.WriteHeader(http.StatusNotFound)
}
}))
}
const teamInfoWithSoftBudget = `{
"team_id": "team-1",
"team_info": {
"team_id": "team-1",
"team_alias": "insights",
"max_budget": 750.0,
"soft_budget": 600.0,
"models": ["claude-haiku-4-5"],
"metadata": {
"department": "customer-insights",
"tags": ["team:customer-insights", "environment:production"],
"soft_budget_alerting_emails": ["finops@example.com"],
"team_member_budget_id": "budget-1"
}
},
"keys": [],
"team_memberships": []
}`
func TestTeamCreateSendsSoftBudgetTagsAndAlertEmails(t *testing.T) {
var captured map[string]interface{}
srv := newTeamTestServer(t, &captured, teamInfoWithSoftBudget)
defer srv.Close()
d := schema.TestResourceDataRaw(t, ResourceLiteLLMTeam().Schema, map[string]interface{}{
"team_alias": "insights",
"max_budget": 750.0,
"soft_budget": 600.0,
"tags": []interface{}{"team:customer-insights", "environment:production"},
"soft_budget_alerting_emails": []interface{}{"finops@example.com"},
"metadata": map[string]interface{}{"department": "customer-insights"},
})
if err := resourceLiteLLMTeamCreate(d, NewClient(srv.URL, "test-key", true)); err != nil {
t.Fatalf("create failed: %v", err)
}
if got := captured["soft_budget"]; got != 600.0 {
t.Fatalf("payload soft_budget = %v, want 600", got)
}
wantTags := []interface{}{"team:customer-insights", "environment:production"}
if got := captured["tags"]; !reflect.DeepEqual(got, wantTags) {
t.Fatalf("payload tags = %v, want %v", got, wantTags)
}
wantMetadata := map[string]interface{}{
"department": "customer-insights",
"soft_budget_alerting_emails": []interface{}{"finops@example.com"},
}
if got := captured["metadata"]; !reflect.DeepEqual(got, wantMetadata) {
t.Fatalf("payload metadata = %v, want %v", got, wantMetadata)
}
}
func TestTeamReadMapsTeamInfoEnvelope(t *testing.T) {
var captured map[string]interface{}
srv := newTeamTestServer(t, &captured, teamInfoWithSoftBudget)
defer srv.Close()
d := schema.TestResourceDataRaw(t, ResourceLiteLLMTeam().Schema, map[string]interface{}{})
d.SetId("team-1")
if err := resourceLiteLLMTeamRead(d, NewClient(srv.URL, "test-key", true)); err != nil {
t.Fatalf("read failed: %v", err)
}
if got := d.Get("team_alias"); got != "insights" {
t.Fatalf("team_alias = %v, want insights", got)
}
if got := d.Get("soft_budget"); got != 600.0 {
t.Fatalf("soft_budget = %v, want 600", got)
}
if got := d.Get("max_budget"); got != 750.0 {
t.Fatalf("max_budget = %v, want 750", got)
}
wantTags := []interface{}{"team:customer-insights", "environment:production"}
if got := d.Get("tags"); !reflect.DeepEqual(got, wantTags) {
t.Fatalf("tags = %v, want %v", got, wantTags)
}
wantEmails := []interface{}{"finops@example.com"}
if got := d.Get("soft_budget_alerting_emails"); !reflect.DeepEqual(got, wantEmails) {
t.Fatalf("soft_budget_alerting_emails = %v, want %v", got, wantEmails)
}
wantMetadata := map[string]interface{}{"department": "customer-insights"}
if got := d.Get("metadata"); !reflect.DeepEqual(got, wantMetadata) {
t.Fatalf("metadata = %v, want %v (server-managed team_member_budget_id dropped)", got, wantMetadata)
}
}
func TestTeamUpdateClearsRemovedTagsAndSoftBudget(t *testing.T) {
var captured map[string]interface{}
srv := newTeamTestServer(t, &captured, `{"team_id":"team-1","team_info":{"team_id":"team-1","team_alias":"insights"},"keys":[],"team_memberships":[]}`)
defer srv.Close()
res := ResourceLiteLLMTeam()
priorData := schema.TestResourceDataRaw(t, res.Schema, map[string]interface{}{
"team_alias": "insights",
"soft_budget": 600.0,
"tags": []interface{}{"team:to-be-removed"},
"soft_budget_alerting_emails": []interface{}{"ops@example.com"},
"metadata": map[string]interface{}{"department": "eng"},
})
priorData.SetId("team-1")
prior := priorData.State()
config := terraform.NewResourceConfigRaw(map[string]interface{}{
"team_alias": "insights",
"metadata": map[string]interface{}{"department": "eng"},
})
diff, err := res.Diff(context.Background(), prior, config, nil)
if err != nil {
t.Fatalf("diff failed: %v", err)
}
d, err := schema.InternalMap(res.Schema).Data(prior, diff)
if err != nil {
t.Fatalf("data failed: %v", err)
}
if err := resourceLiteLLMTeamUpdate(d, NewClient(srv.URL, "test-key", true)); err != nil {
t.Fatalf("update failed: %v", err)
}
if got, ok := captured["soft_budget"]; !ok || got != nil {
t.Fatalf("payload soft_budget = %v (present=%v), want explicit null", got, ok)
}
if got := captured["tags"]; !reflect.DeepEqual(got, []interface{}{}) {
t.Fatalf("payload tags = %v, want []", got)
}
if got := captured["metadata"]; !reflect.DeepEqual(got, map[string]interface{}{"department": "eng"}) {
t.Fatalf("payload metadata = %v, want department only", got)
}
}
func TestTeamReadClearsSoftBudgetWhenProxyReturnsNull(t *testing.T) {
var captured map[string]interface{}
srv := newTeamTestServer(t, &captured, `{"team_id":"team-1","team_info":{"team_id":"team-1","team_alias":"insights","soft_budget":null},"keys":[],"team_memberships":[]}`)
defer srv.Close()
d := schema.TestResourceDataRaw(t, ResourceLiteLLMTeam().Schema, map[string]interface{}{
"team_alias": "insights",
"soft_budget": 600.0,
})
d.SetId("team-1")
if err := resourceLiteLLMTeamRead(d, NewClient(srv.URL, "test-key", true)); err != nil {
t.Fatalf("read failed: %v", err)
}
if got := d.Get("soft_budget"); got != 0.0 {
t.Fatalf("soft_budget = %v, want cleared after the proxy returned null", got)
}
}

View file

@ -33,6 +33,11 @@ type ModelRequest struct {
Additional map[string]interface{} `json:"additional"`
}
type TeamInfoResponse struct {
TeamID string `json:"team_id"`
TeamInfo TeamResponse `json:"team_info"`
}
// TeamResponse represents a response from the API containing team information.
type TeamResponse struct {
TeamID string `json:"team_id,omitempty"`
@ -42,6 +47,7 @@ type TeamResponse struct {
TPMLimit *int `json:"tpm_limit,omitempty"`
RPMLimit *int `json:"rpm_limit,omitempty"`
MaxBudget *float64 `json:"max_budget,omitempty"`
SoftBudget *float64 `json:"soft_budget,omitempty"`
BudgetDuration string `json:"budget_duration,omitempty"`
Models []string `json:"models"`
Blocked bool `json:"blocked,omitempty"`

View file

@ -13,6 +13,7 @@ from litellm.a2a_protocol.card_resolver import (
LiteLLMA2ACardResolver,
fix_agent_card_url,
is_localhost_or_internal_url,
normalize_agent_card_interfaces,
set_agent_card_url,
)
@ -114,3 +115,26 @@ def test_fix_agent_card_url_updates_interface_when_top_level_is_localhost():
assert result.url == "https://my-public-agent.example.com/"
assert result.supported_interfaces[0].url == "https://my-public-agent.example.com/"
def test_normalize_agent_card_interfaces_downgrades_miscased_interfaces_to_the_0_3_dialect():
pb2 = pytest.importorskip("a2a.types.a2a_pb2")
card = pb2.AgentCard(
name="langgraph",
supported_interfaces=[
pb2.AgentInterface(url="http://a/", protocol_binding="jsonrpc", protocol_version="1.0"),
pb2.AgentInterface(url="http://b/", protocol_binding="JSONRPC", protocol_version="1.0"),
pb2.AgentInterface(url="http://c/", protocol_binding="websocket", protocol_version="1.0"),
],
)
normalized = normalize_agent_card_interfaces(card)
assert [(i.protocol_binding, i.protocol_version) for i in normalized.supported_interfaces] == [
("JSONRPC", "0.3"),
("JSONRPC", "1.0"),
("websocket", "1.0"),
]
assert card.supported_interfaces[0].protocol_binding == "jsonrpc"
assert card.supported_interfaces[0].protocol_version == "1.0"

View file

@ -176,10 +176,62 @@ _AGENT_A_HEADERS = {"x-agent-token": "token-for-a", "x-tenant": "tenant-a"}
_AGENT_B_HEADERS = {"x-agent-token": "token-for-b", "x-tenant": "tenant-b"}
_LANGGRAPH_TASK_REPLY = {
"jsonrpc": "2.0",
"id": "reply",
"result": {
"kind": "task",
"id": "run-1:task-1",
"contextId": "thread-1",
"history": [
{
"kind": "message",
"role": "user",
"parts": [{"kind": "text", "text": "hi"}],
"messageId": "m-user",
"taskId": "run-1:task-1",
"contextId": "thread-1",
},
{
"kind": "message",
"role": "agent",
"parts": [{"kind": "text", "text": "langgraph echo: hi"}],
"messageId": "m-agent",
"taskId": "run-1:task-1",
"contextId": "thread-1",
},
],
"status": {"state": "completed", "timestamp": "2026-08-24T00:00:00+00:00"},
"artifacts": [
{
"artifactId": "art-1",
"name": "Assistant Response",
"parts": [{"kind": "text", "text": "langgraph echo: hi"}],
}
],
},
}
_LOWERCASE_BINDING_CARD = {
"name": "langgraph-agent",
"version": "1.0.0",
"capabilities": {"streaming": True},
"defaultInputModes": ["text/plain"],
"defaultOutputModes": ["text/plain"],
"skills": [],
"supportedInterfaces": [
{"url": "http://127.0.0.1:9/", "protocolBinding": "jsonrpc", "protocolVersion": "1.0"}
],
}
class _RequestRecorder:
"""Records the headers httpx put on the wire, per outbound request."""
def __init__(self):
def __init__(self, card=_AGENT_CARD, rpc_reply=_RPC_REPLY):
self.card = card
self.rpc_reply = rpc_reply
self.card_requests = []
self.rpc_requests = []
self.client = None
@ -188,23 +240,23 @@ class _RequestRecorder:
headers = {k.lower(): v for k, v in request.headers.items()}
if request.method == "GET":
self.card_requests.append(headers)
return httpx.Response(200, json=_AGENT_CARD)
return httpx.Response(200, json=self.card)
self.rpc_requests.append(headers)
return httpx.Response(200, json=_RPC_REPLY)
return httpx.Response(200, json=self.rpc_reply)
def _a2a_client_cache_key(timeout: float) -> str:
return "async_httpx_client" + f"timeout_{timeout}" + httpxSpecialProvider.A2AProvider
async def _seed_shared_a2a_client() -> _RequestRecorder:
async def _seed_shared_a2a_client(card=_AGENT_CARD, rpc_reply=_RPC_REPLY) -> _RequestRecorder:
"""Put the one A2A client the cache will hand out behind a mock transport.
Seeding has to happen on the test's own event loop, because the client cache keys on
it. The injected client is a real httpx.AsyncClient, so the merge of per-request
headers over client defaults, which is what these tests are about, stays real.
"""
recorder = _RequestRecorder()
recorder = _RequestRecorder(card=card, rpc_reply=rpc_reply)
handler = AsyncHTTPHandler(timeout=DEFAULT_A2A_AGENT_TIMEOUT)
owned_client = handler.client
handler.client = httpx.AsyncClient(transport=httpx.MockTransport(recorder))
@ -311,6 +363,25 @@ async def test_streaming_send_carries_only_its_own_caller_headers(isolated_clien
assert received["b"]["x-tenant"] == "tenant-b"
@pytest.mark.asyncio
async def test_lowercase_protocol_binding_card_round_trips_the_langgraph_dialect(isolated_client_cache):
"""LangGraph Platform serves cards with protocolBinding "jsonrpc" and answers in the
A2A 0.3 JSON dialect ("kind"-discriminated) while declaring protocolVersion "1.0".
Without binding normalization client creation raises ValueError("no compatible
transports found."); without the version downgrade the SDK's strict v1 transport
rejects the reply with 'Message type "lf.a2a.v1.Task" has no field named "kind"'."""
await _seed_shared_a2a_client(card=_LOWERCASE_BINDING_CARD, rpc_reply=_LANGGRAPH_TASK_REPLY)
a2a_client = await create_a2a_client(base_url="http://127.0.0.1:9")
response = await _send_message(a2a_client, _send_request("lc"))
assert type(response.root.result).__name__ == "Task"
assert response.root.result.artifacts[0].parts[0].root.text == "langgraph echo: hi"
interface = a2a_client._litellm_agent_card.supported_interfaces[0]
assert interface.protocol_binding == "JSONRPC"
assert interface.protocol_version == "0.3"
@pytest.mark.asyncio
async def test_agent_card_fetch_carries_the_callers_headers(isolated_client_cache):
"""Agent cards can sit behind the same auth as the agent, so the card fetch must stay

View file

@ -3762,3 +3762,112 @@ def test_response_incomplete_stream_event_without_details_defaults_to_length():
result = iterator.chunk_parser(chunk)
assert result.choices[0].finish_reason == "length"
def test_assistant_message_with_tool_calls_keeps_its_content():
"""Regression for https://github.com/BerriAI/litellm/issues/24985.
An assistant turn that both answered and called a tool used to lose its whole message:
the branch handling tool_calls emitted the calls and dropped the text.
"""
handler = LiteLLMResponsesTransformationHandler()
messages = [
{"role": "user", "content": "What is the weather in Denver?"},
{
"role": "assistant",
"content": "Let me look that up.",
"tool_calls": [
{
"id": "call_1",
"type": "function",
"function": {"name": "get_weather", "arguments": '{"city": "Denver"}'},
}
],
},
{"role": "tool", "tool_call_id": "call_1", "content": "88F"},
]
input_items, _ = handler.convert_chat_completion_messages_to_responses_api(messages)
assistant_message = next(
item for item in input_items if item.get("type") == "message" and item.get("role") == "assistant"
)
assert assistant_message["content"] == [{"type": "output_text", "text": "Let me look that up."}]
assert [item.get("type") for item in input_items] == [
"message",
"message",
"function_call",
"function_call_output",
]
def test_assistant_thinking_blocks_become_a_reasoning_input_item():
"""Thinking blocks are how an Anthropic-shaped turn carries reasoning into this bridge."""
handler = LiteLLMResponsesTransformationHandler()
messages = [
{"role": "user", "content": "What is the weather in Denver?"},
{
"role": "assistant",
"content": "Denver is sunny.",
"thinking_blocks": [
{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "sig1"},
{"type": "redacted_thinking", "data": "REDACTED"},
],
},
{"role": "user", "content": "Why?"},
]
input_items, _ = handler.convert_chat_completion_messages_to_responses_api(messages)
reasoning_item = next(item for item in input_items if item.get("type") == "reasoning")
assert reasoning_item["summary"] == [{"type": "summary_text", "text": "August in Denver is dry."}]
assert "id" not in reasoning_item
def test_thinking_only_assistant_turn_still_sends_its_reasoning():
"""An assistant turn can be pure reasoning, with no visible text and no tool call."""
handler = LiteLLMResponsesTransformationHandler()
messages = [
{"role": "user", "content": "What is the weather in Denver?"},
{
"role": "assistant",
"content": None,
"thinking_blocks": [
{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "sig1"}
],
},
{"role": "user", "content": "Why?"},
]
input_items, _ = handler.convert_chat_completion_messages_to_responses_api(messages)
reasoning_items = [item for item in input_items if item.get("type") == "reasoning"]
assert len(reasoning_items) == 1
assert reasoning_items[0]["summary"] == [{"type": "summary_text", "text": "August in Denver is dry."}]
def test_stored_reasoning_items_win_over_thinking_blocks():
"""A minted reasoning id beats a re-derived one, so the two must not both be sent."""
handler = LiteLLMResponsesTransformationHandler()
messages = [
{
"role": "assistant",
"content": "Denver is sunny.",
"reasoning_items": [
{
"type": "reasoning",
"id": "rs_real",
"summary": [{"type": "summary_text", "text": "August in Denver is dry."}],
}
],
"thinking_blocks": [
{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "rs_real"}
],
},
]
input_items, _ = handler.convert_chat_completion_messages_to_responses_api(messages)
reasoning_items = [item for item in input_items if item.get("type") == "reasoning"]
assert len(reasoning_items) == 1
assert reasoning_items[0]["id"] == "rs_real"

View file

@ -1599,6 +1599,14 @@ class TestEnableAnthropicPromptCaching:
assert supports_prompt_caching(model=model, custom_llm_provider=provider) is True
assert self._points(model=model, provider=provider) == []
def test_databricks_claude_not_injected_despite_caching_support(self, monkeypatch, local_model_cost_map):
from litellm.utils import supports_prompt_caching
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
model = "databricks/databricks-claude-sonnet-4-5"
assert supports_prompt_caching(model=model, custom_llm_provider="databricks") is True
assert self._points(model=model, provider="databricks") == []
def test_model_without_caching_support_not_injected(self, monkeypatch):
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
assert self._points(model="anthropic.claude-3-5-sonnet-20240620-v1:0", provider="bedrock") == []

View file

@ -0,0 +1,545 @@
import asyncio
import time
from itertools import islice
from typing import Optional
import pytest
from litellm.interactions.background_cost_polling import (
_SETTLED_KEY,
_poll_intervals,
BackgroundInteractionPollContext,
maybe_schedule_background_interaction_cost_polling,
maybe_settle_background_interaction_before_delete,
poll_and_log_background_interaction_cost,
)
from litellm.litellm_core_utils.litellm_logging import Logging as LitellmLogging
from litellm.types.interactions import InteractionsAPIResponse
USAGE_BLOCK = {
"total_tokens": 175,
"total_input_tokens": 100,
"input_tokens_by_modality": [{"modality": "text", "tokens": 100}],
"total_cached_tokens": 0,
"total_output_tokens": 50,
"output_tokens_by_modality": [{"modality": "text", "tokens": 50}],
"total_tool_use_tokens": 0,
"total_thought_tokens": 25,
}
def _logging_obj(
call_type: str = "acreate_interaction",
litellm_params: Optional[dict] = None,
) -> LitellmLogging:
logging_obj = LitellmLogging(
model="gemini-2.5-flash",
messages=[],
stream=False,
call_type=call_type,
start_time=time.time(),
litellm_call_id="bg-interactions-call-id",
function_id="bg-interactions-fn-id",
)
logging_obj.update_environment_variables(
litellm_params=litellm_params or {},
optional_params={},
model="gemini-2.5-flash",
custom_llm_provider="gemini",
input="hi",
)
return logging_obj
def _reservation() -> dict:
return {"reserved_cost": 0.05, "entries": [], "finalized": False, "input_cost": 0.001}
def _logging_obj_with_reservation(reservation: dict) -> LitellmLogging:
return _logging_obj(litellm_params={"metadata": {"user_api_key_budget_reservation": reservation}})
async def _raise_on_billing(result: InteractionsAPIResponse) -> None:
raise RuntimeError("cost calculation failed for a settled background interaction")
def _context(logging_obj: LitellmLogging, timeout_seconds: float = 1.0) -> BackgroundInteractionPollContext:
return BackgroundInteractionPollContext(
interaction_id="interactions/bg-abc",
custom_llm_provider="gemini",
logging_obj=logging_obj,
initial_interval_seconds=0.001,
max_interval_seconds=0.002,
timeout_seconds=timeout_seconds,
)
def _response(status: str, with_usage: bool) -> InteractionsAPIResponse:
return InteractionsAPIResponse(
id="interactions/bg-abc",
model="gemini-2.5-flash",
status=status,
steps=[],
usage=dict(USAGE_BLOCK) if with_usage else None,
)
def _fetch_sequence(*responses):
remaining = list(responses)
calls = []
async def fetch(context):
calls.append(context.interaction_id)
item = remaining.pop(0) if len(remaining) > 1 else remaining[0]
if isinstance(item, Exception):
raise item
return item
return fetch, calls
@pytest.mark.parametrize(
"initial, maximum",
[(0.0, 0.002), (0.001, 0.0), (-1.0, 0.002), (0.0, 0.0)],
)
def test_poll_intervals_stops_instead_of_looping_on_a_non_positive_interval(initial, maximum):
intervals = list(islice(_poll_intervals(initial=initial, maximum=maximum, timeout=3600.0), 10))
assert len(intervals) < 10
assert all(interval > 0 for interval in intervals)
@pytest.mark.asyncio
async def test_poller_bills_once_when_interaction_completes():
logging_obj = _logging_obj()
fetch, calls = _fetch_sequence(
_response("in_progress", with_usage=False),
_response("completed", with_usage=True),
)
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert len(calls) == 2
assert logging_obj.model_call_details["response_cost"] > 0
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
@pytest.mark.asyncio
async def test_poller_bills_an_interaction_paused_for_a_tool_result():
logging_obj = _logging_obj()
fetch, calls = _fetch_sequence(
_response("in_progress", with_usage=False),
_response("requires_action", with_usage=True),
)
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert len(calls) == 2
assert logging_obj.model_call_details["response_cost"] > 0
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
@pytest.mark.asyncio
async def test_poller_does_not_pin_the_budget_for_an_interaction_paused_for_a_tool_result():
reservation = _reservation()
logging_obj = _logging_obj_with_reservation(reservation)
fetch, _ = _fetch_sequence(_response("requires_action", with_usage=True))
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert logging_obj.model_call_details["response_cost"] > 0
assert reservation["finalized"] is False
@pytest.mark.asyncio
async def test_poller_stops_without_billing_on_terminal_status_without_usage():
logging_obj = _logging_obj()
fetch, calls = _fetch_sequence(_response("failed", with_usage=False))
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert len(calls) == 1
assert logging_obj.model_call_details.get("response_cost") is None
@pytest.mark.asyncio
async def test_poller_gives_up_after_timeout_without_billing():
logging_obj = _logging_obj()
fetch, calls = _fetch_sequence(_response("in_progress", with_usage=False))
await poll_and_log_background_interaction_cost(
_context(logging_obj, timeout_seconds=0.01),
fetch_interaction=fetch,
)
assert len(calls) >= 2
assert logging_obj.model_call_details.get("response_cost") is None
@pytest.mark.asyncio
async def test_poller_releases_budget_reservation_when_interaction_ends_without_usage():
reservation = _reservation()
logging_obj = _logging_obj_with_reservation(reservation)
fetch, _ = _fetch_sequence(_response("failed", with_usage=False))
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert reservation["finalized"] is True
@pytest.mark.asyncio
async def test_poller_releases_budget_reservation_on_timeout_give_up():
reservation = _reservation()
logging_obj = _logging_obj_with_reservation(reservation)
fetch, _ = _fetch_sequence(_response("in_progress", with_usage=False))
await poll_and_log_background_interaction_cost(
_context(logging_obj, timeout_seconds=0.01),
fetch_interaction=fetch,
)
assert reservation["finalized"] is True
@pytest.mark.asyncio
async def test_poller_releases_budget_reservation_when_billing_raises():
reservation = _reservation()
logging_obj = _logging_obj_with_reservation(reservation)
fetch, _ = _fetch_sequence(_response("completed", with_usage=True))
logging_obj.async_log_background_interaction_completion = _raise_on_billing
with pytest.raises(RuntimeError):
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert reservation["finalized"] is True
@pytest.mark.asyncio
async def test_poller_leaves_reservation_reconciliation_to_the_completion_event():
reservation = _reservation()
logging_obj = _logging_obj_with_reservation(reservation)
fetch, _ = _fetch_sequence(
_response("in_progress", with_usage=False),
_response("completed", with_usage=True),
)
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert logging_obj.model_call_details["response_cost"] > 0
assert reservation["finalized"] is False
@pytest.mark.asyncio
async def test_poller_retries_after_fetch_error_and_still_bills():
logging_obj = _logging_obj()
fetch, calls = _fetch_sequence(
RuntimeError("transient network error"),
_response("completed", with_usage=True),
)
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert len(calls) == 2
assert logging_obj.model_call_details["response_cost"] > 0
@pytest.mark.asyncio
async def test_schedule_creates_poll_task_for_in_progress_create():
logging_obj = _logging_obj()
task = maybe_schedule_background_interaction_cost_polling(
response=_response("in_progress", with_usage=False),
create_kwargs={"litellm_logging_obj": logging_obj},
custom_llm_provider="gemini",
)
assert isinstance(task, asyncio.Task)
task.cancel()
with pytest.raises(asyncio.CancelledError):
await task
@pytest.mark.asyncio
@pytest.mark.parametrize(
"response,create_kwargs",
[
(_response("completed", with_usage=True), {"litellm_logging_obj": "placeholder"}),
(_response("in_progress", with_usage=False), {}),
("not a response", {"litellm_logging_obj": "placeholder"}),
],
)
async def test_schedule_skips_non_pollable_results(response, create_kwargs):
if create_kwargs.get("litellm_logging_obj") == "placeholder":
create_kwargs = {"litellm_logging_obj": _logging_obj()}
task = maybe_schedule_background_interaction_cost_polling(
response=response,
create_kwargs=create_kwargs,
custom_llm_provider="gemini",
)
assert task is None
def _register_poll(logging_obj: LitellmLogging, poll_fetch=None) -> asyncio.Task:
import litellm.interactions.background_cost_polling as bg
if poll_fetch is None:
poll_fetch, _ = _fetch_sequence(_response("in_progress", with_usage=False))
context = _context(logging_obj)
task = asyncio.create_task(poll_and_log_background_interaction_cost(context, fetch_interaction=poll_fetch))
bg._ACTIVE_POLLS[context.interaction_id] = bg._ActiveBackgroundPoll(task=task, context=context)
task.add_done_callback(lambda finished: bg._discard_poll(context.interaction_id, finished))
return task
@pytest.mark.asyncio
async def test_delete_settlement_bills_an_interaction_paused_for_a_tool_result():
logging_obj = _logging_obj()
task = _register_poll(logging_obj)
fetch, calls = _fetch_sequence(_response("requires_action", with_usage=True))
await maybe_settle_background_interaction_before_delete(
interaction_id="interactions/bg-abc",
fetch_interaction=fetch,
)
assert len(calls) == 1
assert logging_obj.model_call_details["response_cost"] > 0
await asyncio.wait_for(task, timeout=5)
@pytest.mark.asyncio
async def test_delete_settlement_bills_pending_background_interaction():
logging_obj = _logging_obj()
task = _register_poll(logging_obj)
fetch, calls = _fetch_sequence(_response("completed", with_usage=True))
await maybe_settle_background_interaction_before_delete(
interaction_id="interactions/bg-abc",
fetch_interaction=fetch,
)
assert len(calls) == 1
assert logging_obj.model_call_details["response_cost"] > 0
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
await asyncio.wait_for(task, timeout=5)
@pytest.mark.asyncio
async def test_delete_settlement_releases_reservation_when_still_in_progress():
reservation = _reservation()
logging_obj = _logging_obj_with_reservation(reservation)
task = _register_poll(logging_obj)
fetch, _ = _fetch_sequence(_response("in_progress", with_usage=False))
await maybe_settle_background_interaction_before_delete(
interaction_id="interactions/bg-abc",
fetch_interaction=fetch,
)
assert reservation["finalized"] is True
assert logging_obj.model_call_details.get("response_cost") is None
await asyncio.wait_for(task, timeout=5)
@pytest.mark.asyncio
async def test_delete_settlement_releases_reservation_when_prefetch_fails():
reservation = _reservation()
logging_obj = _logging_obj_with_reservation(reservation)
task = _register_poll(logging_obj)
fetch, _ = _fetch_sequence(RuntimeError("interaction already deleted"))
await maybe_settle_background_interaction_before_delete(
interaction_id="interactions/bg-abc",
fetch_interaction=fetch,
)
assert reservation["finalized"] is True
assert logging_obj.model_call_details.get("response_cost") is None
await asyncio.wait_for(task, timeout=5)
@pytest.mark.asyncio
async def test_delete_settlement_releases_reservation_when_billing_raises():
reservation = _reservation()
logging_obj = _logging_obj_with_reservation(reservation)
task = _register_poll(logging_obj)
fetch, _ = _fetch_sequence(_response("completed", with_usage=True))
logging_obj.async_log_background_interaction_completion = _raise_on_billing
with pytest.raises(RuntimeError):
await maybe_settle_background_interaction_before_delete(
interaction_id="interactions/bg-abc",
fetch_interaction=fetch,
)
assert reservation["finalized"] is True
await asyncio.wait_for(task, timeout=5)
@pytest.mark.asyncio
async def test_delete_settlement_ignores_interactions_without_pending_poll():
fetch, calls = _fetch_sequence(_response("completed", with_usage=True))
await maybe_settle_background_interaction_before_delete(
interaction_id="interactions/never-polled",
fetch_interaction=fetch,
)
assert calls == []
@pytest.mark.asyncio
async def test_delete_settlement_noop_after_poll_task_finished():
logging_obj = _logging_obj()
poll_fetch, _ = _fetch_sequence(_response("completed", with_usage=True))
task = _register_poll(logging_obj, poll_fetch=poll_fetch)
await asyncio.wait_for(task, timeout=5)
assert logging_obj.model_call_details["response_cost"] > 0
settle_fetch, settle_calls = _fetch_sequence(_response("completed", with_usage=True))
await maybe_settle_background_interaction_before_delete(
interaction_id="interactions/bg-abc",
fetch_interaction=settle_fetch,
)
assert settle_calls == []
@pytest.mark.asyncio
async def test_delete_settlement_does_not_rebill_when_gate_already_claimed():
logging_obj = _logging_obj()
logging_obj.model_call_details[_SETTLED_KEY] = True
task = _register_poll(logging_obj)
fetch, calls = _fetch_sequence(_response("completed", with_usage=True))
await maybe_settle_background_interaction_before_delete(
interaction_id="interactions/bg-abc",
fetch_interaction=fetch,
)
assert len(calls) == 1
assert logging_obj.model_call_details.get("response_cost") is None
await asyncio.wait_for(task, timeout=5)
@pytest.mark.asyncio
async def test_poller_exits_without_billing_once_settled_elsewhere():
logging_obj = _logging_obj()
logging_obj.model_call_details[_SETTLED_KEY] = True
fetch, calls = _fetch_sequence(_response("completed", with_usage=True))
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert calls == []
assert logging_obj.model_call_details.get("response_cost") is None
@pytest.mark.asyncio
async def test_schedule_respects_kill_switch(monkeypatch):
import litellm.interactions.background_cost_polling as module
monkeypatch.setattr(module, "BACKGROUND_INTERACTION_COST_POLLING_ENABLED", False)
task = maybe_schedule_background_interaction_cost_polling(
response=_response("in_progress", with_usage=False),
create_kwargs={"litellm_logging_obj": _logging_obj()},
custom_llm_provider="gemini",
)
assert task is None
def test_every_status_the_api_can_return_is_either_pollable_or_terminal():
"""
The proxy bills a usage-less create in exactly two ways: it polls the
interaction until it settles, or it recognises the status as terminal and
settles immediately. A status in neither set is billed by nobody, alerts
nobody, and releases its budget reservation, which is the zero-spend bug
this whole module exists to fix.
Pinned against the generated spec enum rather than a hand-written list, so
a status Google adds later breaks this test instead of silently shipping
another unbilled path.
"""
from litellm.interactions.background_cost_polling import _POLLABLE_STATUSES, _TERMINAL_STATUSES
from litellm.types.interactions.generated import Status1
spec_statuses = {member.value for member in Status1}
handled = _POLLABLE_STATUSES | _TERMINAL_STATUSES
assert spec_statuses - handled == set()
assert handled - spec_statuses == set()
@pytest.mark.asyncio
async def test_schedule_creates_poll_task_for_queued_create():
"""
``queued`` is the API's not-started-yet state. It carries no usage, so the
create cannot bill it, and it is not terminal, so nothing settles it:
without a poll task it is never charged at all.
"""
logging_obj = _logging_obj()
task = maybe_schedule_background_interaction_cost_polling(
response=_response("queued", with_usage=False),
create_kwargs={"litellm_logging_obj": logging_obj},
custom_llm_provider="gemini",
)
assert isinstance(task, asyncio.Task)
task.cancel()
with pytest.raises(asyncio.CancelledError):
await task
@pytest.mark.asyncio
async def test_poller_bills_an_interaction_that_started_out_queued():
logging_obj = _logging_obj()
fetch, calls = _fetch_sequence(
_response("queued", with_usage=False),
_response("in_progress", with_usage=False),
_response("completed", with_usage=True),
)
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
assert len(calls) == 3
assert logging_obj.model_call_details["response_cost"] > 0
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
def test_poll_intervals_double_up_to_the_cap_and_stay_inside_the_timeout():
"""
The degenerate cases are covered above; this pins the shape the proxy
actually ships, so an off-by-one in the doubling or in the remaining-budget
check cannot pass green.
"""
intervals = list(_poll_intervals(initial=5.0, maximum=60.0, timeout=3600.0))
assert intervals[:6] == [5.0, 10.0, 20.0, 40.0, 60.0, 60.0]
assert max(intervals) == 60.0
assert sum(intervals) <= 3600.0
assert sum(intervals) + 60.0 > 3600.0
@pytest.mark.asyncio
async def test_giving_up_on_an_unrecognized_status_says_which_status_it_was(monkeypatch):
"""
A status outside both sets polls for the full timeout and then gives up.
The give-up line is the only trace it leaves, so it has to name the status
rather than reporting it as an interaction that was merely still running.
"""
import litellm.interactions.background_cost_polling as bg
errors = []
monkeypatch.setattr(bg.verbose_logger, "error", lambda *args, **kwargs: errors.append(args))
logging_obj = _logging_obj()
fetch, _ = _fetch_sequence(_response("halted_for_review", with_usage=False))
await poll_and_log_background_interaction_cost(
_context(logging_obj, timeout_seconds=0.01), fetch_interaction=fetch
)
assert len(errors) == 1
assert "halted_for_review" in errors[0]

View file

@ -0,0 +1,150 @@
from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import (
InteractionsUsageObjectTransformation,
)
from litellm.types.utils import Usage
OMNI_VIDEO_USAGE = {
"total_tokens": 18247,
"total_input_tokens": 16,
"input_tokens_by_modality": [{"modality": "text", "tokens": 16}],
"total_cached_tokens": 0,
"total_output_tokens": 17937,
"output_tokens_by_modality": [{"modality": "video", "tokens": 17376}],
"total_tool_use_tokens": 0,
"total_thought_tokens": 294,
}
def test_detects_interactions_usage_object():
assert InteractionsUsageObjectTransformation.is_interactions_usage_object(OMNI_VIDEO_USAGE) is True
def test_rejects_chat_and_responses_api_usage_objects():
chat_usage = {"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30}
responses_api_usage = {"input_tokens": 10, "output_tokens": 20, "total_tokens": 30}
assert InteractionsUsageObjectTransformation.is_interactions_usage_object(chat_usage) is False
assert InteractionsUsageObjectTransformation.is_interactions_usage_object(responses_api_usage) is False
assert InteractionsUsageObjectTransformation.is_interactions_usage_object(None) is False
assert InteractionsUsageObjectTransformation.is_interactions_usage_object("usage") is False
def test_transforms_real_omni_video_usage_block():
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(OMNI_VIDEO_USAGE)
assert isinstance(usage, Usage)
assert usage.prompt_tokens == 16
assert usage.completion_tokens == 17937 + 294
assert usage.total_tokens == 18247
assert usage.prompt_tokens_details is not None
assert usage.prompt_tokens_details.text_tokens == 16
assert usage.completion_tokens_details is not None
assert usage.completion_tokens_details.video_tokens == 17376
assert usage.completion_tokens_details.reasoning_tokens == 294
def test_transforms_reasoning_tokens_spec_field_name():
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
{
"total_input_tokens": 10,
"total_output_tokens": 20,
"total_reasoning_tokens": 5,
}
)
assert usage.completion_tokens == 25
assert usage.completion_tokens_details is not None
assert usage.completion_tokens_details.reasoning_tokens == 5
assert usage.total_tokens == 35
def test_cached_tokens_subtracted_from_text_input():
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
{
"total_input_tokens": 1000,
"input_tokens_by_modality": [{"modality": "text", "tokens": 1000}],
"total_cached_tokens": 400,
"total_output_tokens": 50,
}
)
assert usage.prompt_tokens == 1000
assert usage.prompt_tokens_details is not None
assert usage.prompt_tokens_details.text_tokens == 600
assert usage.prompt_tokens_details.cached_tokens == 400
assert usage._cache_read_input_tokens == 400
def test_cached_tokens_subtracted_per_modality_when_breakdown_present():
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
{
"total_input_tokens": 1500,
"input_tokens_by_modality": [
{"modality": "text", "tokens": 1000},
{"modality": "audio", "tokens": 500},
],
"total_cached_tokens": 300,
"cached_tokens_by_modality": [{"modality": "audio", "tokens": 300}],
"total_output_tokens": 50,
}
)
assert usage.prompt_tokens_details is not None
assert usage.prompt_tokens_details.text_tokens == 1000
assert usage.prompt_tokens_details.audio_tokens == 200
assert usage.prompt_tokens_details.cached_tokens == 300
def test_tool_use_tokens_billed_as_input():
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
{
"total_input_tokens": 100,
"input_tokens_by_modality": [{"modality": "text", "tokens": 100}],
"total_tool_use_tokens": 40,
"tool_use_tokens_by_modality": [{"modality": "text", "tokens": 40}],
"total_output_tokens": 10,
}
)
assert usage.prompt_tokens == 140
assert usage.prompt_tokens_details is not None
assert usage.prompt_tokens_details.text_tokens == 140
def test_google_search_grounding_count_maps_to_web_search_requests():
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
{
"total_input_tokens": 103,
"input_tokens_by_modality": [{"modality": "text", "tokens": 103}],
"total_output_tokens": 226,
"total_thought_tokens": 351,
"grounding_tool_count": [
{"type": "google_search", "count": 3},
{"type": "url_context", "count": 2},
],
}
)
assert usage.prompt_tokens_details is not None
assert usage.prompt_tokens_details.web_search_requests == 3
def test_no_grounding_leaves_web_search_requests_unset():
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
{
"total_input_tokens": 10,
"input_tokens_by_modality": [{"modality": "text", "tokens": 10}],
"total_output_tokens": 5,
}
)
assert usage.prompt_tokens_details is not None
assert getattr(usage.prompt_tokens_details, "web_search_requests", None) is None
def test_document_modality_folds_into_text():
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
{
"total_input_tokens": 80,
"input_tokens_by_modality": [
{"modality": "text", "tokens": 30},
{"modality": "document", "tokens": 50},
],
"total_output_tokens": 10,
}
)
assert usage.prompt_tokens_details is not None
assert usage.prompt_tokens_details.text_tokens == 80

View file

@ -378,7 +378,7 @@ def test_shipped_rules_flag_unmapped_fable_as_always_on_thinking(shipped_cost_ma
"model,provider",
[
("claude-opus-4-9@20260101", "vertex_ai"),
("databricks-claude-opus-5-1", "databricks"),
("databricks-claude-haiku-5-1", "databricks"),
],
)
def test_shipped_rules_are_provider_neutral_for_unmapped_ids(shipped_cost_map, model, provider):
@ -388,6 +388,8 @@ def test_shipped_rules_are_provider_neutral_for_unmapped_ids(shipped_cost_map, m
assert info["supports_adaptive_thinking"] is True
assert info["supports_mid_conversation_system"] is True
assert info["supports_function_calling"] is True
assert not info.get("input_cost_per_token")
assert not info.get("output_cost_per_token")
@pytest.mark.parametrize(

View file

@ -4546,6 +4546,323 @@ def test_zero_token_video_usage_preserves_duration_seconds(logging_obj):
assert payload["completion_tokens"] == 0
INTERACTIONS_USAGE_BLOCK = {
"total_tokens": 175,
"total_input_tokens": 100,
"input_tokens_by_modality": [{"modality": "text", "tokens": 100}],
"total_cached_tokens": 0,
"total_output_tokens": 50,
"output_tokens_by_modality": [{"modality": "text", "tokens": 50}],
"total_tool_use_tokens": 0,
"total_thought_tokens": 25,
}
def _interactions_logging_obj(stream: bool, call_type: str = "acreate"):
logging_obj = LitellmLogging(
model="gemini-2.5-flash",
messages=[],
stream=stream,
call_type=call_type,
start_time=time.time(),
litellm_call_id="interactions-call-id",
function_id="interactions-fn-id",
)
logging_obj.update_environment_variables(
litellm_params={},
optional_params={},
model="gemini-2.5-flash",
custom_llm_provider="gemini",
input="hi",
)
return logging_obj
@pytest.mark.parametrize("call_type", ["create", "acreate", "create_interaction", "acreate_interaction"])
def test_interactions_response_is_recognized_for_logging(call_type):
from litellm.types.interactions import InteractionsAPIResponse
logging_obj = _interactions_logging_obj(stream=False, call_type=call_type)
response = InteractionsAPIResponse(
id="interactions/abc",
model="gemini-2.5-flash",
status="completed",
usage=dict(INTERACTIONS_USAGE_BLOCK),
)
assert logging_obj._is_recognized_call_type_for_logging(logging_result=response) is True
@pytest.mark.parametrize("call_type", ["acreate", "acreate_interaction"])
def test_in_progress_background_create_is_not_billed(call_type):
import datetime as dt
from litellm.types.interactions import InteractionsAPIResponse
logging_obj = _interactions_logging_obj(stream=False, call_type=call_type)
response = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
assert logging_obj._is_recognized_call_type_for_logging(logging_result=response) is False
logging_obj._success_handler_helper_fn(
result=response,
start_time=dt.datetime.now(),
end_time=dt.datetime.now(),
cache_hit=False,
)
assert logging_obj.model_call_details.get("response_cost") is None
assert logging_obj.model_call_details.get("standard_logging_object") is None
@pytest.mark.asyncio
async def test_background_interaction_completion_rebills_after_in_progress_success():
import datetime as dt
from litellm.types.interactions import InteractionsAPIResponse
logging_obj = _interactions_logging_obj(stream=False)
in_progress = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
await logging_obj.async_success_handler(
result=in_progress,
start_time=dt.datetime.now(),
end_time=dt.datetime.now(),
)
assert logging_obj.model_call_details.get("response_cost") is None
assert logging_obj.should_run_logging(event_type="async_success") is False
completed = InteractionsAPIResponse(
id="interactions/abc",
model="gemini-2.5-flash",
status="completed",
steps=[],
usage=dict(INTERACTIONS_USAGE_BLOCK),
)
await logging_obj.async_log_background_interaction_completion(result=completed)
assert logging_obj.model_call_details["response_cost"] > 0
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
@pytest.mark.asyncio
async def test_background_interaction_completion_prices_the_settled_body_itself():
"""
The poll fetches the settled body through its own client call, which
prices it against a throwaway logging object holding none of this
request's deployment context. Adopting that price would bill a
custom-priced deployment at the wrong rate, and it would also satisfy the
"already calculated" shortcut and skip repricing, leaving the breakdown at
the zeros the usage-less create stamped and writing those to the spend log.
"""
import datetime as dt
from litellm.types.interactions import InteractionsAPIResponse
logging_obj = _interactions_logging_obj(stream=False)
in_progress = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
await logging_obj.async_success_handler(
result=in_progress,
start_time=dt.datetime.now(),
end_time=dt.datetime.now(),
)
completed = InteractionsAPIResponse(
id="interactions/abc",
model="gemini-2.5-flash",
status="completed",
steps=[],
usage=dict(INTERACTIONS_USAGE_BLOCK),
)
completed._hidden_params = {"response_cost": 99.0}
await logging_obj.async_log_background_interaction_completion(result=completed)
response_cost = logging_obj.model_call_details["response_cost"]
assert response_cost != 99.0
assert response_cost > 0
cost_breakdown = logging_obj.model_call_details["standard_logging_object"]["cost_breakdown"]
assert cost_breakdown["total_cost"] == response_cost
assert cost_breakdown["input_cost"] > 0
assert cost_breakdown["output_cost"] > 0
@pytest.mark.asyncio
async def test_background_interaction_completion_lets_otel_emit_the_cost_span():
"""
OTEL, and every integration that derives from it, dedupes span emission on
a marker kept in the request's own metadata. The in-progress create claims
that marker, so without clearing it the settled completion, the only event
carrying usage and cost, is discarded as a duplicate and every
OTEL-family backend shows the interaction as a span with no cost at all.
"""
import datetime as dt
from litellm.integrations.opentelemetry import OpenTelemetry, OpenTelemetryConfig
from litellm.types.interactions import InteractionsAPIResponse
otel = OpenTelemetry(config=OpenTelemetryConfig(exporter="console"))
logging_obj = _interactions_logging_obj(stream=False)
in_progress = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
await logging_obj.async_success_handler(
result=in_progress,
start_time=dt.datetime.now(),
end_time=dt.datetime.now(),
)
assert otel._emit_once(logging_obj.model_call_details, "success") is True
assert otel._emit_once(logging_obj.model_call_details, "success") is False
completed = InteractionsAPIResponse(
id="interactions/abc",
model="gemini-2.5-flash",
status="completed",
steps=[],
usage=dict(INTERACTIONS_USAGE_BLOCK),
)
await logging_obj.async_log_background_interaction_completion(result=completed)
assert otel._emit_once(logging_obj.model_call_details, "success") is True
@pytest.mark.parametrize(
"call_type",
["aget", "get", "aget_interaction", "adelete_interaction", "acancel_interaction"],
)
def test_interactions_get_poll_is_not_billed(call_type):
import datetime as dt
from litellm.types.interactions import InteractionsAPIResponse
logging_obj = _interactions_logging_obj(stream=False, call_type=call_type)
response = InteractionsAPIResponse(
id="interactions/abc",
model="gemini-2.5-flash",
status="completed",
steps=[],
usage=dict(INTERACTIONS_USAGE_BLOCK),
)
assert logging_obj._is_recognized_call_type_for_logging(logging_result=response) is False
logging_obj._success_handler_helper_fn(
result=response,
start_time=dt.datetime.now(),
end_time=dt.datetime.now(),
cache_hit=False,
)
assert logging_obj.model_call_details.get("response_cost") is None
assert logging_obj.model_call_details.get("standard_logging_object") is None
def test_non_streaming_interactions_success_sets_response_cost_and_usage():
import datetime as dt
from litellm.types.interactions import InteractionsAPIResponse
logging_obj = _interactions_logging_obj(stream=False)
response = InteractionsAPIResponse(
id="interactions/abc",
model="gemini-2.5-flash",
status="completed",
steps=[],
usage=dict(INTERACTIONS_USAGE_BLOCK),
)
logging_obj._success_handler_helper_fn(
result=response,
start_time=dt.datetime.now(),
end_time=dt.datetime.now(),
cache_hit=False,
)
assert logging_obj.model_call_details["response_cost"] > 0
standard_logging_object = logging_obj.model_call_details["standard_logging_object"]
assert standard_logging_object["prompt_tokens"] == 100
assert standard_logging_object["completion_tokens"] == 75
assert standard_logging_object["total_tokens"] == 175
assert standard_logging_object["response_cost"] == logging_obj.model_call_details["response_cost"]
def test_assembled_streaming_response_from_completed_interaction_event():
import datetime as dt
from litellm.types.interactions import (
InteractionsAPIResponse,
InteractionsAPIStreamingResponse,
)
logging_obj = _interactions_logging_obj(stream=True)
completed_event = InteractionsAPIStreamingResponse(
event_type="interaction.completed",
interaction={
"id": "interactions/abc",
"model": "gemini-2.5-flash",
"status": "completed",
"steps": [],
"usage": dict(INTERACTIONS_USAGE_BLOCK),
},
)
assembled = logging_obj._get_assembled_streaming_response(
result=completed_event,
start_time=dt.datetime.now(),
end_time=dt.datetime.now(),
is_async=True,
streaming_chunks=[],
)
assert isinstance(assembled, InteractionsAPIResponse)
assert assembled.usage == INTERACTIONS_USAGE_BLOCK
in_progress_event = InteractionsAPIStreamingResponse(event_type="interaction.in_progress")
assert (
logging_obj._get_assembled_streaming_response(
result=in_progress_event,
start_time=dt.datetime.now(),
end_time=dt.datetime.now(),
is_async=True,
streaming_chunks=[],
)
is None
)
def test_assembled_streaming_response_from_legacy_completed_chunk():
from litellm.types.interactions import (
InteractionsAPIResponse,
InteractionsAPIStreamingResponse,
)
legacy_chunk = InteractionsAPIStreamingResponse(
event_type="interaction.complete",
id="interactions/legacy",
model="gemini-2.5-flash",
status="completed",
outputs=[],
usage=dict(INTERACTIONS_USAGE_BLOCK),
)
assembled = LitellmLogging._assemble_completed_interaction_response(legacy_chunk)
assert isinstance(assembled, InteractionsAPIResponse)
assert assembled.id == "interactions/legacy"
assert assembled.usage == INTERACTIONS_USAGE_BLOCK
def test_standard_logging_payload_maps_interactions_usage():
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(
response_obj={"usage": dict(INTERACTIONS_USAGE_BLOCK)}
)
assert usage.prompt_tokens == 100
assert usage.completion_tokens == 75
assert usage.total_tokens == 175
def test_pre_call_does_not_pin_request_in_module_state(logging_obj):
"""
pre_call/post_call must not stash their locals (full messages, the Logging

View file

@ -359,6 +359,48 @@ def test_translate_anthropic_messages_to_openai_thinking_blocks():
assert result[1]["tool_calls"][0]["id"] == "toolu_01234"
def test_translate_anthropic_messages_to_openai_sets_reasoning_content():
"""Reasoning-aware chat providers read reasoning_content, so thinking text must land there.
Without it Moonshot and DeepSeek fill in a single-space placeholder and the model gets
a blank where its own prior reasoning belongs.
"""
anthropic_messages = [
AnthropicMessagesUserMessageParam(
role="user",
content=[{"type": "text", "text": "Which city is best for a picnic?"}],
),
AnthopicMessagesAssistantMessageParam(
role="assistant",
content=[
{"type": "thinking", "thinking": "Denver is dry in August.", "signature": "sig1"},
{"type": "thinking", "thinking": "San Francisco is foggy.", "signature": "sig2"},
{"type": "redacted_thinking", "data": "REDACTED"},
{"type": "text", "text": "Denver."},
],
),
]
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(messages=anthropic_messages)
assert result[1]["reasoning_content"] == "Denver is dry in August.\nSan Francisco is foggy."
assert result[1]["content"] == "Denver."
def test_translate_anthropic_messages_to_openai_sets_no_reasoning_content_without_thinking():
anthropic_messages = [
AnthopicMessagesAssistantMessageParam(
role="assistant",
content=[{"type": "text", "text": "Denver."}],
),
]
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(messages=anthropic_messages)
assert "reasoning_content" not in result[0]
def test_translate_anthropic_messages_to_openai_tool_message_placement():
"""Test that tool result messages are placed before user messages in the conversation order."""

View file

@ -144,9 +144,15 @@ class TestReasoningItemWithoutSummaryText:
("content_block_delta", 1),
("content_block_stop", 1),
]
assert chunks[1]["content_block"] == {"type": "thinking", "thinking": ""}
assert chunks[1]["content_block"] == {"type": "thinking", "thinking": "", "signature": ""}
assert "".join(c["delta"]["thinking"] for c in chunks[2:4]) == "Weighing options"
def test_the_reasoning_item_id_is_never_streamed_as_a_signature(self):
"""A stand-in signature would be replayed as a real one, so none is ever sent."""
chunks = _drain_async(self._gpt_turn(reasoning_summary_deltas=["Weighing options"]))
assert not [c for c in chunks if c.get("delta", {}).get("type") == "signature_delta"]
class TestToolUseBlockClosedExactlyOnce:
"""Regression for https://github.com/BerriAI/litellm/issues/37273.

View file

@ -486,8 +486,8 @@ class TestTranslateMessagesToResponsesInput:
}
]
def test_assistant_thinking_block_becomes_output_text(self):
"""Assistant thinking block text is included as output_text."""
def test_assistant_thinking_block_becomes_reasoning_item(self):
"""Assistant thinking block becomes a reasoning item, never visible assistant prose."""
messages = [
{
"role": "assistant",
@ -495,7 +495,77 @@ class TestTranslateMessagesToResponsesInput:
}
]
result = _translate_messages(messages)
assert result[0]["content"] == [{"type": "output_text", "text": "Let me reason step by step."}]
assert result == [
{
"type": "reasoning",
"summary": [{"type": "summary_text", "text": "Let me reason step by step."}],
}
]
def test_reasoning_item_carries_no_id(self):
"""A fabricated reasoning id 404s upstream, so the item must go out without one."""
messages = [
{
"role": "assistant",
"content": [{"type": "thinking", "thinking": "Private reasoning.", "signature": "rs_abc123"}],
}
]
result = _translate_messages(messages)
assert "id" not in result[0]
def test_consecutive_thinking_blocks_become_one_reasoning_item(self):
"""Summary parts of one upstream reasoning item are regrouped into that item."""
messages = [
{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "First part."},
{"type": "thinking", "thinking": "Second part."},
],
}
]
result = _translate_messages(messages)
assert result == [
{
"type": "reasoning",
"summary": [
{"type": "summary_text", "text": "First part."},
{"type": "summary_text", "text": "Second part."},
],
}
]
def test_a_tool_call_splits_the_reasoning_items_around_it(self):
"""Thinking on either side of a tool call belongs to two different reasoning items."""
messages = [
{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "Before the call."},
{"type": "tool_use", "id": "call_1", "name": "get_weather", "input": {"city": "Denver"}},
{"type": "thinking", "thinking": "After the call."},
],
}
]
result = _translate_messages(messages)
assert [item["type"] for item in result] == ["reasoning", "function_call", "reasoning"]
assert result[0]["summary"] == [{"type": "summary_text", "text": "Before the call."}]
assert result[2]["summary"] == [{"type": "summary_text", "text": "After the call."}]
def test_thinking_and_text_stay_separate(self):
"""The visible answer stays the only thing in the assistant message."""
messages = [
{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "The user wants Denver."},
{"type": "text", "text": "Denver is the best pick."},
],
}
]
result = _translate_messages(messages)
assert [item["type"] for item in result] == ["reasoning", "message"]
assert result[1]["content"] == [{"type": "output_text", "text": "Denver is the best pick."}]
def test_assistant_empty_thinking_block_skipped(self):
"""Assistant thinking block with empty thinking text is skipped."""
@ -1094,7 +1164,7 @@ def _make_function_call_item(call_id: str, name: str, arguments: str) -> MagicMo
return item
def _make_reasoning_item(summaries: List[str]) -> MagicMock:
def _make_reasoning_item(summaries: List[str], item_id: str = "rs_test_1") -> MagicMock:
"""Build a mock ResponseReasoningItem."""
from openai.types.responses import ResponseReasoningItem # type: ignore[import]
@ -1105,6 +1175,7 @@ def _make_reasoning_item(summaries: List[str]) -> MagicMock:
summary_mocks.append(s)
item = MagicMock(spec=ResponseReasoningItem)
item.id = item_id
item.summary = summary_mocks
return item
@ -1178,6 +1249,53 @@ class TestTranslateResponse:
result: Any = _ADAPTER.translate_response(response)
assert result["content"] == []
def test_null_summary_text_skipped_rather_than_stringified(self):
"""A summary part whose text is null must not reach the client as the word "None"."""
response = _make_mock_response(
output=[
{
"type": "reasoning",
"id": "rs_null_1",
"summary": [{"type": "summary_text", "text": None}],
}
]
)
result: Any = _ADAPTER.translate_response(response)
assert result["content"] == []
def test_reasoning_item_id_never_becomes_a_thinking_signature(self):
"""Only Anthropic can sign a thinking block, so a stand-in signature is never invented."""
reasoning = _make_reasoning_item(["Part one.", "Part two."], item_id="rs_abc123")
response = _make_mock_response(output=[reasoning])
result: Any = _ADAPTER.translate_response(response)
assert [block["signature"] for block in result["content"]] == [None, None]
def test_dict_reasoning_item_becomes_thinking_block(self):
"""A reasoning item arriving as a plain dict is kept, not dropped."""
response = _make_mock_response(
output=[
{
"type": "reasoning",
"id": "rs_dict_1",
"summary": [{"type": "summary_text", "text": "Weighing the options."}],
}
]
)
result: Any = _ADAPTER.translate_response(response)
assert result["content"] == [
{"type": "thinking", "thinking": "Weighing the options.", "signature": None}
]
def test_thinking_blocks_are_dropped_when_replayed_to_anthropic(self):
"""Replaying this turn to an Anthropic model must not send a signature it cannot verify."""
from litellm.litellm_core_utils.prompt_templates.factory import (
_drop_unsignable_thinking_blocks,
)
response = _make_mock_response(output=[_make_reasoning_item(["Part one."], item_id="rs_abc123")])
result: Any = _ADAPTER.translate_response(response)
assert _drop_unsignable_thinking_blocks(result["content"]) == []
def test_usage_mapped_correctly(self):
"""Input/output tokens from ResponseAPIUsage are mapped to AnthropicUsage."""
response = _make_mock_response(

View file

@ -413,6 +413,7 @@ def test_select_azure_base_url_called(setup_mocks):
"avector_store_create",
"avector_store_search",
"acreate_skill",
"acreate_interaction",
]
],
)

View file

@ -300,6 +300,7 @@ def test_azure_ai_strips_non_openai_spec_message_fields():
"cache_control": {"type": "ephemeral"},
}
],
"reasoning_content": "The user wants me to read a file.",
"provider_specific_fields": {"thought_signature": "sig-top"},
"tool_calls": [
{
@ -327,6 +328,7 @@ def test_azure_ai_strips_non_openai_spec_message_fields():
transformed_messages = request["messages"]
assert not _find_key_anywhere(transformed_messages, "thinking_blocks")
assert not _find_key_anywhere(transformed_messages, "reasoning_content")
assert not _find_key_anywhere(transformed_messages, "provider_specific_fields")
assert not _find_key_anywhere(transformed_messages, "cache_control")

View file

@ -0,0 +1,264 @@
import json
from decimal import Decimal
from pathlib import Path
from typing import Final
import pytest
import litellm
from litellm.llms.databricks.cost_calculator import cost_per_token
from litellm.types.utils import ModelInfo, Usage
REPO_ROOT: Final = Path(__file__).parents[4]
MAIN_PRICES: Final = REPO_ROOT / "model_prices_and_context_window.json"
BACKUP_PRICES: Final = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
NEW_MODELS: Final = (
"databricks/databricks-claude-opus-4-7",
"databricks/databricks-claude-opus-4-8",
"databricks/databricks-claude-opus-5",
"databricks/databricks-claude-sonnet-5",
"databricks/databricks-claude-fable-5",
)
DOLLARS_PER_DBU: Final = Decimal("0.070")
PRICE_FIELDS: Final = (
"input_cost_per_token",
"output_cost_per_token",
"cache_creation_input_token_cost",
"cache_read_input_token_cost",
)
PUBLISHED_DBU_PER_MILLION: Final = {
"databricks/databricks-claude-fable-5": ("142.858", "714.286", "178.572", "14.286"),
"databricks/databricks-claude-opus-5": ("71.429", "357.143", "89.286", "7.143"),
"databricks/databricks-claude-opus-4-8": ("71.429", "357.143", "89.286", "7.143"),
"databricks/databricks-claude-opus-4-7": ("71.429", "357.143", "89.286", "7.143"),
"databricks/databricks-claude-opus-4-6": ("71.429", "357.143", "89.286", "7.143"),
"databricks/databricks-claude-opus-4-5": ("71.429", "357.143", "89.286", "7.143"),
"databricks/databricks-claude-opus-4-1": ("214.286", "1071.429", "267.857", "21.429"),
"databricks/databricks-claude-opus-4": ("214.286", "1071.429", "267.857", "21.429"),
"databricks/databricks-claude-sonnet-5": ("42.857", "214.286", "53.571", "4.286"),
"databricks/databricks-claude-sonnet-4-6": ("42.857", "214.286", "53.571", "4.286"),
"databricks/databricks-claude-sonnet-4-5": ("42.857", "214.286", "53.571", "4.286"),
"databricks/databricks-claude-sonnet-4-1": ("42.857", "214.286", "53.571", "4.286"),
"databricks/databricks-claude-sonnet-4": ("42.857", "214.286", "53.571", "4.286"),
"databricks/databricks-claude-3-7-sonnet": ("42.857", "214.286", "53.571", "4.286"),
"databricks/databricks-claude-haiku-4-5": ("14.286", "71.429", "17.857", "1.429"),
"databricks/databricks-gpt-5": ("17.857", "142.857", "17.857", "1.786"),
"databricks/databricks-gpt-5-1": ("17.857", "142.857", "17.857", "1.786"),
"databricks/databricks-gpt-5-1-codex-max": ("17.857", "142.857", "17.857", "1.786"),
"databricks/databricks-gpt-5-1-codex-mini": ("3.571", "28.571", "3.571", "0.357"),
"databricks/databricks-gpt-5-mini": ("3.571", "28.571", "3.571", "0.357"),
"databricks/databricks-gpt-5-nano": ("0.714", "5.714", "0.714", "0.071"),
"databricks/databricks-gpt-5-2": ("25.000", "200.000", "25.000", "2.500"),
"databricks/databricks-gpt-5-2-codex": ("25.000", "200.000", "25.000", "2.500"),
"databricks/databricks-gpt-5-3-codex": ("25.000", "200.000", "25.000", "2.500"),
"databricks/databricks-gpt-5-4": ("35.714", "214.286", "35.714", "3.571"),
"databricks/databricks-gpt-5-4-mini": ("10.714", "64.286", "10.714", "1.071"),
"databricks/databricks-gpt-5-4-nano": ("2.857", "17.857", "2.857", "0.286"),
"databricks/databricks-gemini-3-1-pro": ("35.714", "214.286", "35.714", "3.571"),
"databricks/databricks-gemini-3-pro": ("35.714", "214.286", "35.714", "3.571"),
"databricks/databricks-gemini-3-flash": ("8.929", "53.571", "8.929", "0.893"),
"databricks/databricks-gemini-3-1-flash-lite": ("4.464", "26.786", "4.464", "0.446"),
"databricks/databricks-gemini-2-5-pro": ("22.321", "178.571", "22.321", "2.232"),
"databricks/databricks-gemini-2-5-flash": ("5.357", "44.643", "5.357", "0.536"),
}
PROMOTIONAL_DISCOUNT: Final = 0.80
PROMOTION_EXPIRES: Final = "2027-01-31"
ENTRIES_STORING_PROMOTIONAL_RATE: Final = (
"databricks/databricks-gemini-2-5-pro",
"databricks/databricks-gemini-2-5-flash",
)
ENTRIES_STORING_LIST_RATE_DESPITE_PROMOTION: Final = (
"databricks/databricks-gemini-3-1-pro",
"databricks/databricks-gemini-3-pro",
"databricks/databricks-gemini-3-flash",
"databricks/databricks-gemini-3-1-flash-lite",
)
CACHE_FIELDS: Final = ("cache_creation_input_token_cost", "cache_read_input_token_cost")
def _model_info(model: str) -> ModelInfo:
return litellm.get_model_info(model=model, custom_llm_provider="databricks")
def _dollars_per_token(dbu_per_million: str) -> float:
return float(Decimal(dbu_per_million) * DOLLARS_PER_DBU / Decimal(10) ** 6)
@pytest.mark.parametrize(
"model",
[
"databricks/databricks-claude-opus-4-8",
"databricks/databricks-claude-opus-5",
"databricks/databricks-claude-sonnet-5",
],
)
def test_cached_tokens_bill_at_cache_rates(local_model_cost_map: None, model: str) -> None:
info: Final = _model_info(model)
usage: Final = Usage(
prompt_tokens=11000,
completion_tokens=500,
total_tokens=11500,
cache_creation_input_tokens=2000,
cache_read_input_tokens=8000,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
assert prompt_cost == pytest.approx(
1000 * info["input_cost_per_token"]
+ 2000 * info["cache_creation_input_token_cost"]
+ 8000 * info["cache_read_input_token_cost"]
)
assert completion_cost == pytest.approx(500 * info["output_cost_per_token"])
assert prompt_cost < 11000 * info["input_cost_per_token"]
def test_uncached_request_bills_every_prompt_token_at_the_input_rate(local_model_cost_map: None) -> None:
model: Final = "databricks/databricks-claude-sonnet-5"
info: Final = _model_info(model)
usage: Final = Usage(prompt_tokens=1000, completion_tokens=200, total_tokens=1200)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
assert prompt_cost == pytest.approx(1000 * info["input_cost_per_token"])
assert completion_cost == pytest.approx(200 * info["output_cost_per_token"])
def test_legacy_endpoint_names_still_resolve(local_model_cost_map: None) -> None:
info: Final = _model_info("databricks/databricks-mixtral-8x7b-instruct")
usage: Final = Usage(prompt_tokens=100, completion_tokens=100, total_tokens=200)
prompt_cost, completion_cost = cost_per_token(model="databricks/mixtral-8x7b-instruct-v0.1", usage=usage)
assert prompt_cost == pytest.approx(100 * info["input_cost_per_token"])
assert completion_cost == pytest.approx(100 * info["output_cost_per_token"])
@pytest.mark.parametrize("model", NEW_MODELS)
def test_new_models_price_at_published_dbu_rates(local_model_cost_map: None, model: str) -> None:
info: Final = _model_info(model)
for field, dbu_per_million in zip(PRICE_FIELDS, PUBLISHED_DBU_PER_MILLION[model]):
assert info[field] == _dollars_per_token(dbu_per_million), field
@pytest.mark.parametrize("model", sorted(set(PUBLISHED_DBU_PER_MILLION) - set(ENTRIES_STORING_PROMOTIONAL_RATE)))
def test_cache_rates_derive_from_published_cache_dbu(local_model_cost_map: None, model: str) -> None:
info: Final = _model_info(model)
cache_dbu_per_million: Final = PUBLISHED_DBU_PER_MILLION[model][2:]
for field, dbu_per_million in zip(CACHE_FIELDS, cache_dbu_per_million):
assert info[field] == _dollars_per_token(dbu_per_million), field
@pytest.mark.parametrize("model", NEW_MODELS)
def test_new_models_carry_cache_pricing(local_model_cost_map: None, model: str) -> None:
info: Final = _model_info(model)
assert info["input_cost_per_token"] > 0
assert info["output_cost_per_token"] > 0
assert info["cache_creation_input_token_cost"] > info["input_cost_per_token"]
assert info["cache_read_input_token_cost"] < info["input_cost_per_token"]
assert info["supports_prompt_caching"] is True
def test_every_priced_databricks_model_declares_cache_rates(local_model_cost_map: None) -> None:
undeclared: Final = [
model
for model, info in litellm.model_cost.items()
if model.startswith("databricks/")
and info.get("input_cost_per_token") is not None
and any(info.get(field) is None for field in CACHE_FIELDS)
]
assert undeclared == []
def test_models_without_a_cache_discount_bill_cache_tokens_at_the_input_rate(
local_model_cost_map: None,
) -> None:
model: Final = "databricks/databricks-meta-llama-3-3-70b-instruct"
info: Final = _model_info(model)
usage: Final = Usage(
prompt_tokens=10000,
completion_tokens=100,
total_tokens=10100,
cache_read_input_tokens=8000,
)
prompt_cost, _ = cost_per_token(model=model, usage=usage)
assert prompt_cost == pytest.approx(10000 * info["input_cost_per_token"])
assert prompt_cost > 8000 * info["input_cost_per_token"]
def test_every_model_without_published_cache_dbu_bills_cache_at_its_own_input_rate(
local_model_cost_map: None,
) -> None:
without_published_rates: Final = [
model
for model, info in litellm.model_cost.items()
if model.startswith("databricks/")
and info.get("input_cost_per_token")
and model not in PUBLISHED_DBU_PER_MILLION
]
assert len(without_published_rates) == 14
for model in without_published_rates:
info = _model_info(model)
for field in CACHE_FIELDS:
assert info[field] == pytest.approx(info["input_cost_per_token"]), (model, field)
@pytest.mark.parametrize("model", NEW_MODELS)
def test_backup_price_map_matches_main(model: str) -> None:
main_cost: Final = json.loads(MAIN_PRICES.read_text())
backup_cost: Final = json.loads(BACKUP_PRICES.read_text())
assert model in main_cost
assert model in backup_cost
assert backup_cost[model] == main_cost[model]
def test_sonnet_5_ships_standard_rates_not_introductory(local_model_cost_map: None) -> None:
sonnet_5: Final = _model_info("databricks/databricks-claude-sonnet-5")
sonnet_4_6: Final = _model_info("databricks/databricks-claude-sonnet-4-6")
for field in PRICE_FIELDS:
assert sonnet_5[field] == pytest.approx(sonnet_4_6[field]), field
@pytest.mark.parametrize("model", ENTRIES_STORING_PROMOTIONAL_RATE)
def test_entries_storing_the_promotional_rate_price_below_the_published_table(
local_model_cost_map: None,
model: str,
) -> None:
info: Final = _model_info(model)
input_dbu, output_dbu, _, _ = PUBLISHED_DBU_PER_MILLION[model]
expiry_hint: Final = f"the gemini promotion expires {PROMOTION_EXPIRES}, after which the list rate applies"
assert info["input_cost_per_token"] == pytest.approx(
_dollars_per_token(input_dbu) * PROMOTIONAL_DISCOUNT, rel=2e-4
), expiry_hint
assert info["output_cost_per_token"] == pytest.approx(
_dollars_per_token(output_dbu) * PROMOTIONAL_DISCOUNT, rel=2e-4
), expiry_hint
assert info["cache_creation_input_token_cost"] == pytest.approx(info["input_cost_per_token"])
assert info["cache_read_input_token_cost"] == pytest.approx(0.1 * info["input_cost_per_token"])
@pytest.mark.parametrize("model", ENTRIES_STORING_LIST_RATE_DESPITE_PROMOTION)
def test_entries_storing_the_list_rate_bill_above_the_promotional_price(
local_model_cost_map: None,
model: str,
) -> None:
info: Final = _model_info(model)
input_dbu, _, _, _ = PUBLISHED_DBU_PER_MILLION[model]
list_rate: Final = _dollars_per_token(input_dbu)
assert info["input_cost_per_token"] == pytest.approx(list_rate, rel=2e-4), (
f"{model} moved off the list rate; if it now stores the discount that runs to "
f"{PROMOTION_EXPIRES}, move it into ENTRIES_STORING_PROMOTIONAL_RATE"
)
assert info["cache_creation_input_token_cost"] == pytest.approx(info["input_cost_per_token"])

View file

@ -473,12 +473,14 @@ def test_transform_messages_helper_strips_thinking_blocks():
"thinking_blocks": [
{"type": "thinking", "thinking": "internal", "signature": ""}
],
"reasoning_content": "internal",
},
]
out = config._transform_messages_helper(
messages, model="accounts/fireworks/models/glm-5p1", litellm_params={}
)
assert "thinking_blocks" not in out[1]
assert "reasoning_content" not in out[1]
assert out[1]["content"] == "I can help."

View file

@ -200,6 +200,7 @@ def test_hosted_vllm_thinking_blocks_prepended_to_assistant_content():
"signature": "abc123",
}
],
"reasoning_content": "Let me reason about this...",
},
{
"role": "user",
@ -218,6 +219,7 @@ def test_hosted_vllm_thinking_blocks_prepended_to_assistant_content():
assert isinstance(assistant_msg["content"], str)
assert assistant_msg["content"] == "Here is my answer."
assert "thinking_blocks" not in assistant_msg
assert "reasoning_content" not in assistant_msg
def test_hosted_vllm_thinking_blocks_with_list_content():

View file

@ -245,13 +245,21 @@ def _fake_get_async_httpx_client_factory(captured_calls: list):
return _fake_get_async_httpx_client
async def _fake_create_client(base_url, client_config=None, **kwargs):
async def _fake_create_client(agent_card, client_config=None, **kwargs):
client = MagicMock()
if client_config is not None:
client._litellm_httpx_client = client_config.httpx_client
return client
def _fake_card_resolver(httpx_client, base_url, **kwargs):
resolver = MagicMock()
card = MagicMock()
card.supported_interfaces = ()
resolver.get_agent_card = AsyncMock(return_value=card)
return resolver
@pytest.mark.asyncio
async def test_create_a2a_client_leaves_the_shared_client_untouched():
"""
@ -276,6 +284,10 @@ async def test_create_a2a_client_leaves_the_shared_client_untouched():
"litellm.a2a_protocol.main.create_client",
new=AsyncMock(side_effect=_fake_create_client),
),
patch(
"litellm.a2a_protocol.main.A2ACardResolver",
side_effect=_fake_card_resolver,
),
):
await create_a2a_client(
base_url="http://agent-a:9999",
@ -321,6 +333,10 @@ async def test_create_a2a_client_default_timeout_matches_constant():
"litellm.a2a_protocol.main.create_client",
new=AsyncMock(side_effect=_fake_create_client),
),
patch(
"litellm.a2a_protocol.main.A2ACardResolver",
side_effect=_fake_card_resolver,
),
):
await create_a2a_client(base_url="http://127.0.0.1:9")
@ -352,6 +368,10 @@ async def test_create_a2a_client_explicit_timeout_overrides_default():
"litellm.a2a_protocol.main.create_client",
new=AsyncMock(side_effect=_fake_create_client),
),
patch(
"litellm.a2a_protocol.main.A2ACardResolver",
side_effect=_fake_card_resolver,
),
):
await create_a2a_client(base_url="http://127.0.0.1:9", timeout=42.5)

View file

@ -18,6 +18,9 @@ from litellm.types.proxy.claude_code_endpoints import (
UpdatePluginRequest,
)
from litellm.proxy.anthropic_endpoints.claude_code_endpoints.claude_code_marketplace import (
delete_plugin,
disable_plugin,
enable_plugin,
get_marketplace,
register_plugin,
update_plugin,
@ -72,6 +75,12 @@ _USER = UserAPIKeyAuth(
user_id="test-user",
)
_NON_ADMIN_USER = UserAPIKeyAuth(
user_role=LitellmUserRoles.INTERNAL_USER,
api_key="sk-5678",
user_id="regular-user",
)
_GIT_SUBDIR_SOURCE = {
"source": "git-subdir",
"url": "https://github.com/org/monorepo.git",
@ -151,6 +160,7 @@ async def test_update_plugin_replaces_existing_source():
response = await update_plugin(
plugin_name=name,
request=UpdatePluginRequest(source=new_source, version="2.0.0", description="updated"),
user_api_key_dict=_USER,
)
assert response.status == "success"
@ -170,6 +180,7 @@ async def test_update_plugin_not_found():
await update_plugin(
plugin_name="does-not-exist",
request=UpdatePluginRequest(source=_GIT_SUBDIR_SOURCE),
user_api_key_dict=_USER,
)
assert exc_info.value.status_code == 404
@ -213,6 +224,7 @@ async def test_update_plugin_db_error_maps_to_structured_500():
await update_plugin(
plugin_name=name,
request=UpdatePluginRequest(source={"source": "github", "repo": "org/replacement"}),
user_api_key_dict=_USER,
)
assert exc_info.value.status_code == 500
@ -341,3 +353,62 @@ async def test_register_plugin_unknown_source_type():
assert exc_info.value.status_code == 400
assert "git-subdir" in exc_info.value.detail["error"]
@pytest.mark.asyncio
async def test_register_plugin_rejects_non_admin():
"""A non-admin key cannot add an entry to the marketplace catalog."""
request = RegisterPluginRequest(name="attacker-plugin", source=_GIT_SUBDIR_SOURCE)
with pytest.raises(HTTPException) as exc_info:
await register_plugin(request=request, user_api_key_dict=_NON_ADMIN_USER)
assert exc_info.value.status_code == 403
table = litellm.proxy.proxy_server.prisma_client.db.litellm_claudecodeplugintable
assert await table.find_unique(where={"name": "attacker-plugin"}) is None
@pytest.mark.asyncio
async def test_update_plugin_rejects_non_admin_overwrite():
"""A non-admin key cannot overwrite an existing plugin's source."""
name = "trusted-plugin"
await register_plugin(
request=RegisterPluginRequest(name=name, source=_GIT_SUBDIR_SOURCE, version="1.0.0"),
user_api_key_dict=_USER,
)
malicious_source = {"source": "github", "repo": "attacker/malicious-repo"}
with pytest.raises(HTTPException) as exc_info:
await update_plugin(
plugin_name=name,
request=UpdatePluginRequest(source=malicious_source),
user_api_key_dict=_NON_ADMIN_USER,
)
assert exc_info.value.status_code == 403
stored = await _read_stored_manifest(name)
assert stored["source"] == _GIT_SUBDIR_SOURCE
@pytest.mark.asyncio
async def test_enable_disable_delete_plugin_reject_non_admin():
"""Non-admin keys cannot enable, disable, or delete catalog entries."""
name = "trusted-plugin-2"
await register_plugin(
request=RegisterPluginRequest(name=name, source=_GIT_SUBDIR_SOURCE, version="1.0.0"),
user_api_key_dict=_USER,
)
for coro in (
enable_plugin(plugin_name=name, user_api_key_dict=_NON_ADMIN_USER),
disable_plugin(plugin_name=name, user_api_key_dict=_NON_ADMIN_USER),
delete_plugin(plugin_name=name, user_api_key_dict=_NON_ADMIN_USER),
):
with pytest.raises(HTTPException) as exc_info:
await coro
assert exc_info.value.status_code == 403
table = litellm.proxy.proxy_server.prisma_client.db.litellm_claudecodeplugintable
assert (await table.find_unique(where={"name": name})).enabled is True

View file

@ -731,6 +731,318 @@ async def test_track_cost_callback_skips_when_no_standard_logging_object():
mock_proxy_logging.failed_tracking_alert.assert_not_called()
@pytest.mark.asyncio
async def test_track_cost_callback_defers_in_progress_background_interaction(): # test-quality-ok: writing no spend row and raising no alert is the whole observable contract of the deferral path
"""
A background=true interaction create returns in_progress with no usage
block, so its success event has a model but no standard_logging_object.
The callback must skip quietly (billing happens later via the background
poll task) instead of raising 'Cost tracking failed' and alerting.
"""
from litellm.types.interactions import InteractionsAPIResponse
logger = _ProxyDBLogger()
kwargs = {
"call_type": "acreate_interaction",
"model": "gemini/gemini-3-flash-preview",
"litellm_call_id": "test-call-id",
"litellm_params": {},
"stream": False,
}
in_progress_response = InteractionsAPIResponse(
id="interactions/bg-abc",
model="gemini-3-flash-preview",
status="in_progress",
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
) as mock_proxy_logging:
mock_proxy_logging.failed_tracking_alert = AsyncMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
await logger._PROXY_track_cost_callback(
kwargs=kwargs,
completion_response=in_progress_response,
start_time=datetime.now(),
end_time=datetime.now(),
)
mock_proxy_logging.db_spend_update_writer.update_database.assert_not_called()
mock_proxy_logging.failed_tracking_alert.assert_not_called()
def _in_progress_interaction_kwargs(reservation: dict) -> dict:
return {
"call_type": "acreate_interaction",
"model": "gemini/gemini-3-flash-preview",
"litellm_call_id": "test-call-id",
"litellm_params": {"metadata": {"user_api_key_budget_reservation": reservation}},
"stream": False,
}
@pytest.mark.asyncio
@pytest.mark.parametrize("status", ["in_progress", "queued"])
async def test_track_cost_callback_keeps_reservation_open_for_in_progress_background_interaction(status):
"""
The pre-call budget reservation must stay open while a background
interaction is in flight, so concurrent creates cannot stack past the
budget; the poll task's completion event reconciles it to the actual cost.
``queued`` is in flight for the same reason ``in_progress`` is: it has not
reached a terminal status, so releasing its reservation here would drop the
estimate off the spend counters while the interaction is still going to run
and still going to cost money.
"""
from litellm.types.interactions import InteractionsAPIResponse
logger = _ProxyDBLogger()
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
in_progress_response = InteractionsAPIResponse(
id="interactions/bg-abc",
model="gemini-3-flash-preview",
status=status,
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
) as mock_proxy_logging:
mock_proxy_logging.failed_tracking_alert = AsyncMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
await logger._PROXY_track_cost_callback(
kwargs=_in_progress_interaction_kwargs(reservation),
completion_response=in_progress_response,
start_time=datetime.now(),
end_time=datetime.now(),
)
assert reservation["finalized"] is False
mock_proxy_logging.failed_tracking_alert.assert_not_called()
@pytest.mark.asyncio
async def test_track_cost_callback_releases_reservation_for_in_progress_interaction_when_polling_disabled(
monkeypatch,
):
"""
With the poll task kill switch off nothing will ever reconcile the
reservation, so the callback must release it or the spend counters stay
pinned at the estimated cost forever.
"""
import litellm.proxy.hooks.proxy_track_cost_callback as callback_module
from litellm.types.interactions import InteractionsAPIResponse
monkeypatch.setattr(callback_module, "BACKGROUND_INTERACTION_COST_POLLING_ENABLED", False)
logger = _ProxyDBLogger()
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
in_progress_response = InteractionsAPIResponse(
id="interactions/bg-abc",
model="gemini-3-flash-preview",
status="in_progress",
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
) as mock_proxy_logging:
mock_proxy_logging.failed_tracking_alert = AsyncMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
await logger._PROXY_track_cost_callback(
kwargs=_in_progress_interaction_kwargs(reservation),
completion_response=in_progress_response,
start_time=datetime.now(),
end_time=datetime.now(),
)
assert reservation["finalized"] is True
mock_proxy_logging.failed_tracking_alert.assert_not_called()
@pytest.mark.asyncio
@pytest.mark.parametrize(
"status",
["failed", "cancelled", "incomplete", "budget_exceeded"],
)
async def test_track_cost_callback_releases_reservation_for_unpollable_interaction(status):
"""
Only an in-progress create gets a poll task, so a create that comes back
terminal with no usage has nobody left to reconcile its reservation. The
callback must release it there and then, or the pre-call estimate stays
added to the key, user, team and org spend counters and starts refusing
traffic against budget that was never actually spent.
None of these statuses produced output, so their missing usage is a normal
outcome rather than a cost-tracking failure, and the callback must not fire
``failed_tracking_alert``: doing so would flood operators with false alerts
and mask real cost-tracking failures.
"""
from litellm.types.interactions import InteractionsAPIResponse
logger = _ProxyDBLogger()
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
terminal_response = InteractionsAPIResponse(
id="interactions/bg-abc",
model="gemini-3-flash-preview",
status=status,
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
) as mock_proxy_logging:
mock_proxy_logging.failed_tracking_alert = AsyncMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
await logger._PROXY_track_cost_callback(
kwargs=_in_progress_interaction_kwargs(reservation),
completion_response=terminal_response,
start_time=datetime.now(),
end_time=datetime.now(),
)
assert reservation["finalized"] is True
mock_proxy_logging.failed_tracking_alert.assert_not_called()
@pytest.mark.asyncio
@pytest.mark.parametrize("status", ["completed", "requires_action"])
async def test_track_cost_callback_alerts_when_an_interaction_that_produced_output_has_no_usage(status):
"""
``completed`` and ``requires_action`` both mean the model produced output,
so a usage block is always expected with them. One arriving without it
means the charge for real work was lost, which is exactly what the
cost-tracking alert is for: silencing it here would let an operator's
interactions bill nothing with no signal that anything went wrong.
The reservation still has to be released, since suppressing the alert was
never what freed it.
"""
from litellm.types.interactions import InteractionsAPIResponse
logger = _ProxyDBLogger()
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
usageless_response = InteractionsAPIResponse(
id="interactions/bg-abc",
model="gemini-3-flash-preview",
status=status,
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
) as mock_proxy_logging:
mock_proxy_logging.failed_tracking_alert = AsyncMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
await logger._PROXY_track_cost_callback(
kwargs=_in_progress_interaction_kwargs(reservation),
completion_response=usageless_response,
start_time=datetime.now(),
end_time=datetime.now(),
)
assert reservation["finalized"] is True
mock_proxy_logging.failed_tracking_alert.assert_called_once()
@pytest.mark.asyncio
async def test_track_cost_callback_releases_reservation_for_interaction_without_an_id():
"""
The scheduler also refuses a response with no id, since it has nothing to
poll for, so the callback must not defer to a poll task that will never
exist, and it must not fire ``failed_tracking_alert`` for what is a
legitimate no-usage response rather than a cost-tracking failure.
"""
from litellm.types.interactions import InteractionsAPIResponse
logger = _ProxyDBLogger()
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
idless_response = InteractionsAPIResponse(
id="",
model="gemini-3-flash-preview",
status="in_progress",
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
) as mock_proxy_logging:
mock_proxy_logging.failed_tracking_alert = AsyncMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
await logger._PROXY_track_cost_callback(
kwargs=_in_progress_interaction_kwargs(reservation),
completion_response=idless_response,
start_time=datetime.now(),
end_time=datetime.now(),
)
assert reservation["finalized"] is True
mock_proxy_logging.failed_tracking_alert.assert_not_called()
@pytest.mark.asyncio
async def test_callback_handles_every_status_the_interactions_api_can_return():
"""
Whatever status a usage-less create comes back with, exactly one of two
things has to happen to its budget reservation: the callback holds it open
for a poll task that will settle it, or it releases it on the spot. A
status that falls through both leaves the pre-call estimate pinned to the
key, user, team and org spend counters forever, refusing traffic against
budget nobody spent.
Driven off the generated spec enum so a status Google adds later fails here
instead of quietly leaking reservations in production.
"""
from litellm.types.interactions import InteractionsAPIResponse
from litellm.types.interactions.generated import Status1
deferred = set()
released = set()
for status in sorted(member.value for member in Status1):
logger = _ProxyDBLogger()
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
response = InteractionsAPIResponse(
id="interactions/bg-abc",
model="gemini-3-flash-preview",
status=status,
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
) as mock_proxy_logging:
mock_proxy_logging.failed_tracking_alert = AsyncMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
await logger._PROXY_track_cost_callback(
kwargs=_in_progress_interaction_kwargs(reservation),
completion_response=response,
start_time=datetime.now(),
end_time=datetime.now(),
)
(deferred if reservation["finalized"] is False else released).add(status)
assert deferred == {"in_progress", "queued"}
assert released == {
"completed",
"requires_action",
"failed",
"cancelled",
"incomplete",
"budget_exceeded",
}
@pytest.mark.asyncio
async def test_async_post_call_failure_hook_propagates_trace_id_from_logging_obj():
"""

View file

@ -9,7 +9,7 @@ import litellm
from litellm.litellm_core_utils.health_check_helpers import HealthCheckHelpers
from litellm.proxy import health_check as hc_module
from litellm.proxy.health_check import (
_is_semantic_auto_router_deployment,
_is_strategy_router_deployment,
_resolve_health_check_max_tokens,
_resolve_health_check_mode,
_update_litellm_params_for_health_check,
@ -499,33 +499,22 @@ def test_autodetected_embedding_skips_reasoning_effort():
assert "max_tokens" not in updated
# ---------------------------------------------------------------------------
# auto_router (semantic router) deployments must be skipped by health checks.
#
# These are meta-routers that select among real LLM deployments at request
# time. They have no LLM endpoint to probe. Before this fix, the health check
# passed model="auto_router/router_1" to get_llm_provider(), which raised
# BadRequestError: "Unmapped LLM provider for this endpoint" because
# auto_router is not a real LLM provider.
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
"model, expected",
[
("auto_router/router_1", True),
("auto_router/my_router", True),
("auto_router/complexity_router", False),
("auto_router/adaptive_router", False),
("auto_router/quality_router", False),
("auto_router/adaptive_router/subpath", False),
("auto_router/complexity_router", True),
("auto_router/adaptive_router", True),
("auto_router/quality_router", True),
("auto_router/adaptive_router/subpath", True),
("gpt-4", False),
("openai/gpt-4", False),
("bedrock/claude", False),
],
)
def test_is_semantic_auto_router_deployment(model, expected):
assert _is_semantic_auto_router_deployment({"model": model}) == expected
def test_is_strategy_router_deployment(model, expected):
assert _is_strategy_router_deployment({"model": model}) == expected
@pytest.mark.asyncio
@ -676,3 +665,22 @@ async def test_bedrock_invoke_body_has_no_media_source_without_health_check_para
body = await _pegasus_health_check_request_body({"mode": "chat"}, monkeypatch)
assert "mediaSource" not in body
@pytest.mark.asyncio
async def test_run_model_health_check_skips_complexity_router_deployment():
fake_ahealth_check = AsyncMock(return_value={})
model = {
"litellm_params": {
"model": "auto_router/complexity_router",
"complexity_router_config": {"tiers": {"simple": "gpt-4o-mini"}},
"complexity_router_default_model": "gpt-4o-mini",
},
"model_info": {},
}
with patch.object(hc_module.litellm, "ahealth_check", fake_ahealth_check):
result = await hc_module._run_model_health_check(model)
fake_ahealth_check.assert_not_called()
assert result == {}

View file

@ -114,3 +114,46 @@ def test_assistant_message_after_tool_call_is_folded_into_it():
tool_call_idx = next(i for i, m in enumerate(msgs) if isinstance(m, dict) and m.get("tool_calls"))
assert msgs[tool_call_idx].get("role") == "assistant"
assert msgs[tool_call_idx + 1].get("role") == "tool"
def test_assistant_message_before_function_call_keeps_one_assistant_turn():
"""The chat->responses bridge emits an assistant message ahead of its function_call.
Round-tripping that order back to chat must fold both into a single assistant
turn, so the tool result still follows the message that made the call.
"""
msgs = LiteLLMCompletionResponsesConfig._transform_response_input_param_to_chat_completion_message(
input=[
{
"role": "user",
"type": "message",
"content": [{"type": "input_text", "text": "What is the weather?"}],
},
{
"role": "assistant",
"type": "message",
"content": [{"type": "output_text", "text": "Let me check."}],
},
{
"type": "function_call",
"name": "get_weather",
"call_id": "call_1",
"arguments": "{}",
},
{
"type": "function_call_output",
"call_id": "call_1",
"output": "sunny",
},
]
)
assistant_msgs = [m for m in msgs if isinstance(m, dict) and m.get("role") == "assistant"]
assert len(assistant_msgs) == 1
assistant = assistant_msgs[0]
assert assistant["content"] == [{"type": "text", "text": "Let me check."}]
assert [tc["function"]["name"] for tc in assistant["tool_calls"]] == ["get_weather"]
assistant_idx = msgs.index(assistant)
assert msgs[assistant_idx + 1].get("role") == "tool"
assert msgs[assistant_idx + 1].get("tool_call_id") == "call_1"

View file

@ -3584,6 +3584,104 @@ def test_batch_cost_calculator_cache_creation_falls_back_to_input_rate():
assert prompt_cost == pytest.approx((1000 * 3e-6 + 8000 * 3e-7 + 2000 * 3e-6) / 2)
def test_completion_cost_bills_interactions_api_response():
from litellm.types.interactions import InteractionsAPIResponse
model_info = litellm.get_model_info(model="gemini-2.5-flash", custom_llm_provider="gemini")
response = InteractionsAPIResponse(
id="interactions/abc123",
model="gemini-2.5-flash",
status="completed",
steps=[],
usage={
"total_tokens": 175,
"total_input_tokens": 100,
"input_tokens_by_modality": [{"modality": "text", "tokens": 100}],
"total_cached_tokens": 0,
"total_output_tokens": 50,
"output_tokens_by_modality": [{"modality": "text", "tokens": 50}],
"total_tool_use_tokens": 0,
"total_thought_tokens": 25,
},
)
cost = completion_cost(completion_response=response, custom_llm_provider="gemini")
reasoning_rate = model_info.get("output_cost_per_reasoning_token") or model_info["output_cost_per_token"]
expected = (
100 * model_info["input_cost_per_token"]
+ 50 * model_info["output_cost_per_token"]
+ 25 * reasoning_rate
)
assert cost == pytest.approx(expected)
assert cost > 0
def test_completion_cost_bills_interactions_google_search_per_query():
from litellm.types.interactions import InteractionsAPIResponse
model_info = litellm.get_model_info(model="gemini-3-flash-preview", custom_llm_provider="gemini")
response = InteractionsAPIResponse(
id="interactions/search123",
model="gemini-3-flash-preview",
status="completed",
steps=[],
usage={
"total_tokens": 680,
"total_input_tokens": 103,
"input_tokens_by_modality": [{"modality": "text", "tokens": 103}],
"total_cached_tokens": 0,
"total_output_tokens": 226,
"total_tool_use_tokens": 0,
"total_thought_tokens": 351,
"grounding_tool_count": [{"type": "google_search", "count": 3}],
},
)
cost = completion_cost(completion_response=response, custom_llm_provider="gemini")
per_query_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"]
reasoning_rate = model_info.get("output_cost_per_reasoning_token") or model_info["output_cost_per_token"]
expected = (
103 * model_info["input_cost_per_token"]
+ 226 * model_info["output_cost_per_token"]
+ 351 * reasoning_rate
+ 3 * per_query_cost
)
assert model_info.get("web_search_billing_unit") == "per_query"
assert cost == pytest.approx(expected)
assert cost > 3 * per_query_cost
def test_completion_cost_bills_interactions_video_output_at_video_rate():
from litellm.types.interactions import InteractionsAPIResponse
model_info = litellm.get_model_info(model="gemini-omni-flash-preview", custom_llm_provider="gemini")
video_tokens = 5792 * 8
response = InteractionsAPIResponse(
id="interactions/video123",
model="gemini-omni-flash-preview",
status="completed",
steps=[],
usage={
"total_tokens": 10 + video_tokens,
"total_input_tokens": 10,
"input_tokens_by_modality": [{"modality": "text", "tokens": 10}],
"total_cached_tokens": 0,
"total_output_tokens": video_tokens,
"output_tokens_by_modality": [{"modality": "video", "tokens": video_tokens}],
"total_tool_use_tokens": 0,
"total_thought_tokens": 0,
},
)
cost = completion_cost(completion_response=response, custom_llm_provider="gemini")
expected = 10 * model_info["input_cost_per_token"] + video_tokens * model_info["output_cost_per_video_token"]
assert model_info["output_cost_per_video_token"] != model_info["output_cost_per_token"]
assert cost == pytest.approx(expected)
@pytest.mark.parametrize(
"batch_rate,expected_prompt,expected_completion",
[

View file

@ -165,6 +165,20 @@ describe("OrganizationsTable", () => {
expect(screen.getByText("RPM: Unlimited")).toBeInTheDocument();
});
it("renders a tpm/rpm limit of 0 as 0, never as Unlimited", () => {
render(
<OrganizationsTable
{...baseProps}
organizations={[makeOrganization({ litellm_budget_table: { max_budget: null, tpm_limit: 0, rpm_limit: 0 } })]}
/>,
);
expect(screen.getByText("TPM: 0")).toBeInTheDocument();
expect(screen.getByText("RPM: 0")).toBeInTheDocument();
expect(screen.queryByText("TPM: Unlimited")).not.toBeInTheDocument();
expect(screen.queryByText("RPM: Unlimited")).not.toBeInTheDocument();
});
it("renders loading skeletons instead of rows while loading", () => {
render(
<OrganizationsTable

View file

@ -28,8 +28,8 @@ function OrganizationLimitsCell({ organization }: { organization: Organization }
const { tpm_limit, rpm_limit } = getOrganizationBudget(organization);
return (
<div className="flex flex-col text-xs text-muted-foreground">
<span>TPM: {tpm_limit ? tpm_limit : "Unlimited"}</span>
<span>RPM: {rpm_limit ? rpm_limit : "Unlimited"}</span>
<span>TPM: {tpm_limit ?? "Unlimited"}</span>
<span>RPM: {rpm_limit ?? "Unlimited"}</span>
</div>
);
}

View file

@ -87,6 +87,21 @@ describe("ChatMessageBubble", () => {
expect(screen.getByText("Hi there")).toBeInTheDocument();
});
it.each([
{ role: "user" as const, bubble: ["bg-info/10", "border-info/20"], avatar: "bg-info/20" },
{ role: "assistant" as const, bubble: ["bg-card", "border-border"], avatar: "bg-muted" },
])("should paint the $role surface from theme tokens, not fixed colours", ({ role, bubble, avatar }) => {
render(<ChatMessageBubble {...defaultProps} message={{ role, content: "Hello" }} />);
const header = screen.getByText(role).closest("div") as HTMLElement;
const surface = header.parentElement as HTMLElement;
expect(surface).toHaveClass(...bubble);
expect(surface).not.toHaveAttribute("style");
expect(header.firstElementChild).toHaveClass(avatar);
expect(header.firstElementChild).not.toHaveAttribute("style");
});
it("should show model badge for assistant messages when model is provided", () => {
render(<ChatMessageBubble {...defaultProps} message={{ role: "assistant", content: "Reply", model: "gpt-4" }} />);

View file

@ -46,20 +46,16 @@ function ChatMessageBubble({
return (
<div className={`mb-4 min-w-0 ${isUser ? "text-right" : "text-left"}`}>
<div
className="inline-block min-w-0 max-w-[92%] overflow-hidden rounded-lg p-3 shadow-xs sm:max-w-[85%] sm:px-4"
style={{
backgroundColor: isUser ? "#f0f8ff" : "#ffffff",
border: isUser ? "1px solid #e6f0fa" : "1px solid #f0f0f0",
textAlign: "left",
}}
className={`inline-block min-w-0 max-w-[92%] overflow-hidden rounded-lg border p-3 text-left text-card-foreground shadow-xs sm:max-w-[85%] sm:px-4 ${
isUser ? "border-info/20 bg-info/10" : "border-border bg-card"
}`}
>
{/* Header: role icon + name + model badge */}
<div className="mb-1.5 flex min-w-0 items-center gap-2">
<div
className="flex items-center justify-center w-6 h-6 rounded-full mr-1"
style={{
backgroundColor: isUser ? "#e6f0fa" : "#f5f5f5",
}}
className={`flex items-center justify-center w-6 h-6 rounded-full mr-1 ${
isUser ? "bg-info/20" : "bg-muted"
}`}
>
{isUser ? (
<User className="size-3 text-info" aria-hidden="true" />

View file

@ -1784,19 +1784,9 @@ const ChatUI: React.FC<ChatUIProps> = ({
chatHistory.length > 0 &&
chatHistory[chatHistory.length - 1].role === "user" && (
<div className="mb-4 text-left">
<div
className="inline-block max-w-[80%] rounded-lg p-3.5 px-4 shadow-xs"
style={{
backgroundColor: "#ffffff",
border: "1px solid #f0f0f0",
textAlign: "left",
}}
>
<div className="inline-block max-w-[80%] rounded-lg border border-border bg-card p-3.5 px-4 text-left text-card-foreground shadow-xs">
<div className="mb-1.5 flex items-center gap-2">
<div
className="mr-1 flex h-6 w-6 items-center justify-center rounded-full"
style={{ backgroundColor: "#f5f5f5" }}
>
<div className="mr-1 flex h-6 w-6 items-center justify-center rounded-full bg-muted">
<Bot className="size-3 text-muted-foreground" aria-hidden="true" />
</div>
<strong className="text-sm capitalize">Assistant</strong>

View file

@ -8,6 +8,7 @@ import { KeyResponse, Team } from "../key_team_helpers/key_list";
import { useKeyInfo } from "@/app/(dashboard)/hooks/keys/useKeyInfo";
import { KeysResponse, useKeys } from "@/app/(dashboard)/hooks/keys/useKeys";
import useTeams from "@/app/(dashboard)/hooks/useTeams";
import { regenerateKeyCall } from "../networking";
// Resolve debounced values synchronously so an applied filter lands in the useKeys query within the test tick.
vi.mock("@tanstack/react-pacer/debouncer", async () => {
@ -25,6 +26,11 @@ vi.mock("@tanstack/react-pacer/debouncer", async () => {
vi.mock("next/navigation", () => ({ useRouter: () => ({ push: vi.fn() }) }));
vi.mock("../networking", async (importOriginal) => ({
...(await importOriginal<typeof import("../networking")>()),
regenerateKeyCall: vi.fn(),
}));
vi.mock("@/app/(dashboard)/hooks/useAuthorized", () => ({
default: vi.fn(() => ({
accessToken: "test-token",
@ -389,6 +395,28 @@ it("renders KeyInfoView when the URL has ?key= for a key on the current page, wi
expect(screen.getByTestId("pagination-range")).toBeInTheDocument();
});
it("repoints ?key= to the rotated hash once the regenerate dialog is dismissed", async () => {
const user = userEvent.setup();
vi.mocked(regenerateKeyCall).mockResolvedValue({
key: "sk-rotated-plaintext",
token: null,
token_id: "rotated-hash-456",
});
const onUrlUpdate = vi.fn<OnUrlUpdateFunction>();
renderWithProviders(<VirtualKeysTable />, { searchParams: { key: mockKey.token }, onUrlUpdate });
await user.click(await screen.findByRole("button", { name: /regenerate key/i }));
await user.click(await screen.findByRole("button", { name: /^Regenerate$/ }));
expect(await screen.findAllByText("sk-rotated-plaintext")).not.toHaveLength(0);
expect(lastKeyParam(onUrlUpdate)).toBeUndefined();
await user.click(screen.getAllByRole("button", { name: "Close" })[0]);
await waitFor(() => {
expect(lastKeyParam(onUrlUpdate)).toBe("rotated-hash-456");
});
});
it("fetches the key by id when the URL has ?key= for a key not in the loaded page", async () => {
mockUseKeyInfo.mockReturnValue(
keyInfoResult({ ...mockKey, token: "other-key-hash", key_alias: "Fetched Key Alias" }),

View file

@ -20,7 +20,7 @@ import { KeyRound } from "lucide-react";
import { parseAsString, useQueryState } from "nuqs";
import React, { useCallback, useMemo, useState } from "react";
import { Team } from "../key_team_helpers/key_list";
import { KeyResponse, Team } from "../key_team_helpers/key_list";
import KeyInfoView from "../templates/key_info_view";
import { getKeyTableColumns, KEY_TABLE_HIDDEN_COLUMNS } from "./keyTableColumns";
@ -139,6 +139,16 @@ export function VirtualKeysTable({ headerActions }: VirtualKeysTableProps) {
[organizations],
);
const handleSelectedKeyDataUpdate = useCallback(
(updated: Partial<KeyResponse>) => {
const rotatedToken = updated.token ?? updated.token_id;
if (!rotatedToken || rotatedToken === selectedKeyId) return;
void setSelectedKeyId(rotatedToken);
void refetch();
},
[refetch, selectedKeyId, setSelectedKeyId],
);
const formatFilterValue = useCallback(
(columnId: string, value: unknown): string => {
const raw = String(value);
@ -165,6 +175,7 @@ export function VirtualKeysTable({ headerActions }: VirtualKeysTableProps) {
keyData={selectedKey}
teams={allTeams}
onDelete={refetch}
onKeyDataUpdate={handleSelectedKeyDataUpdate}
/>
</div>
);

View file

@ -34,6 +34,8 @@ const modelMappingsRule = {
},
};
const tooltipCodeClassName = "rounded-sm bg-background/20 px-1 py-0.5 font-mono text-xs";
const ConditionalPublicModelName: React.FC = () => {
const form = useFormContext<MountedFormValues>();
@ -123,22 +125,22 @@ const ConditionalPublicModelName: React.FC = () => {
if (!showPublicModelName) return null;
const publicNameTooltipContent = (
<>
<div className="mb-2 font-normal">The name you specify in your API calls to LiteLLM Proxy</div>
<div className="mb-2 font-normal">
<div className="flex flex-col gap-2 text-left font-normal">
<div>The name you specify in your API calls to LiteLLM Proxy</div>
<div>
<strong>Example:</strong> If you name your public model{" "}
<code className="bg-muted px-1 py-0.5 rounded-sm text-xs">example-name</code>, and choose{" "}
<code className="bg-muted px-1 py-0.5 rounded-sm text-xs">openai/qwen-plus-latest</code> as the LiteLLM model
<code className={tooltipCodeClassName}>example-name</code>, and choose{" "}
<code className={tooltipCodeClassName}>openai/qwen-plus-latest</code> as the LiteLLM model
</div>
<div className="mb-2 font-normal">
<div>
<strong>Usage:</strong> You make an API call to the LiteLLM proxy with{" "}
<code className="bg-muted px-1 py-0.5 rounded-sm text-xs">model = &quot;example-name&quot;</code>
<code className={tooltipCodeClassName}>model = &quot;example-name&quot;</code>
</div>
<div className="font-normal">
<strong>Result:</strong> LiteLLM sends{" "}
<code className="bg-muted px-1 py-0.5 rounded-sm text-xs">qwen-plus-latest</code> to the provider
<div>
<strong>Result:</strong> LiteLLM sends <code className={tooltipCodeClassName}>qwen-plus-latest</code> to the
provider
</div>
</>
</div>
);
const liteLLMModelTooltipContent = <div>The model name LiteLLM will send to the LLM API</div>;

View file

@ -19,13 +19,13 @@ import {
isValidSubPath,
buildMarketplaceSettingsSnippet,
} from "./helpers";
import { MarketplacePluginEntry, PluginSource } from "./types";
import { MarketplacePluginEntry } from "./types";
describe("buildMarketplaceSettingsSnippet", () => {
it("nests the url under a source object so Claude Code accepts the marketplace", () => {
expect(JSON.parse(buildMarketplaceSettingsSnippet("https://proxy.example.com"))).toEqual({
extraKnownMarketplaces: {
"my-org": {
litellm: {
source: {
source: "url",
url: "https://proxy.example.com/claude-code/marketplace.json",
@ -37,28 +37,12 @@ describe("buildMarketplaceSettingsSnippet", () => {
});
describe("formatInstallCommand", () => {
it("formats github source with repo", () => {
const source: PluginSource = { source: "github", repo: "org/repo" };
expect(formatInstallCommand({ name: "my-plugin", source })).toBe("/plugin marketplace add org/repo");
it("produces a /plugin install command scoped to the litellm marketplace", () => {
expect(formatInstallCommand({ name: "my-plugin" })).toBe("/plugin install my-plugin@litellm");
});
it("formats url source", () => {
const source: PluginSource = { source: "url", url: "https://example.com/plugin" };
expect(formatInstallCommand({ name: "my-plugin", source })).toBe(
"/plugin marketplace add https://example.com/plugin",
);
});
it("formats git-subdir source using its url", () => {
const source: PluginSource = { source: "git-subdir", url: "https://github.com/org/repo", path: "plugins/x" };
expect(formatInstallCommand({ name: "my-plugin", source })).toBe(
"/plugin marketplace add https://github.com/org/repo",
);
});
it("falls back to plugin name when no repo or url", () => {
const source: PluginSource = { source: "github" };
expect(formatInstallCommand({ name: "my-plugin", source })).toBe("/plugin marketplace add my-plugin");
it("uses the plugin name as the identifier", () => {
expect(formatInstallCommand({ name: "code-review" })).toBe("/plugin install code-review@litellm");
});
});

View file

@ -179,13 +179,14 @@ export const parseSkillSource = (rawUrl: string, subPath?: string): SkillSourceP
/**
* Build the `~/.claude/settings.json` snippet that registers the proxy as a marketplace.
* Claude Code expects `extraKnownMarketplaces.<name>.source` to be a source object, not a
* bare `"url"` string, so the url/source pair is nested one level deeper.
* bare `"url"` string, so the url/source pair is nested one level deeper. The key must be
* "litellm" to match the name the proxy returns in marketplace.json.
*/
export const buildMarketplaceSettingsSnippet = (proxyOrigin: string): string =>
JSON.stringify(
{
extraKnownMarketplaces: {
"my-org": {
litellm: {
source: {
source: "url",
url: `${proxyOrigin}/claude-code/marketplace.json`,
@ -198,20 +199,10 @@ export const buildMarketplaceSettingsSnippet = (proxyOrigin: string): string =>
);
/**
* Generate install command for Claude Code CLI
* Format: /plugin marketplace add org/repo OR /plugin marketplace add url
* Generate install command for Claude Code CLI.
* Installs the named plugin from the "litellm" marketplace registered in settings.json.
*/
export const formatInstallCommand = (plugin: { name: string; source: PluginSource }): string => {
const { source } = plugin;
if (source.source === "github" && source.repo) {
return `/plugin marketplace add ${source.repo}`;
}
if ((source.source === "url" || source.source === "git-subdir") && source.url) {
return `/plugin marketplace add ${source.url}`;
}
// Fallback to plugin name
return `/plugin marketplace add ${plugin.name}`;
};
export const formatInstallCommand = (plugin: { name: string }): string => `/plugin install ${plugin.name}@litellm`;
/**
* Extract unique categories from plugins list

View file

@ -261,6 +261,32 @@ const SkillDetail: React.FC<SkillDetailProps> = ({ skill, onBack }) => {
</pre>
</div>
{/* Shown when the marketplace catalog is stale and the plugin isn't found yet */}
<div
style={{
border: "1px solid #fce8b2",
borderRadius: 8,
padding: "12px 16px",
backgroundColor: "#fefce8",
marginBottom: 16,
}}
>
<p style={{ fontSize: 13, color: "#5f6368", lineHeight: 1.6, margin: "0 0 8px 0" }}>
If you see &quot;Plugin {skill.name} not found in marketplace&quot;, update the catalog first:
</p>
<pre
style={{
margin: 0,
fontSize: 13,
fontFamily: "monospace",
color: "#202124",
backgroundColor: "transparent",
}}
>
/plugin marketplace update litellm
</pre>
</div>
<p style={{ fontSize: 13, color: "#5f6368", lineHeight: 1.6, margin: 0 }}>
Don&apos;t have the marketplace configured yet?{" "}
<span onClick={() => setActiveTab("setup")} style={{ color: "#1a73e8", cursor: "pointer" }}>
@ -276,12 +302,73 @@ const SkillDetail: React.FC<SkillDetailProps> = ({ skill, onBack }) => {
<h2 style={{ fontSize: 18, fontWeight: 400, color: "#202124", margin: "0 0 8px 0" }}>
One-time marketplace setup
</h2>
<p style={{ fontSize: 14, color: "#5f6368", margin: "0 0 24px 0", lineHeight: 1.6 }}>
Add this to{" "}
{/* Option 1: single command — fastest path for most users */}
<p style={{ fontSize: 14, color: "#5f6368", margin: "0 0 12px 0", lineHeight: 1.6 }}>
Run this command in Claude Code to register the marketplace:
</p>
<div
style={{
border: "1px solid #dadce0",
borderRadius: 8,
overflow: "hidden",
marginBottom: 24,
}}
>
<div
style={{
display: "flex",
alignItems: "center",
justifyContent: "space-between",
padding: "10px 16px",
backgroundColor: "#f8f9fa",
borderBottom: "1px solid #dadce0",
}}
>
<span style={{ fontSize: 13, color: "#3c4043", fontWeight: 500 }}>Run in Claude Code</span>
<button
onClick={() => {
const origin = typeof window !== "undefined" ? window.location.origin : "";
copyToClipboard(`/plugin marketplace add ${origin}/claude-code/marketplace.json`, "marketplace-cmd");
}}
style={{
display: "flex",
alignItems: "center",
gap: 4,
fontSize: 12,
color: copiedKey === "marketplace-cmd" ? "#137333" : "#1a73e8",
background: "none",
border: "none",
cursor: "pointer",
padding: 0,
}}
>
{copiedKey === "marketplace-cmd" ? <CheckOutlined /> : <CopyOutlined />}
{copiedKey === "marketplace-cmd" ? "Copied" : "Copy"}
</button>
</div>
<pre
style={{
margin: 0,
padding: "14px 16px",
fontSize: 13,
fontFamily: "monospace",
color: "#202124",
backgroundColor: "#fff",
}}
>
{`/plugin marketplace add ${typeof window !== "undefined" ? window.location.origin : "<proxy-url>"}/claude-code/marketplace.json`}
</pre>
</div>
{/* Option 2: settings.json for persistent config or managed deployments.
extraKnownMarketplaces requires source to be a nested object, not a flat string. */}
<p style={{ fontSize: 14, color: "#5f6368", margin: "0 0 12px 0", lineHeight: 1.6 }}>
Or add this to{" "}
<code style={{ fontSize: 13, backgroundColor: "#f1f3f4", padding: "1px 6px", borderRadius: 4 }}>
~/.claude/settings.json
</code>{" "}
to point Claude Code at your proxy:
for a persistent configuration:
</p>
<div
style={{

View file

@ -427,4 +427,21 @@ describe("RegenerateKeyModal", () => {
);
});
});
it("should report the rotated hash from token_id when the API leaves token null", async () => {
const user = userEvent.setup();
mockRegenerateKeyCall.mockResolvedValue({
key: "sk-new-regenerated-key",
token: null,
token_id: "rotated-hash-456",
});
renderWithProviders(<RegenerateKeyModal {...defaultProps} />);
await user.click(screen.getByRole("button", { name: /Regenerate/ }));
await waitFor(() => {
expect(mockOnKeyUpdate).toHaveBeenCalledOnce();
});
expect(mockOnKeyUpdate.mock.calls[0][0].token).toBe("rotated-hash-456");
});
});

View file

@ -107,7 +107,7 @@ export function RegenerateKeyModal({ selectedToken, visible, onClose, onKeyUpdat
// formatted preview, otherwise downstream expiry parsing breaks.
const updatedKeyData: Partial<KeyResponse> = {
...response,
token: response.token || response.key_id || selectedToken.token,
token: response.token_id || response.token || selectedToken.token,
key_name: response.key,
max_budget: formValues.max_budget,
tpm_limit: formValues.tpm_limit,

View file

@ -1,5 +1,5 @@
import React from "react";
import { fireEvent, screen, waitFor } from "@testing-library/react";
import { fireEvent, screen, waitFor, within } from "@testing-library/react";
import userEvent from "@testing-library/user-event";
import { vi, test, expect, beforeEach } from "vitest";
import { renderWithProviders } from "../../../tests/test-utils";
@ -290,3 +290,37 @@ test("should keep unsaved settings edits when switching tabs and back", async ()
expect(screen.getByLabelText(/Organization Name/i)).toHaveValue("Renamed Org");
});
test("renders a tpm/rpm limit of 0 as 0 in the overview and settings tabs, never as Unlimited", async () => {
const zeroLimitOrg = {
...mockOrg,
litellm_budget_table: { ...mockOrg.litellm_budget_table, tpm_limit: 0, rpm_limit: 0 },
};
mockUseOrganization.mockReturnValue({ data: zeroLimitOrg, isLoading: false } as unknown as ReturnType<
typeof useOrganization
>);
const user = userEvent.setup();
renderWithProviders(
<OrganizationInfoView
organizationId="org_123"
onClose={() => {}}
accessToken="test-token"
is_org_admin={false}
is_proxy_admin={true}
userModels={[]}
editOrg={false}
/>,
);
const overview = await screen.findByRole("tabpanel", { name: "Overview" });
expect(within(overview).getByText("TPM: 0")).toBeInTheDocument();
expect(within(overview).getByText("RPM: 0")).toBeInTheDocument();
await user.click(screen.getByRole("tab", { name: "Settings" }));
const settings = await screen.findByRole("tabpanel", { name: "Settings" });
expect(within(settings).getByText("TPM: 0")).toBeInTheDocument();
expect(within(settings).getByText("RPM: 0")).toBeInTheDocument();
expect(screen.queryByText("TPM: Unlimited")).not.toBeInTheDocument();
expect(screen.queryByText("RPM: Unlimited")).not.toBeInTheDocument();
});

View file

@ -206,8 +206,8 @@ const OrganizationInfoView: React.FC<OrganizationInfoProps> = ({
<CardContent>
<p className="text-sm text-muted-foreground">Rate Limits</p>
<div className="mt-2 text-sm text-foreground">
<p>TPM: {orgData.litellm_budget_table.tpm_limit || "Unlimited"}</p>
<p>RPM: {orgData.litellm_budget_table.rpm_limit || "Unlimited"}</p>
<p>TPM: {orgData.litellm_budget_table.tpm_limit ?? "Unlimited"}</p>
<p>RPM: {orgData.litellm_budget_table.rpm_limit ?? "Unlimited"}</p>
{orgData.litellm_budget_table.max_parallel_requests && (
<p>Max Parallel Requests: {orgData.litellm_budget_table.max_parallel_requests}</p>
)}
@ -311,8 +311,8 @@ const OrganizationInfoView: React.FC<OrganizationInfoProps> = ({
</div>
<div>
<p className="font-medium text-foreground">Rate Limits</p>
<div>TPM: {orgData.litellm_budget_table.tpm_limit || "Unlimited"}</div>
<div>RPM: {orgData.litellm_budget_table.rpm_limit || "Unlimited"}</div>
<div>TPM: {orgData.litellm_budget_table.tpm_limit ?? "Unlimited"}</div>
<div>RPM: {orgData.litellm_budget_table.rpm_limit ?? "Unlimited"}</div>
</div>
<div>
<p className="font-medium text-foreground">Budget</p>

View file

@ -19,6 +19,15 @@ describe("CreatedKeyDisplay", () => {
expect(screen.getByText("sk-test-123")).toBeInTheDocument();
});
it("should theme the key box with tokens instead of a hardcoded light background", () => {
render(<CreatedKeyDisplay apiKey="sk-test-123" />);
const keyBox = screen.getByText("sk-test-123").parentElement as HTMLElement;
expect(keyBox).not.toHaveAttribute("style");
expect(keyBox).toHaveClass("bg-muted");
});
it("should display the security warning", () => {
render(<CreatedKeyDisplay apiKey="sk-test-123" />);
expect(screen.getByText(/you will not be able to view it again/i)).toBeInTheDocument();

View file

@ -29,15 +29,8 @@ const CreatedKeyDisplay: React.FC<CreatedKeyDisplayProps> = ({ apiKey }) => {
</p>
<p className="text-sm text-muted-foreground mt-3 mb-1">Virtual Key:</p>
<div
style={{
background: "#f8f8f8",
padding: "10px",
borderRadius: "5px",
marginBottom: "10px",
}}
>
<pre style={{ wordWrap: "break-word", whiteSpace: "normal", margin: 0 }}>{apiKey}</pre>
<div className="bg-muted rounded-md p-2.5 mb-2.5">
<pre className="m-0 whitespace-normal break-words text-foreground">{apiKey}</pre>
</div>
<CopyToClipboard text={apiKey} onCopy={handleCopy}>

View file

@ -110,7 +110,7 @@ describe("EditMembership submit payload", () => {
expect(submitted()).toStrictEqual(expected);
});
it("collapses falsy budget and limit values to null and a missing model list to an empty array", async () => {
it("keeps stored 0 budget and limits as 0 on an untouched save, collapsing only empty strings and a missing model list", async () => {
renderEdit(teamMemberConfig, {
user_id: "u1",
user_email: "a@b.com",
@ -128,10 +128,10 @@ describe("EditMembership submit payload", () => {
user_email: "a@b.com",
user_id: "u1",
role: "user",
max_budget_in_team: null,
max_budget_in_team: 0,
budget_duration: null,
tpm_limit: null,
rpm_limit: null,
tpm_limit: 0,
rpm_limit: 0,
allowed_models: [],
});
});

View file

@ -319,6 +319,34 @@ describe("TeamInfoView", () => {
expect(screen.getByText(/of \$1,000\.00/)).toBeInTheDocument();
});
it("renders a tpm/rpm/budget limit of 0 as 0 in the overview and settings tabs, never as Unlimited or No Limit", async () => {
vi.mocked(networking.teamInfoCall).mockResolvedValue(
createMockTeamData({
tpm_limit: 0,
rpm_limit: 0,
team_member_budget_table: { max_budget: 0, budget_duration: null, tpm_limit: 0, rpm_limit: 0 },
}),
);
renderWithProviders(<TeamInfoView {...defaultProps} />);
const overview = await screen.findByRole("tabpanel", { name: "Overview" });
expect(within(overview).getByText("TPM: 0")).toBeInTheDocument();
expect(within(overview).getByText("RPM: 0")).toBeInTheDocument();
await userEvent.setup({ delay: null }).click(screen.getByRole("tab", { name: "Settings" }));
const settings = await screen.findByRole("tabpanel", { name: "Settings" });
expect(within(settings).getByText("TPM: 0")).toBeInTheDocument();
expect(within(settings).getByText("RPM: 0")).toBeInTheDocument();
expect(within(settings).getByText("TPM Limit: 0")).toBeInTheDocument();
expect(within(settings).getByText("RPM Limit: 0")).toBeInTheDocument();
expect(within(settings).getByText("Max Budget: 0")).toBeInTheDocument();
expect(screen.queryByText("TPM: Unlimited")).not.toBeInTheDocument();
expect(screen.queryByText("RPM: Unlimited")).not.toBeInTheDocument();
expect(screen.queryByText("TPM Limit: No Limit")).not.toBeInTheDocument();
expect(screen.queryByText("RPM Limit: No Limit")).not.toBeInTheDocument();
});
it("should display guardrails in overview when present", async () => {
vi.mocked(networking.teamInfoCall).mockResolvedValue(
createMockTeamData({

View file

@ -971,8 +971,8 @@ const TeamInfoView: React.FC<TeamInfoProps> = ({
<Card className="block p-6">
<p>Rate Limits</p>
<div className="mt-2">
<p>TPM: {info.tpm_limit || "Unlimited"}</p>
<p>RPM: {info.rpm_limit || "Unlimited"}</p>
<p>TPM: {info.tpm_limit ?? "Unlimited"}</p>
<p>RPM: {info.rpm_limit ?? "Unlimited"}</p>
{info.max_parallel_requests && <p>Max Parallel Requests: {info.max_parallel_requests}</p>}
{(() => {
const modelTpm = (info.metadata?.model_tpm_limit ?? {}) as Record<string, number>;
@ -1760,8 +1760,8 @@ const TeamInfoView: React.FC<TeamInfoProps> = ({
</div>
<div>
<p className="font-medium">Rate Limits</p>
<div>TPM: {info.tpm_limit || "Unlimited"}</div>
<div>RPM: {info.rpm_limit || "Unlimited"}</div>
<div>TPM: {info.tpm_limit ?? "Unlimited"}</div>
<div>RPM: {info.rpm_limit ?? "Unlimited"}</div>
{(() => {
const modelTpm = (info.metadata?.model_tpm_limit ?? {}) as Record<string, number>;
const modelRpm = (info.metadata?.model_rpm_limit ?? {}) as Record<string, number>;
@ -1811,11 +1811,11 @@ const TeamInfoView: React.FC<TeamInfoProps> = ({
<Info className="ml-1 inline size-3.5 align-text-bottom" />
</SimpleTooltip>
</p>
<div>Max Budget: {info.team_member_budget_table?.max_budget || "No Limit"}</div>
<div>Max Budget: {info.team_member_budget_table?.max_budget ?? "No Limit"}</div>
<div>Budget Duration: {info.team_member_budget_table?.budget_duration || "No Limit"}</div>
<div>Key Duration: {info.metadata?.team_member_key_duration || "No Limit"}</div>
<div>TPM Limit: {info.team_member_budget_table?.tpm_limit || "No Limit"}</div>
<div>RPM Limit: {info.team_member_budget_table?.rpm_limit || "No Limit"}</div>
<div>TPM Limit: {info.team_member_budget_table?.tpm_limit ?? "No Limit"}</div>
<div>RPM Limit: {info.team_member_budget_table?.rpm_limit ?? "No Limit"}</div>
</div>
<div>
<p className="font-medium">Router Settings</p>

View file

@ -1,4 +1,4 @@
import { screen } from "@testing-library/react";
import { screen, within } from "@testing-library/react";
import userEvent from "@testing-library/user-event";
import { beforeEach, describe, expect, it, vi } from "vitest";
import { renderWithProviders } from "../../../tests/test-utils";
@ -322,6 +322,43 @@ describe("TeamMembersComponent", () => {
expect(mockSetSelectedEditMember).toHaveBeenCalled();
});
it("keeps a member's stored 0 limits as 0 in the table and in the edit payload, never unlimited", async () => {
const user = userEvent.setup();
vi.mocked(isProxyAdminRole).mockReturnValue(true);
const baseTeamData = createMockTeamData();
const teamData = {
...baseTeamData,
team_memberships: baseTeamData.team_memberships.map((membership, index) =>
index === 0
? {
...membership,
litellm_budget_table: { ...membership.litellm_budget_table, max_budget: 0, tpm_limit: 0, rpm_limit: 0 },
}
: membership,
),
};
renderWithProviders(
<TeamMembersComponent
teamData={teamData}
canEditTeam={true}
handleMemberDelete={mockHandleMemberDelete}
setSelectedEditMember={mockSetSelectedEditMember}
setIsEditMemberModalVisible={mockSetIsEditMemberModalVisible}
setIsAddMemberModalVisible={mockSetIsAddMemberModalVisible}
/>,
);
const memberRow = screen.getByRole("row", { name: /user1@test\.com/ });
expect(within(memberRow).getByText("0 RPM / 0 TPM")).toBeInTheDocument();
expect(within(memberRow).queryByText("No Limits")).not.toBeInTheDocument();
await user.click(within(memberRow).getByTestId("edit-member"));
const zeroLimitsMember = { user_id: "user1@test.com", max_budget_in_team: 0, tpm_limit: 0, rpm_limit: 0 };
expect(mockSetSelectedEditMember).toHaveBeenCalledWith(expect.objectContaining(zeroLimitsMember));
});
it("should call setIsAddMemberModalVisible when Add Member button is clicked", async () => {
const user = userEvent.setup();

View file

@ -71,8 +71,8 @@ export default function TeamMemberTab({
const rpmLimit = membership?.litellm_budget_table?.rpm_limit;
const tpmLimit = membership?.litellm_budget_table?.tpm_limit;
const rpmText = rpmLimit ? `${formatNumber(rpmLimit)} RPM` : null;
const tpmText = tpmLimit ? `${formatNumber(tpmLimit)} TPM` : null;
const rpmText = rpmLimit != null ? `${formatNumber(rpmLimit)} RPM` : null;
const tpmText = tpmLimit != null ? `${formatNumber(tpmLimit)} TPM` : null;
const limits = [rpmText, tpmText].filter(Boolean);
return limits.length > 0 ? limits.join(" / ") : "No Limits";
@ -191,9 +191,9 @@ export default function TeamMemberTab({
const membership = teamData.team_memberships.find((tm) => tm.user_id === record.user_id);
const enhancedMember = {
...record,
max_budget_in_team: membership?.litellm_budget_table?.max_budget || null,
tpm_limit: membership?.litellm_budget_table?.tpm_limit || null,
rpm_limit: membership?.litellm_budget_table?.rpm_limit || null,
max_budget_in_team: membership?.litellm_budget_table?.max_budget ?? null,
tpm_limit: membership?.litellm_budget_table?.tpm_limit ?? null,
rpm_limit: membership?.litellm_budget_table?.rpm_limit ?? null,
budget_duration: membership?.litellm_budget_table?.budget_duration || null,
allowed_models: membership?.litellm_budget_table?.allowed_models || [],
};

View file

@ -82,7 +82,7 @@ describe("buildMemberFormValues", () => {
});
});
it("collapses falsy budgets and limits to null and a missing model list to an empty array", () => {
it("keeps a stored budget or limit of 0 as 0 because only null means unlimited", () => {
expect(
buildMemberFormValues(
"edit",
@ -90,6 +90,19 @@ describe("buildMemberFormValues", () => {
teamConfig,
),
).toStrictEqual({
user_email: "a@b.com",
user_id: "u1",
role: "user",
max_budget_in_team: 0,
budget_duration: null,
tpm_limit: 0,
rpm_limit: 0,
allowed_models: [],
});
});
it("collapses missing budgets and limits to null and a missing model list to an empty array", () => {
const unlimitedMember = {
user_email: "a@b.com",
user_id: "u1",
role: "user",
@ -98,7 +111,10 @@ describe("buildMemberFormValues", () => {
tpm_limit: null,
rpm_limit: null,
allowed_models: [],
});
};
expect(
buildMemberFormValues("edit", { user_email: "a@b.com", user_id: "u1", role: "user" }, teamConfig),
).toStrictEqual(unlimitedMember);
});
it("falls back to the configured default role when the member has none", () => {

View file

@ -43,9 +43,9 @@ export const buildMemberFormValues = (
const seeded: MemberFormValues = {
...initialData,
role: (initialData.role as string) || config.defaultRole,
max_budget_in_team: initialData.max_budget_in_team || null,
tpm_limit: initialData.tpm_limit || null,
rpm_limit: initialData.rpm_limit || null,
max_budget_in_team: initialData.max_budget_in_team ?? null,
tpm_limit: initialData.tpm_limit ?? null,
rpm_limit: initialData.rpm_limit ?? null,
budget_duration: initialData.budget_duration || null,
allowed_models: initialData.allowed_models || [],
};

View file

@ -100,6 +100,9 @@ export default function KeyInfoView({
// Add local state to maintain key data and track regeneration
const [currentKeyData, setCurrentKeyData] = useState<KeyResponse | undefined>(keyData);
const [lastRegeneratedAt, setLastRegeneratedAt] = useState<Date | null>(null);
const [keyDataUpdateHeldUntilModalClose, setKeyDataUpdateHeldUntilModalClose] = useState<Partial<KeyResponse> | null>(
null,
);
const [isRecentlyRegenerated, setIsRecentlyRegenerated] = useState(false);
const [policyGuardrails, setPolicyGuardrails] = useState<Record<string, string[]>>({});
const [loadingPolicies, setLoadingPolicies] = useState(false);
@ -352,6 +355,7 @@ export default function KeyInfoView({
};
const handleRegenerateKeyUpdate = (updatedKeyData: Partial<KeyResponse>) => {
const regeneratedAt = new Date();
// Update local state immediately with ALL the new data
setCurrentKeyData((prevData) => {
if (!prevData) return undefined;
@ -359,20 +363,26 @@ export default function KeyInfoView({
...prevData,
...updatedKeyData, // This should include the new token (key-id)
// Update the created_at to show when it was regenerated
created_at: new Date().toLocaleString(),
created_at: regeneratedAt.toLocaleString(),
};
return newData;
});
// Track regeneration timestamp
setLastRegeneratedAt(new Date());
setLastRegeneratedAt(regeneratedAt);
setIsRecentlyRegenerated(true);
if (onKeyDataUpdate) {
onKeyDataUpdate({
...updatedKeyData,
created_at: new Date().toLocaleString(),
});
setKeyDataUpdateHeldUntilModalClose({
...updatedKeyData,
created_at: regeneratedAt.toLocaleString(),
});
};
const handleRegenerateModalClose = () => {
setIsRegenerateModalOpen(false);
if (keyDataUpdateHeldUntilModalClose) {
setKeyDataUpdateHeldUntilModalClose(null);
onKeyDataUpdate?.(keyDataUpdateHeldUntilModalClose);
}
};
@ -506,7 +516,7 @@ export default function KeyInfoView({
<RegenerateKeyModal
selectedToken={currentKeyData}
visible={isRegenerateModalOpen}
onClose={() => setIsRegenerateModalOpen(false)}
onClose={handleRegenerateModalClose}
onKeyUpdate={handleRegenerateKeyUpdate}
/>

View file

@ -23142,7 +23142,7 @@ export interface components {
* CallTypes
* @enum {string}
*/
CallTypes: "embedding" | "aembedding" | "completion" | "acompletion" | "atext_completion" | "text_completion" | "image_generation" | "aimage_generation" | "image_edit" | "aimage_edit" | "moderation" | "amoderation" | "atranscription" | "transcription" | "aspeech" | "speech" | "rerank" | "arerank" | "search" | "asearch" | "_arealtime" | "_aresponses_websocket" | "create_batch" | "acreate_batch" | "aretrieve_batch" | "retrieve_batch" | "acancel_batch" | "cancel_batch" | "pass_through_endpoint" | "anthropic_messages" | "aanthropic_messages" | "get_assistants" | "aget_assistants" | "create_assistants" | "acreate_assistants" | "delete_assistant" | "adelete_assistant" | "acreate_thread" | "create_thread" | "aget_thread" | "get_thread" | "a_add_message" | "add_message" | "aget_messages" | "get_messages" | "arun_thread" | "run_thread" | "arun_thread_stream" | "run_thread_stream" | "afile_retrieve" | "file_retrieve" | "afile_delete" | "file_delete" | "afile_list" | "file_list" | "acreate_file" | "create_file" | "afile_content" | "file_content" | "create_fine_tuning_job" | "acreate_fine_tuning_job" | "create_video" | "acreate_video" | "avideo_retrieve" | "video_retrieve" | "avideo_content" | "video_content" | "video_remix" | "avideo_remix" | "video_list" | "avideo_list" | "video_retrieve_job" | "avideo_retrieve_job" | "video_delete" | "avideo_delete" | "video_create_character" | "avideo_create_character" | "video_get_character" | "avideo_get_character" | "video_edit" | "avideo_edit" | "video_extension" | "avideo_extension" | "vector_store_file_create" | "avector_store_file_create" | "vector_store_file_list" | "avector_store_file_list" | "vector_store_file_retrieve" | "avector_store_file_retrieve" | "vector_store_file_content" | "avector_store_file_content" | "vector_store_file_update" | "avector_store_file_update" | "vector_store_file_delete" | "avector_store_file_delete" | "vector_store_create" | "avector_store_create" | "vector_store_search" | "avector_store_search" | "ingest" | "aingest" | "query" | "aquery" | "create_container" | "acreate_container" | "list_containers" | "alist_containers" | "retrieve_container" | "aretrieve_container" | "delete_container" | "adelete_container" | "list_container_files" | "alist_container_files" | "upload_container_file" | "aupload_container_file" | "create_sandbox" | "acreate_sandbox" | "delete_sandbox" | "adelete_sandbox" | "run_code" | "arun_code" | "code_interpreter_tool" | "acode_interpreter_tool" | "acancel_fine_tuning_job" | "cancel_fine_tuning_job" | "alist_fine_tuning_jobs" | "list_fine_tuning_jobs" | "aretrieve_fine_tuning_job" | "retrieve_fine_tuning_job" | "responses" | "aresponses" | "alist_input_items" | "llm_passthrough_route" | "allm_passthrough_route" | "generate_content" | "agenerate_content" | "generate_content_stream" | "agenerate_content_stream" | "ocr" | "aocr" | "call_mcp_tool" | "list_mcp_tools" | "asend_message" | "send_message" | "acreate_skill";
CallTypes: "embedding" | "aembedding" | "completion" | "acompletion" | "atext_completion" | "text_completion" | "image_generation" | "aimage_generation" | "image_edit" | "aimage_edit" | "moderation" | "amoderation" | "atranscription" | "transcription" | "aspeech" | "speech" | "rerank" | "arerank" | "search" | "asearch" | "_arealtime" | "_aresponses_websocket" | "create_batch" | "acreate_batch" | "aretrieve_batch" | "retrieve_batch" | "acancel_batch" | "cancel_batch" | "pass_through_endpoint" | "anthropic_messages" | "aanthropic_messages" | "get_assistants" | "aget_assistants" | "create_assistants" | "acreate_assistants" | "delete_assistant" | "adelete_assistant" | "acreate_thread" | "create_thread" | "aget_thread" | "get_thread" | "a_add_message" | "add_message" | "aget_messages" | "get_messages" | "arun_thread" | "run_thread" | "arun_thread_stream" | "run_thread_stream" | "afile_retrieve" | "file_retrieve" | "afile_delete" | "file_delete" | "afile_list" | "file_list" | "acreate_file" | "create_file" | "afile_content" | "file_content" | "create_fine_tuning_job" | "acreate_fine_tuning_job" | "create_video" | "acreate_video" | "avideo_retrieve" | "video_retrieve" | "avideo_content" | "video_content" | "video_remix" | "avideo_remix" | "video_list" | "avideo_list" | "video_retrieve_job" | "avideo_retrieve_job" | "video_delete" | "avideo_delete" | "video_create_character" | "avideo_create_character" | "video_get_character" | "avideo_get_character" | "video_edit" | "avideo_edit" | "video_extension" | "avideo_extension" | "vector_store_file_create" | "avector_store_file_create" | "vector_store_file_list" | "avector_store_file_list" | "vector_store_file_retrieve" | "avector_store_file_retrieve" | "vector_store_file_content" | "avector_store_file_content" | "vector_store_file_update" | "avector_store_file_update" | "vector_store_file_delete" | "avector_store_file_delete" | "vector_store_create" | "avector_store_create" | "vector_store_search" | "avector_store_search" | "ingest" | "aingest" | "query" | "aquery" | "create_interaction" | "acreate_interaction" | "create_container" | "acreate_container" | "list_containers" | "alist_containers" | "retrieve_container" | "aretrieve_container" | "delete_container" | "adelete_container" | "list_container_files" | "alist_container_files" | "upload_container_file" | "aupload_container_file" | "create_sandbox" | "acreate_sandbox" | "delete_sandbox" | "adelete_sandbox" | "run_code" | "arun_code" | "code_interpreter_tool" | "acode_interpreter_tool" | "acancel_fine_tuning_job" | "cancel_fine_tuning_job" | "alist_fine_tuning_jobs" | "list_fine_tuning_jobs" | "aretrieve_fine_tuning_job" | "retrieve_fine_tuning_job" | "responses" | "aresponses" | "alist_input_items" | "llm_passthrough_route" | "allm_passthrough_route" | "generate_content" | "agenerate_content" | "generate_content_stream" | "agenerate_content_stream" | "ocr" | "aocr" | "call_mcp_tool" | "list_mcp_tools" | "asend_message" | "send_message" | "acreate_skill";
/** CallbackDelete */
CallbackDelete: {
/** Callback Name */