mirror of
https://github.com/BerriAI/litellm.git
synced 2026-08-28 05:25:59 +00:00
Merge remote-tracking branch 'origin/litellm_internal_staging' into litellm_health_test_connection_health_check_params
# Conflicts: # tests/test_litellm/proxy/test_health_check_max_tokens.py
This commit is contained in:
commit
531aafdf75
76 changed files with 4393 additions and 304 deletions
|
|
@ -4,6 +4,7 @@ Custom A2A Card Resolver for LiteLLM.
|
|||
Extends the A2A SDK's card resolver to support multiple well-known paths.
|
||||
"""
|
||||
|
||||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
|
|
@ -48,6 +49,43 @@ def is_localhost_or_internal_url(url: str | None) -> bool:
|
|||
return any(pattern in url_lower for pattern in LOCALHOST_URL_PATTERNS)
|
||||
|
||||
|
||||
_CANONICAL_PROTOCOL_BINDINGS: Final = MappingProxyType(
|
||||
{
|
||||
"jsonrpc": "JSONRPC",
|
||||
"http+json": "HTTP+JSON",
|
||||
"grpc": "GRPC",
|
||||
}
|
||||
)
|
||||
|
||||
_LEGACY_PROTOCOL_VERSION: Final = "0.3"
|
||||
|
||||
|
||||
def normalize_agent_card_interfaces(agent_card: "AgentCard") -> "AgentCard":
|
||||
"""
|
||||
Canonicalize the supported interfaces of spec-adjacent agent cards.
|
||||
|
||||
Some A2A servers (e.g. LangGraph Platform) serve agent cards with lowercase
|
||||
bindings like "jsonrpc", but a2a-sdk's ClientFactory matches bindings
|
||||
case-sensitively against its uppercase TransportProtocol constants and fails
|
||||
with "no compatible transports found." for spec-adjacent casings.
|
||||
|
||||
The same servers also speak the A2A 0.3 JSON dialect ("kind"-discriminated
|
||||
payloads) while declaring protocolVersion "1.0", which a2a-sdk's strict v1
|
||||
proto parsing rejects. A mis-cased binding fingerprints such a server, so its
|
||||
declared version is downgraded to 0.3 to route the SDK's ClientFactory onto
|
||||
its v0.3 compat transport, which speaks that dialect.
|
||||
"""
|
||||
normalized: Final = type(agent_card)()
|
||||
normalized.CopyFrom(agent_card)
|
||||
for interface in normalized.supported_interfaces:
|
||||
canonical: str | None = _CANONICAL_PROTOCOL_BINDINGS.get(interface.protocol_binding.lower())
|
||||
if canonical is None or canonical == interface.protocol_binding:
|
||||
continue
|
||||
interface.protocol_binding = canonical
|
||||
interface.protocol_version = _LEGACY_PROTOCOL_VERSION
|
||||
return normalized
|
||||
|
||||
|
||||
def get_agent_card_url(agent_card: "AgentCard") -> str | None:
|
||||
"""Return the agent endpoint URL from the resolved SDK card."""
|
||||
url: Final = getattr(agent_card, "url", None)
|
||||
|
|
|
|||
|
|
@ -73,6 +73,7 @@ except ImportError:
|
|||
from litellm.a2a_protocol.card_resolver import (
|
||||
LiteLLMA2ACardResolver,
|
||||
get_agent_card_url,
|
||||
normalize_agent_card_interfaces,
|
||||
)
|
||||
from litellm.a2a_protocol.exception_mapping_utils import (
|
||||
handle_a2a_localhost_retry,
|
||||
|
|
@ -782,13 +783,17 @@ async def create_a2a_client(
|
|||
if extra_headers:
|
||||
verbose_proxy_logger.debug("A2A client created with extra_headers=%s", list(extra_headers.keys()))
|
||||
|
||||
resolver: Final = A2ACardResolver(httpx_client=httpx_client, base_url=base_url)
|
||||
agent_card: Final = normalize_agent_card_interfaces(
|
||||
await resolver.get_agent_card(http_kwargs={"headers": extra_headers} if extra_headers else None)
|
||||
)
|
||||
|
||||
a2a_client: Final = await create_client( # pyright: ignore[reportOptionalCall]
|
||||
base_url,
|
||||
agent_card,
|
||||
client_config=ClientConfig( # pyright: ignore[reportOptionalCall]
|
||||
httpx_client=httpx_client,
|
||||
streaming=streaming,
|
||||
),
|
||||
resolver_http_kwargs={"headers": extra_headers} if extra_headers else None,
|
||||
)
|
||||
# Stash LiteLLM-owned handles on the client so the localhost-retry path can reuse
|
||||
# the configured httpx client and this agent's headers without excavating
|
||||
|
|
@ -799,9 +804,7 @@ async def create_a2a_client(
|
|||
if extra_headers
|
||||
else None
|
||||
)
|
||||
agent_card: Final = getattr(a2a_client, "_card", None)
|
||||
if agent_card is not None:
|
||||
a2a_client._litellm_agent_card = agent_card
|
||||
a2a_client._litellm_agent_card = agent_card
|
||||
|
||||
verbose_logger.info("A2A client created for %s", base_url)
|
||||
|
||||
|
|
|
|||
|
|
@ -21,6 +21,9 @@ from pydantic import BaseModel
|
|||
import litellm
|
||||
from litellm import ModelResponse
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
responses_reasoning_item_from_thinking_blocks,
|
||||
)
|
||||
from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator
|
||||
from litellm.llms.base_llm.bridges.completion_transformation import (
|
||||
CompletionTransformationBridge,
|
||||
|
|
@ -85,6 +88,22 @@ def _get_reasoning_items(
|
|||
return []
|
||||
|
||||
|
||||
def _reasoning_input_items(msg: "AllMessageValues") -> list[dict[str, object]]: # mutable-ok: API message payload
|
||||
"""Reasoning input items for an assistant message.
|
||||
|
||||
Stored reasoning items win because they carry an id the Responses API minted; thinking
|
||||
blocks are the fallback for turns that arrived over another API surface.
|
||||
"""
|
||||
items: Final = _get_reasoning_items(msg)
|
||||
stored: Final = [_reasoning_item_to_response_input(item) for item in items] # mutable-ok: API message payload
|
||||
if stored:
|
||||
return stored
|
||||
raw_blocks: Final = msg.get("thinking_blocks") or ()
|
||||
blocks: Final = cast("Iterable[ChatCompletionThinkingBlock]", raw_blocks) # cast-ok: untyped client json
|
||||
from_thinking: Final = responses_reasoning_item_from_thinking_blocks(blocks)
|
||||
return [] if from_thinking is None else [dict(from_thinking)] # mutable-ok: API message payload
|
||||
|
||||
|
||||
def _build_reasoning_item(
|
||||
item_id: str,
|
||||
encrypted_content: str | None,
|
||||
|
|
@ -372,8 +391,15 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
)
|
||||
)
|
||||
elif role == "assistant" and tool_calls and isinstance(tool_calls, list):
|
||||
for r_item in _get_reasoning_items(msg):
|
||||
input_items.append(_reasoning_item_to_response_input(r_item))
|
||||
input_items.extend(_reasoning_input_items(msg))
|
||||
if content:
|
||||
input_items.append(
|
||||
{ # mutable-ok: API message payload
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"content": self._convert_content_to_responses_format(content, "assistant"),
|
||||
}
|
||||
)
|
||||
for tool_call in tool_calls:
|
||||
function = tool_call.get("function")
|
||||
custom = tool_call.get("custom")
|
||||
|
|
@ -400,15 +426,16 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
raise ValueError(f"tool call not supported: {tool_call}")
|
||||
elif content is not None:
|
||||
if role == "assistant":
|
||||
for r_item in _get_reasoning_items(msg):
|
||||
input_items.append(_reasoning_item_to_response_input(r_item))
|
||||
input_items.extend(_reasoning_input_items(msg))
|
||||
input_items.append(
|
||||
{
|
||||
{ # mutable-ok: API message payload
|
||||
"type": "message",
|
||||
"role": role,
|
||||
"content": self._convert_content_to_responses_format(content, cast(str, role)),
|
||||
}
|
||||
)
|
||||
elif role == "assistant":
|
||||
input_items.extend(_reasoning_input_items(msg))
|
||||
|
||||
return input_items, instructions
|
||||
|
||||
|
|
|
|||
|
|
@ -1563,6 +1563,19 @@ STALE_OBJECT_CLEANUP_BATCH_SIZE: Final = max(1, int(os.getenv("STALE_OBJECT_CLEA
|
|||
# installations with large numbers of stale managed objects).
|
||||
_batch_polling_env: Final = os.getenv("PROXY_BATCH_POLLING_ENABLED", "true").lower()
|
||||
PROXY_BATCH_POLLING_ENABLED: Final = _batch_polling_env == "true"
|
||||
BACKGROUND_INTERACTION_COST_POLL_INITIAL_INTERVAL_SECONDS: Final = float(
|
||||
os.getenv("BACKGROUND_INTERACTION_COST_POLL_INITIAL_INTERVAL_SECONDS", "5")
|
||||
)
|
||||
BACKGROUND_INTERACTION_COST_POLL_MAX_INTERVAL_SECONDS: Final = float(
|
||||
os.getenv("BACKGROUND_INTERACTION_COST_POLL_MAX_INTERVAL_SECONDS", "60")
|
||||
)
|
||||
BACKGROUND_INTERACTION_COST_POLL_TIMEOUT_SECONDS: Final = float(
|
||||
os.getenv("BACKGROUND_INTERACTION_COST_POLL_TIMEOUT_SECONDS", "3600")
|
||||
)
|
||||
_background_interaction_cost_polling_env: Final = os.getenv(
|
||||
"BACKGROUND_INTERACTION_COST_POLLING_ENABLED", "true"
|
||||
).lower()
|
||||
BACKGROUND_INTERACTION_COST_POLLING_ENABLED: Final = _background_interaction_cost_polling_env == "true"
|
||||
PROXY_BUDGET_RESCHEDULER_MAX_TIME: Final = int(os.getenv("PROXY_BUDGET_RESCHEDULER_MAX_TIME", 605))
|
||||
PROXY_BATCH_WRITE_AT: Final = int(os.getenv("PROXY_BATCH_WRITE_AT", 10)) # in seconds, increased from 10
|
||||
PROXY_CONFIG_RELOAD_INTERVAL_SECONDS: Final = get_env_int("PROXY_CONFIG_RELOAD_INTERVAL_SECONDS", 30)
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
|
|||
StandardBuiltInToolCostTracking,
|
||||
)
|
||||
from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import (
|
||||
InteractionsUsageObjectTransformation,
|
||||
TranscriptionUsageObjectTransformation,
|
||||
)
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import (
|
||||
|
|
@ -912,6 +913,8 @@ def _get_usage_object(
|
|||
usage_obj,
|
||||
)
|
||||
)
|
||||
elif isinstance(usage_obj, dict) and InteractionsUsageObjectTransformation.is_interactions_usage_object(usage_obj):
|
||||
return InteractionsUsageObjectTransformation.transform_interactions_usage_object(usage_obj)
|
||||
elif isinstance(usage_obj, dict):
|
||||
return Usage(**usage_obj)
|
||||
elif isinstance(usage_obj, BaseModel):
|
||||
|
|
@ -1288,6 +1291,10 @@ def completion_cost(
|
|||
)
|
||||
if tr_usage is not None:
|
||||
_usage = tr_usage.model_dump()
|
||||
elif InteractionsUsageObjectTransformation.is_interactions_usage_object(_usage):
|
||||
_usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
|
||||
_usage
|
||||
).model_dump()
|
||||
else:
|
||||
_usage = _usage
|
||||
|
||||
|
|
|
|||
313
litellm/interactions/background_cost_polling.py
Normal file
313
litellm/interactions/background_cost_polling.py
Normal file
|
|
@ -0,0 +1,313 @@
|
|||
"""
|
||||
Cost tracking for background interactions.
|
||||
|
||||
A create request with ``background=true`` returns ``in_progress`` with no
|
||||
usage block, and GET polls are deliberately never billed (billing them would
|
||||
double-charge every poll; the GET response also does not echo ``background``,
|
||||
so a poll cannot be told apart from a re-fetch of an already-billed
|
||||
interaction). The create call is therefore the only place that can own
|
||||
billing: it schedules a poll task that fetches the interaction until it
|
||||
reaches a terminal status and logs the final usage as a single success event
|
||||
attributed to the original request.
|
||||
|
||||
``requires_action`` is terminal for the interaction it names. The API has no
|
||||
operation that resumes one: a caller answers a tool request by creating a new
|
||||
interaction whose ``previous_interaction_id`` points at it, and that new
|
||||
interaction bills itself. The paused interaction keeps the tokens it already
|
||||
spent producing the tool request, so it is billed and settled where it stops
|
||||
rather than polled until the timeout, which would both lose that usage and
|
||||
hold its budget reservation open for the whole timeout window.
|
||||
|
||||
Deleting an interaction makes every subsequent poll fail, which would let a
|
||||
caller retrieve the completed output themselves and then delete it before the
|
||||
poll task settles, leaving the work unbilled and the budget reservation
|
||||
refunded at the poll timeout. ``adelete`` therefore settles any pending poll
|
||||
for the interaction before dispatching the delete: it fetches the current
|
||||
state with the create's credentials, bills it if it is terminal with usage,
|
||||
and releases the reservation otherwise. A settlement gate on the create's
|
||||
logging object makes the poll task and the delete path mutually exclusive, so
|
||||
the interaction is billed exactly once no matter who settles first.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
from collections.abc import Awaitable, Callable, Iterator, Mapping
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING, Final, TypeAlias
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import (
|
||||
BACKGROUND_INTERACTION_COST_POLL_INITIAL_INTERVAL_SECONDS,
|
||||
BACKGROUND_INTERACTION_COST_POLL_MAX_INTERVAL_SECONDS,
|
||||
BACKGROUND_INTERACTION_COST_POLL_TIMEOUT_SECONDS,
|
||||
BACKGROUND_INTERACTION_COST_POLLING_ENABLED,
|
||||
)
|
||||
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
||||
_TERMINAL_STATUSES: Final = frozenset(
|
||||
{"completed", "failed", "cancelled", "incomplete", "budget_exceeded", "requires_action"}
|
||||
)
|
||||
|
||||
_POLLABLE_STATUSES: Final = frozenset({"in_progress", "queued"})
|
||||
|
||||
_STATUSES_THAT_PRODUCED_OUTPUT: Final = frozenset({"completed", "requires_action"})
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class BackgroundInteractionPollContext:
|
||||
interaction_id: str
|
||||
custom_llm_provider: str
|
||||
logging_obj: "LiteLLMLoggingObj"
|
||||
api_key: str | None = None
|
||||
api_base: str | None = None
|
||||
initial_interval_seconds: float = BACKGROUND_INTERACTION_COST_POLL_INITIAL_INTERVAL_SECONDS
|
||||
max_interval_seconds: float = BACKGROUND_INTERACTION_COST_POLL_MAX_INTERVAL_SECONDS
|
||||
timeout_seconds: float = BACKGROUND_INTERACTION_COST_POLL_TIMEOUT_SECONDS
|
||||
|
||||
|
||||
FetchInteraction: TypeAlias = Callable[[BackgroundInteractionPollContext], Awaitable[InteractionsAPIResponse]]
|
||||
|
||||
|
||||
async def _fetch_interaction(context: BackgroundInteractionPollContext) -> InteractionsAPIResponse:
|
||||
from litellm.interactions import aget
|
||||
|
||||
return await aget(
|
||||
interaction_id=context.interaction_id,
|
||||
custom_llm_provider=context.custom_llm_provider,
|
||||
api_key=context.api_key,
|
||||
api_base=context.api_base,
|
||||
**{
|
||||
"no-log": True
|
||||
}, # mutable-ok: "no-log" is not a valid identifier, so it can only be passed through a mapping
|
||||
)
|
||||
|
||||
|
||||
def _poll_intervals(initial: float, maximum: float, timeout: float) -> Iterator[float]:
|
||||
elapsed = 0.0
|
||||
interval = initial
|
||||
while interval > 0 and elapsed + interval <= timeout:
|
||||
yield interval
|
||||
elapsed += interval
|
||||
interval = min(interval * 2, maximum)
|
||||
|
||||
|
||||
_SETTLED_KEY = "background_interaction_settled"
|
||||
|
||||
|
||||
def _is_settled(logging_obj: "LiteLLMLoggingObj") -> bool:
|
||||
return logging_obj.model_call_details.get(_SETTLED_KEY) is True
|
||||
|
||||
|
||||
def _claim_settlement(logging_obj: "LiteLLMLoggingObj") -> bool:
|
||||
"""
|
||||
Exactly-once gate between the poll task and the delete-time settlement:
|
||||
both run on the same event loop and neither awaits between reading and
|
||||
setting the flag, so whichever claims first owns billing or release.
|
||||
"""
|
||||
if _is_settled(logging_obj):
|
||||
return False
|
||||
logging_obj.model_call_details[_SETTLED_KEY] = True # rebind-ok: both settlers must see the same settlement flag
|
||||
return True
|
||||
|
||||
|
||||
async def poll_and_log_background_interaction_cost(
|
||||
context: BackgroundInteractionPollContext,
|
||||
fetch_interaction: FetchInteraction = _fetch_interaction,
|
||||
) -> None:
|
||||
last_seen_status: str | None = None
|
||||
for interval in _poll_intervals(
|
||||
initial=context.initial_interval_seconds,
|
||||
maximum=context.max_interval_seconds,
|
||||
timeout=context.timeout_seconds,
|
||||
):
|
||||
await asyncio.sleep(interval)
|
||||
if _is_settled(context.logging_obj):
|
||||
return
|
||||
try:
|
||||
response = await fetch_interaction(context)
|
||||
except Exception as e: # noqa: BLE001 # any fetch error must not kill the billing poll loop
|
||||
verbose_logger.debug(
|
||||
"Background interaction cost poll for %s failed, will retry: %s",
|
||||
context.interaction_id,
|
||||
e,
|
||||
)
|
||||
continue
|
||||
last_seen_status = response.status
|
||||
if response.status not in _TERMINAL_STATUSES:
|
||||
continue
|
||||
if not _claim_settlement(context.logging_obj):
|
||||
return
|
||||
if response.usage is not None:
|
||||
await _bill_settled_interaction(logging_obj=context.logging_obj, response=response)
|
||||
else:
|
||||
await _release_open_budget_reservation(logging_obj=context.logging_obj)
|
||||
return
|
||||
if not _claim_settlement(context.logging_obj):
|
||||
return
|
||||
if last_seen_status is not None and last_seen_status not in _POLLABLE_STATUSES:
|
||||
verbose_logger.error(
|
||||
"Gave up cost polling for background interaction %s after %ss: its last status %r is in neither "
|
||||
"the pollable nor the terminal set, so this proxy never learned how to settle it and its usage "
|
||||
"will not be tracked",
|
||||
context.interaction_id,
|
||||
context.timeout_seconds,
|
||||
last_seen_status,
|
||||
)
|
||||
else:
|
||||
verbose_logger.warning(
|
||||
"Gave up cost polling for background interaction %s after %ss; its usage will not be tracked",
|
||||
context.interaction_id,
|
||||
context.timeout_seconds,
|
||||
)
|
||||
await _release_open_budget_reservation(logging_obj=context.logging_obj)
|
||||
|
||||
|
||||
async def _release_open_budget_reservation(logging_obj: "LiteLLMLoggingObj") -> None:
|
||||
"""
|
||||
The proxy keeps the pre-call budget reservation open for an in-progress
|
||||
background interaction so concurrent creates cannot stack past the budget.
|
||||
The completion success event reconciles it to the actual cost; when the
|
||||
interaction terminates without billable usage (or polling gives up, or it
|
||||
is deleted before settling), no such event fires, so whoever claims the
|
||||
settlement must release the reservation here or the spend counters stay
|
||||
pinned at the estimated cost.
|
||||
"""
|
||||
metadata = get_litellm_metadata_from_kwargs(kwargs=logging_obj.model_call_details)
|
||||
budget_reservation = metadata.get("user_api_key_budget_reservation")
|
||||
if not isinstance(budget_reservation, dict):
|
||||
return
|
||||
|
||||
from litellm.proxy.spend_tracking.budget_reservation import release_budget_reservation
|
||||
|
||||
try:
|
||||
await release_budget_reservation(budget_reservation=budget_reservation)
|
||||
except Exception: # noqa: BLE001 # a failed release must not crash the poll task; counters expire via TTL
|
||||
verbose_logger.exception("Failed to release budget reservation for an unbilled background interaction")
|
||||
|
||||
|
||||
async def _bill_settled_interaction(logging_obj: "LiteLLMLoggingObj", response: InteractionsAPIResponse) -> None:
|
||||
"""
|
||||
Claiming the settlement makes the claimer solely responsible for the
|
||||
reservation, and no one retries a claim that is already set. A billing
|
||||
failure here must therefore release the reservation on its way out, or it
|
||||
stays pinned at the estimated cost until the whole poll times out.
|
||||
"""
|
||||
try:
|
||||
await logging_obj.async_log_background_interaction_completion(result=response)
|
||||
except Exception:
|
||||
await _release_open_budget_reservation(logging_obj=logging_obj)
|
||||
raise
|
||||
|
||||
|
||||
def is_pollable_background_interaction(response: InteractionsAPIResponse) -> bool:
|
||||
"""
|
||||
The single gate deciding whether a create's response gets a poll task.
|
||||
The proxy's success callback defers releasing the budget reservation for
|
||||
exactly these responses, on the promise that a poll task will settle them,
|
||||
so a response one site accepts and the other refuses strands its
|
||||
reservation on the spend counters with nothing left to reconcile it.
|
||||
|
||||
``queued`` belongs here alongside ``in_progress``. It is the API's
|
||||
not-started-yet state, so it reaches a terminal status the same way and
|
||||
needs polling for the same reason: nothing else in the proxy ever bills a
|
||||
create that came back without usage, so a status missing from both this
|
||||
set and ``_TERMINAL_STATUSES`` is billed nowhere and alerts nobody.
|
||||
"""
|
||||
return response.status in _POLLABLE_STATUSES and bool(response.id)
|
||||
|
||||
|
||||
def missing_usage_is_expected(response: InteractionsAPIResponse) -> bool:
|
||||
"""
|
||||
Whether a response arriving with no usage block is a normal outcome rather
|
||||
than lost billing data. An interaction that is still running, or that
|
||||
stopped at ``failed``, ``cancelled``, ``incomplete`` or ``budget_exceeded``,
|
||||
has nothing to charge for and should not raise a cost-tracking alarm.
|
||||
|
||||
``completed`` and ``requires_action`` both mean the model produced output,
|
||||
so a usage block is always expected with them. If one arrives without it
|
||||
the charge for real work has been lost, which is precisely what the
|
||||
proxy's cost-tracking alert exists to surface.
|
||||
"""
|
||||
return response.status not in _STATUSES_THAT_PRODUCED_OUTPUT
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _ActiveBackgroundPoll:
|
||||
task: "asyncio.Task[None]"
|
||||
context: BackgroundInteractionPollContext
|
||||
|
||||
|
||||
_ACTIVE_POLLS: dict[str, _ActiveBackgroundPoll] = {} # mutable-ok: asyncio needs strong refs to running poll tasks
|
||||
|
||||
|
||||
def _discard_poll(interaction_id: str, task: "asyncio.Task[None]") -> None:
|
||||
entry = _ACTIVE_POLLS.get(interaction_id)
|
||||
if entry is not None and entry.task is task:
|
||||
del _ACTIVE_POLLS[interaction_id]
|
||||
|
||||
|
||||
def maybe_schedule_background_interaction_cost_polling(
|
||||
response: object,
|
||||
create_kwargs: Mapping[str, object],
|
||||
custom_llm_provider: str,
|
||||
) -> "asyncio.Task[None] | None":
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging
|
||||
|
||||
if not BACKGROUND_INTERACTION_COST_POLLING_ENABLED:
|
||||
return None
|
||||
if not isinstance(response, InteractionsAPIResponse):
|
||||
return None
|
||||
if not is_pollable_background_interaction(response):
|
||||
return None
|
||||
logging_obj = create_kwargs.get("litellm_logging_obj")
|
||||
if not isinstance(logging_obj, Logging):
|
||||
return None
|
||||
try:
|
||||
asyncio.get_running_loop()
|
||||
except RuntimeError:
|
||||
return None
|
||||
api_key = create_kwargs.get("api_key")
|
||||
api_base = create_kwargs.get("api_base")
|
||||
context = BackgroundInteractionPollContext(
|
||||
interaction_id=response.id,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
logging_obj=logging_obj,
|
||||
api_key=api_key if isinstance(api_key, str) else None,
|
||||
api_base=api_base if isinstance(api_base, str) else None,
|
||||
)
|
||||
task = asyncio.create_task(poll_and_log_background_interaction_cost(context))
|
||||
_ACTIVE_POLLS[context.interaction_id] = _ActiveBackgroundPoll(task=task, context=context)
|
||||
task.add_done_callback(
|
||||
lambda finished, interaction_id=context.interaction_id: _discard_poll(interaction_id, finished)
|
||||
)
|
||||
return task
|
||||
|
||||
|
||||
async def maybe_settle_background_interaction_before_delete(
|
||||
interaction_id: str,
|
||||
fetch_interaction: FetchInteraction = _fetch_interaction,
|
||||
) -> None:
|
||||
entry = _ACTIVE_POLLS.get(interaction_id)
|
||||
if entry is None:
|
||||
return
|
||||
context = entry.context
|
||||
try:
|
||||
response = await fetch_interaction(context)
|
||||
except Exception as e: # noqa: BLE001 # unfetchable pre-delete state settles by releasing the reservation
|
||||
verbose_logger.debug(
|
||||
"Could not fetch background interaction %s before delete, releasing its reservation: %s",
|
||||
interaction_id,
|
||||
e,
|
||||
)
|
||||
if _claim_settlement(context.logging_obj):
|
||||
await _release_open_budget_reservation(logging_obj=context.logging_obj)
|
||||
return
|
||||
if not _claim_settlement(context.logging_obj):
|
||||
return
|
||||
if response.status in _TERMINAL_STATUSES and response.usage is not None:
|
||||
await _bill_settled_interaction(logging_obj=context.logging_obj, response=response)
|
||||
return
|
||||
await _release_open_budget_reservation(logging_obj=context.logging_obj)
|
||||
|
|
@ -40,6 +40,10 @@ from typing import Any, Final
|
|||
import httpx
|
||||
|
||||
import litellm
|
||||
from litellm.interactions.background_cost_polling import (
|
||||
maybe_schedule_background_interaction_cost_polling,
|
||||
maybe_settle_background_interaction_before_delete,
|
||||
)
|
||||
from litellm.interactions.http_handler import interactions_http_handler
|
||||
from litellm.interactions.utils import (
|
||||
InteractionsAPIRequestUtils,
|
||||
|
|
@ -171,6 +175,12 @@ async def acreate(
|
|||
else:
|
||||
response = init_response
|
||||
|
||||
maybe_schedule_background_interaction_cost_polling(
|
||||
response=response,
|
||||
create_kwargs=kwargs,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
return response
|
||||
except Exception as e:
|
||||
raise litellm.exception_type(
|
||||
|
|
@ -462,6 +472,8 @@ async def adelete(
|
|||
loop: Final = asyncio.get_event_loop()
|
||||
kwargs["adelete_interaction"] = True
|
||||
|
||||
await maybe_settle_background_interaction_before_delete(interaction_id=interaction_id)
|
||||
|
||||
func: Final = partial(
|
||||
delete,
|
||||
interaction_id=interaction_id,
|
||||
|
|
|
|||
|
|
@ -71,6 +71,9 @@ from litellm.litellm_core_utils.llm_cost_calc.guardrail_cost import (
|
|||
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
|
||||
StandardBuiltInToolCostTracking,
|
||||
)
|
||||
from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import (
|
||||
InteractionsUsageObjectTransformation,
|
||||
)
|
||||
from litellm.litellm_core_utils.logging_utils import truncate_base64_in_messages
|
||||
from litellm.litellm_core_utils.model_param_helper import ModelParamHelper
|
||||
from litellm.litellm_core_utils.redact_messages import (
|
||||
|
|
@ -83,6 +86,10 @@ from litellm.llms.base_llm.search.transformation import SearchResponse
|
|||
from litellm.responses.utils import ResponseAPILoggingUtils
|
||||
from litellm.types.agents import LiteLLMSendMessageResponse
|
||||
from litellm.types.containers.main import ContainerObject
|
||||
from litellm.types.interactions import (
|
||||
InteractionsAPIResponse,
|
||||
InteractionsAPIStreamingResponse,
|
||||
)
|
||||
from litellm.types.llms.openai import (
|
||||
AllMessageValues,
|
||||
Batch,
|
||||
|
|
@ -2145,6 +2152,11 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
or isinstance(logging_result, OpenAIModerationResponse)
|
||||
or isinstance(logging_result, OCRResponse) # OCR
|
||||
or isinstance(logging_result, SearchResponse) # Search API
|
||||
or (
|
||||
isinstance(logging_result, InteractionsAPIResponse)
|
||||
and logging_result.usage is not None
|
||||
and self._is_interactions_create_call_type()
|
||||
)
|
||||
or isinstance(logging_result, dict)
|
||||
and logging_result.get("object") == "vector_store.search_results.page"
|
||||
or isinstance(logging_result, dict)
|
||||
|
|
@ -2157,6 +2169,87 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
return True
|
||||
return False
|
||||
|
||||
def _is_interactions_create_call_type(self) -> bool:
|
||||
"""
|
||||
Only interaction creation is billable. GET polls, deletes, and cancels
|
||||
also return an ``InteractionsAPIResponse`` (with usage once completed),
|
||||
so recognizing those would write spend on every poll of a background
|
||||
interaction. The proxy sets ``call_type`` from its route_type
|
||||
(``create_interaction``/``acreate_interaction``); the SDK sets it from
|
||||
the decorated function name (``create``/``acreate``).
|
||||
|
||||
Recognition additionally requires a usage block (checked at the call
|
||||
site): a ``background=true`` create returns ``in_progress`` without
|
||||
usage, and billing it would write a $0 spend log under the interaction
|
||||
id that collides with the row the background poll task writes once the
|
||||
interaction completes (see
|
||||
``litellm.interactions.background_cost_polling``).
|
||||
"""
|
||||
return self.call_type in (
|
||||
CallTypes.create_interaction.value,
|
||||
CallTypes.acreate_interaction.value,
|
||||
"create",
|
||||
"acreate",
|
||||
)
|
||||
|
||||
async def async_log_background_interaction_completion(
|
||||
self,
|
||||
result: InteractionsAPIResponse,
|
||||
) -> None:
|
||||
"""
|
||||
Log the terminal result of a background interaction as a fresh success
|
||||
event. The create request already ran success logging for its
|
||||
``in_progress`` response (no usage, so no cost was tracked); clearing
|
||||
the dedup flags lets the completed result flow through cost calculation
|
||||
and spend tracking exactly once, spanning create to completion.
|
||||
|
||||
The poll fetched this body through its own client call, which priced it
|
||||
against a throwaway logging object holding none of this request's
|
||||
deployment context: no ``model_info``, no router ``model_id``, no
|
||||
deployment ``litellm_params``. Keeping that price would bill a
|
||||
custom-priced deployment at the wrong rate, and it would also satisfy
|
||||
the "already calculated" shortcut and skip repricing here, leaving the
|
||||
cost breakdown at the zeros the usage-less create stamped and writing
|
||||
those zeros to the spend log. Dropping it makes this event price the
|
||||
settled body itself, against the deployment that served the create.
|
||||
|
||||
The same throwaway call stamped the deployment identity that travels
|
||||
with the price, so ``model_id`` and ``litellm_model_name`` go with it.
|
||||
Left in place they overwrite the create's real deployment with the
|
||||
poll's empty one in the payload every logging integration reads.
|
||||
"""
|
||||
settled_hidden_params: Final = getattr(result, "_hidden_params", None)
|
||||
if isinstance(settled_hidden_params, dict):
|
||||
for poll_scoped_key in ("response_cost", "model_id", "litellm_model_name"):
|
||||
settled_hidden_params.pop(poll_scoped_key, None)
|
||||
self._reset_success_emission_dedupe()
|
||||
await self.async_success_handler(result=result)
|
||||
|
||||
def _reset_success_emission_dedupe(self) -> None:
|
||||
"""
|
||||
Success callbacks dedupe per request, because the sync and async
|
||||
handlers both fire on some paths and would otherwise report one call
|
||||
twice. A settled background interaction is a genuinely second success
|
||||
event on the same request, so every such marker has to be cleared or
|
||||
the completion, the only event that carries usage and cost, is
|
||||
discarded as a duplicate of the in-progress create.
|
||||
"""
|
||||
self.model_call_details.pop("has_logged_async_success", None)
|
||||
litellm_params = self.model_call_details.get("litellm_params")
|
||||
if not isinstance(litellm_params, dict):
|
||||
return
|
||||
metadata = litellm_params.get("metadata")
|
||||
if not isinstance(metadata, dict):
|
||||
return
|
||||
otel_internal = metadata.get("_otel_internal")
|
||||
if not isinstance(otel_internal, dict):
|
||||
return
|
||||
spans_logged = otel_internal.get("spans_logged")
|
||||
if not isinstance(spans_logged, dict):
|
||||
return
|
||||
for scope in [key for key in spans_logged if isinstance(key, tuple) and key[-1:] == ("success",)]:
|
||||
del spans_logged[scope]
|
||||
|
||||
def _flush_passthrough_collected_chunks_helper(
|
||||
self,
|
||||
raw_bytes: list[bytes],
|
||||
|
|
@ -2282,7 +2375,9 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
is_sync_request: Final = self._is_sync_litellm_request(litellm_params)
|
||||
try:
|
||||
## BUILD COMPLETE STREAMED RESPONSE
|
||||
complete_streaming_response: ModelResponse | TextCompletionResponse | ResponsesAPIResponse | None = None
|
||||
complete_streaming_response: (
|
||||
ModelResponse | TextCompletionResponse | ResponsesAPIResponse | InteractionsAPIResponse | None
|
||||
) = None
|
||||
if "complete_streaming_response" in self.model_call_details:
|
||||
return # break out of this.
|
||||
complete_streaming_response = self._get_assembled_streaming_response(
|
||||
|
|
@ -2768,14 +2863,14 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
## BUILD COMPLETE STREAMED RESPONSE
|
||||
if "async_complete_streaming_response" in self.model_call_details:
|
||||
return # break out of this.
|
||||
complete_streaming_response: Final[ModelResponse | TextCompletionResponse | ResponsesAPIResponse | None] = (
|
||||
self._get_assembled_streaming_response(
|
||||
result=result,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
is_async=True,
|
||||
streaming_chunks=self.streaming_chunks,
|
||||
)
|
||||
complete_streaming_response: Final[
|
||||
ModelResponse | TextCompletionResponse | ResponsesAPIResponse | InteractionsAPIResponse | None
|
||||
] = self._get_assembled_streaming_response(
|
||||
result=result,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
is_async=True,
|
||||
streaming_chunks=self.streaming_chunks,
|
||||
)
|
||||
|
||||
if complete_streaming_response is not None:
|
||||
|
|
@ -3558,7 +3653,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
end_time: datetime.datetime,
|
||||
is_async: bool,
|
||||
streaming_chunks: list[object],
|
||||
) -> ModelResponse | TextCompletionResponse | ResponsesAPIResponse | None:
|
||||
) -> ModelResponse | TextCompletionResponse | ResponsesAPIResponse | InteractionsAPIResponse | None:
|
||||
if self.stream is not True:
|
||||
return None
|
||||
if isinstance(result, ModelResponse) or isinstance(result, TextCompletionResponse):
|
||||
|
|
@ -3583,9 +3678,40 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
),
|
||||
)
|
||||
return result.response
|
||||
elif isinstance(result, InteractionsAPIStreamingResponse):
|
||||
return self._assemble_completed_interaction_response(result)
|
||||
else:
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _assemble_completed_interaction_response(
|
||||
result: InteractionsAPIStreamingResponse,
|
||||
) -> InteractionsAPIResponse | None:
|
||||
"""
|
||||
The Interactions API streaming iterator hands the terminal event to the
|
||||
success handlers: the new schema (Api-Revision: 2026-05-20) emits
|
||||
``interaction.completed`` carrying the full interaction object, the
|
||||
legacy schema (2026-05-07) emits a chunk with ``status="completed"``
|
||||
and usage on the chunk itself. Build the equivalent non-streaming
|
||||
response so cost calculation and spend tracking see one shape.
|
||||
"""
|
||||
if result.event_type == "interaction.completed" and result.interaction is not None:
|
||||
return InteractionsAPIResponse(**result.interaction)
|
||||
if result.status == "completed":
|
||||
return InteractionsAPIResponse(
|
||||
**result.model_dump(
|
||||
exclude={ # mutable-ok: pydantic types exclude as set[str], which a frozenset does not satisfy
|
||||
"event_type",
|
||||
"delta",
|
||||
"index",
|
||||
"step",
|
||||
"interaction_id",
|
||||
"interaction",
|
||||
}
|
||||
)
|
||||
)
|
||||
return None
|
||||
|
||||
def _handle_anthropic_messages_response_logging(self, result: Any) -> ModelResponse:
|
||||
"""
|
||||
Handles logging for Anthropic messages responses.
|
||||
|
|
@ -5092,6 +5218,8 @@ class StandardLoggingPayloadSetup:
|
|||
elif isinstance(usage, dict):
|
||||
if ResponseAPILoggingUtils._is_response_api_usage(usage):
|
||||
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(usage)
|
||||
if InteractionsUsageObjectTransformation.is_interactions_usage_object(usage):
|
||||
return InteractionsUsageObjectTransformation.transform_interactions_usage_object(usage)
|
||||
return Usage(**usage)
|
||||
|
||||
raise ValueError(f"usage is required, got={usage} of type {type(usage)}")
|
||||
|
|
@ -5118,6 +5246,8 @@ class StandardLoggingPayloadSetup:
|
|||
if isinstance(_raw, dict):
|
||||
if ResponseAPILoggingUtils._is_response_api_usage(_raw):
|
||||
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_raw).model_dump()
|
||||
if InteractionsUsageObjectTransformation.is_interactions_usage_object(_raw):
|
||||
return InteractionsUsageObjectTransformation.transform_interactions_usage_object(_raw).model_dump()
|
||||
return _raw
|
||||
if isinstance(_raw, Usage):
|
||||
return _raw.model_dump()
|
||||
|
|
|
|||
|
|
@ -1,6 +1,9 @@
|
|||
from collections.abc import Mapping, Sequence
|
||||
from types import MappingProxyType
|
||||
from typing import Any
|
||||
|
||||
from litellm.types.utils import (
|
||||
CompletionTokensDetailsWrapper,
|
||||
PromptTokensDetailsWrapper,
|
||||
TranscriptionUsageDurationObject,
|
||||
TranscriptionUsageTokensObject,
|
||||
|
|
@ -34,3 +37,127 @@ class TranscriptionUsageObjectTransformation:
|
|||
),
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
_INTERACTIONS_MODALITY_FIELDS: Mapping[str, str] = MappingProxyType(
|
||||
{
|
||||
"text": "text_tokens",
|
||||
"audio": "audio_tokens",
|
||||
"image": "image_tokens",
|
||||
"video": "video_tokens",
|
||||
"document": "text_tokens",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _modality_field(entry: Mapping[str, Any]) -> str | None:
|
||||
return _INTERACTIONS_MODALITY_FIELDS.get(str(entry.get("modality", "")).lower())
|
||||
|
||||
|
||||
def _token_count(value: object) -> int:
|
||||
return value if isinstance(value, int) else 0
|
||||
|
||||
|
||||
def _modality_token_sums(entries: Sequence[Mapping[str, Any]]) -> Mapping[str, int]:
|
||||
fields = frozenset(field for entry in entries if (field := _modality_field(entry)) is not None)
|
||||
return MappingProxyType(
|
||||
{
|
||||
field: sum(_token_count(entry.get("tokens")) for entry in entries if _modality_field(entry) == field)
|
||||
for field in fields
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _google_search_query_count(usage_object: Mapping[str, Any]) -> int:
|
||||
return sum(
|
||||
_token_count(entry.get("count"))
|
||||
for entry in tuple(usage_object.get("grounding_tool_count") or ())
|
||||
if isinstance(entry, Mapping) and entry.get("type") == "google_search" # pyright: ignore[reportUnnecessaryIsInstance] # provider JSON, not the empty tuple inferred from `or ()`
|
||||
)
|
||||
|
||||
|
||||
def _subtract_cached_from_input(
|
||||
input_sums: Mapping[str, int],
|
||||
cached_sums: Mapping[str, int],
|
||||
total_cached_tokens: int,
|
||||
) -> Mapping[str, int]:
|
||||
if cached_sums:
|
||||
return MappingProxyType(
|
||||
{field: max(0, tokens - cached_sums.get(field, 0)) for field, tokens in input_sums.items()}
|
||||
)
|
||||
if total_cached_tokens and "text_tokens" in input_sums:
|
||||
return MappingProxyType(
|
||||
{
|
||||
**input_sums,
|
||||
"text_tokens": max(0, input_sums["text_tokens"] - total_cached_tokens),
|
||||
}
|
||||
)
|
||||
return input_sums
|
||||
|
||||
|
||||
class InteractionsUsageObjectTransformation:
|
||||
"""
|
||||
Maps the Google Interactions API usage block (total_input_tokens,
|
||||
output_tokens_by_modality, ...) into LiteLLM's chat-format ``Usage`` so the
|
||||
generic cost calculator and spend tracking can bill it.
|
||||
"""
|
||||
|
||||
@staticmethod
|
||||
def is_interactions_usage_object(usage_object: object) -> bool:
|
||||
if not isinstance(usage_object, dict):
|
||||
return False
|
||||
if "prompt_tokens" in usage_object or "input_tokens" in usage_object:
|
||||
return False
|
||||
return "total_input_tokens" in usage_object or "total_output_tokens" in usage_object
|
||||
|
||||
@staticmethod
|
||||
def transform_interactions_usage_object(usage_object: Mapping[str, Any]) -> Usage:
|
||||
input_entries = tuple(usage_object.get("input_tokens_by_modality") or ()) + tuple(
|
||||
usage_object.get("tool_use_tokens_by_modality") or ()
|
||||
)
|
||||
cached_sums = _modality_token_sums(tuple(usage_object.get("cached_tokens_by_modality") or ()))
|
||||
output_sums = _modality_token_sums(tuple(usage_object.get("output_tokens_by_modality") or ()))
|
||||
|
||||
total_cached_tokens = _token_count(usage_object.get("total_cached_tokens"))
|
||||
input_sums = _subtract_cached_from_input(
|
||||
input_sums=_modality_token_sums(input_entries),
|
||||
cached_sums=cached_sums,
|
||||
total_cached_tokens=total_cached_tokens,
|
||||
)
|
||||
|
||||
reasoning_tokens = _token_count(usage_object.get("total_reasoning_tokens")) or _token_count(
|
||||
usage_object.get("total_thought_tokens")
|
||||
)
|
||||
prompt_tokens = _token_count(usage_object.get("total_input_tokens")) + _token_count(
|
||||
usage_object.get("total_tool_use_tokens")
|
||||
)
|
||||
completion_tokens = _token_count(usage_object.get("total_output_tokens")) + reasoning_tokens
|
||||
total_tokens = _token_count(usage_object.get("total_tokens")) or (prompt_tokens + completion_tokens)
|
||||
|
||||
web_search_requests = _google_search_query_count(usage_object)
|
||||
prompt_tokens_details = (
|
||||
PromptTokensDetailsWrapper(
|
||||
cached_tokens=total_cached_tokens or None,
|
||||
web_search_requests=web_search_requests or None,
|
||||
**input_sums,
|
||||
)
|
||||
if input_sums or total_cached_tokens or web_search_requests
|
||||
else None
|
||||
)
|
||||
completion_tokens_details = (
|
||||
CompletionTokensDetailsWrapper(
|
||||
reasoning_tokens=reasoning_tokens or None,
|
||||
**output_sums,
|
||||
)
|
||||
if output_sums or reasoning_tokens
|
||||
else None
|
||||
)
|
||||
|
||||
return Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=total_tokens,
|
||||
prompt_tokens_details=prompt_tokens_details,
|
||||
completion_tokens_details=completion_tokens_details,
|
||||
cache_read_input_tokens=total_cached_tokens or None,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -28,8 +28,12 @@ from litellm.types.llms.openai import (
|
|||
ChatCompletionAssistantMessage,
|
||||
ChatCompletionFileObject,
|
||||
ChatCompletionImageObject,
|
||||
ChatCompletionReasoningItem,
|
||||
ChatCompletionReasoningSummaryTextBlock,
|
||||
ChatCompletionRedactedThinkingBlock,
|
||||
ChatCompletionResponseMessage,
|
||||
ChatCompletionTextObject,
|
||||
ChatCompletionThinkingBlock,
|
||||
ChatCompletionToolParam,
|
||||
ChatCompletionUserMessage,
|
||||
)
|
||||
|
|
@ -1549,6 +1553,44 @@ def _extract_reasoning_content(message: dict) -> tuple[str | None, str | None]:
|
|||
return None, message_content
|
||||
|
||||
|
||||
def _readable_thinking_text(
|
||||
block: ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock,
|
||||
) -> str:
|
||||
"""The text a chat model can read back, empty for redacted blocks and malformed ones."""
|
||||
if block.get("type") != "thinking":
|
||||
return ""
|
||||
thinking: Final = cast(ChatCompletionThinkingBlock, block).get("thinking") # cast-ok: narrowed by the type tag
|
||||
return str(thinking or "")
|
||||
|
||||
|
||||
def reasoning_content_from_thinking_blocks(
|
||||
thinking_blocks: Iterable[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock],
|
||||
) -> str:
|
||||
"""Flatten Anthropic thinking blocks into the `reasoning_content` string chat models expect.
|
||||
|
||||
Redacted blocks carry no readable text, so they contribute nothing.
|
||||
"""
|
||||
return "\n".join(text for block in thinking_blocks if (text := _readable_thinking_text(block)))
|
||||
|
||||
|
||||
def responses_reasoning_item_from_thinking_blocks(
|
||||
thinking_blocks: Iterable[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock],
|
||||
) -> ChatCompletionReasoningItem | None:
|
||||
"""Build a Responses API `reasoning` input item from Anthropic thinking blocks.
|
||||
|
||||
The item carries no `id`: the Responses API rejects an empty one and 404s on any id it
|
||||
did not mint itself, while an item without an id is always accepted.
|
||||
"""
|
||||
summary: Final[list[ChatCompletionReasoningSummaryTextBlock]] = [ # mutable-ok: API message payload
|
||||
ChatCompletionReasoningSummaryTextBlock(type="summary_text", text=text)
|
||||
for block in thinking_blocks
|
||||
if (text := _readable_thinking_text(block))
|
||||
]
|
||||
if not summary:
|
||||
return None
|
||||
return ChatCompletionReasoningItem(type="reasoning", summary=summary)
|
||||
|
||||
|
||||
def _parse_content_for_reasoning(
|
||||
message_text: str | None,
|
||||
) -> tuple[str | None, str | None]:
|
||||
|
|
|
|||
|
|
@ -64,6 +64,7 @@ from openai.types.chat.chat_completion_chunk import Choice as OpenAIStreamingCho
|
|||
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
parse_tool_call_arguments,
|
||||
reasoning_content_from_thinking_blocks,
|
||||
with_prompt_cache_breakpoint,
|
||||
)
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
|
|
@ -592,6 +593,9 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
assistant_message["tool_calls"] = tool_calls
|
||||
if len(thinking_blocks) > 0:
|
||||
assistant_message["thinking_blocks"] = thinking_blocks
|
||||
reasoning_content = reasoning_content_from_thinking_blocks(thinking_blocks)
|
||||
if reasoning_content:
|
||||
assistant_message["reasoning_content"] = reasoning_content
|
||||
new_messages.append(assistant_message)
|
||||
|
||||
return new_messages
|
||||
|
|
|
|||
|
|
@ -152,7 +152,10 @@ class AnthropicResponsesStreamWrapper:
|
|||
if block_idx < 0:
|
||||
if not delta:
|
||||
return
|
||||
block_idx = self._open_block(item_id, {"type": "thinking", "thinking": ""})
|
||||
block_idx = self._open_block(
|
||||
item_id,
|
||||
{"type": "thinking", "thinking": "", "signature": ""}, # mutable-ok: API message payload
|
||||
)
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_delta",
|
||||
|
|
|
|||
|
|
@ -6,12 +6,14 @@ path used for OpenAI and Azure models.
|
|||
"""
|
||||
|
||||
import json
|
||||
from collections.abc import Iterable
|
||||
from collections.abc import Iterable, Mapping
|
||||
from itertools import groupby
|
||||
from typing import Any, Final, cast
|
||||
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
TOOL_RESULT_IMAGE_BOUNDARY,
|
||||
TOOL_RESULT_IMAGE_PLACEHOLDER,
|
||||
responses_reasoning_item_from_thinking_blocks,
|
||||
with_prompt_cache_breakpoint,
|
||||
)
|
||||
from litellm.litellm_core_utils.reasoning_effort_utils import (
|
||||
|
|
@ -36,7 +38,11 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
|
|||
AnthropicMessagesResponse,
|
||||
AnthropicUsage,
|
||||
)
|
||||
from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse
|
||||
from litellm.types.llms.openai import (
|
||||
ChatCompletionThinkingBlock,
|
||||
ResponseAPIUsage,
|
||||
ResponsesAPIResponse,
|
||||
)
|
||||
|
||||
|
||||
class LiteLLMAnthropicToResponsesAPIAdapter:
|
||||
|
|
@ -100,6 +106,58 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
if isinstance(block, dict) and block.get("type") == "text" and (text := block.get("text")) # pyright: ignore[reportUnnecessaryIsInstance] # untrusted client payload
|
||||
]
|
||||
|
||||
@staticmethod
|
||||
def _summary_part_text(part: object) -> str:
|
||||
if isinstance(part, Mapping):
|
||||
mapping: Final = cast(Mapping[str, Any], part) # cast-ok: summary parts are untyped provider json
|
||||
return str(mapping.get("text") or "")
|
||||
return str(getattr(part, "text", None) or "")
|
||||
|
||||
@classmethod
|
||||
def _thinking_blocks_from_reasoning_item(
|
||||
cls,
|
||||
summary: Iterable[object],
|
||||
) -> tuple[dict[str, Any], ...]: # mutable-ok: API message payload
|
||||
"""Anthropic thinking blocks for one Responses reasoning item.
|
||||
|
||||
The signature stays empty: only Anthropic can sign a thinking block, and a stand-in
|
||||
value would be replayed as a real one and rejected by every backend that verifies it.
|
||||
"""
|
||||
return tuple(
|
||||
AnthropicResponseContentBlockThinking(
|
||||
type="thinking",
|
||||
thinking=text,
|
||||
signature=None,
|
||||
).model_dump()
|
||||
for part in summary
|
||||
if (text := cls._summary_part_text(part))
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _assistant_block_group_key(indexed_block: tuple[int, Mapping[str, Any]]) -> str:
|
||||
"""Group a run of consecutive thinking blocks together; keep every other block alone."""
|
||||
index, block = indexed_block
|
||||
return "thinking" if block.get("type") == "thinking" else f"block:{index}"
|
||||
|
||||
@classmethod
|
||||
def _assistant_group_to_input_item(
|
||||
cls, group: tuple[Mapping[str, Any], ...]
|
||||
) -> dict[str, Any] | None: # mutable-ok: API message payload
|
||||
first: Final = group[0]
|
||||
btype: Final = first.get("type")
|
||||
if btype == "thinking":
|
||||
blocks: Final = cast(tuple[ChatCompletionThinkingBlock, ...], group) # cast-ok: untrusted client payload
|
||||
reasoning_item: Final = responses_reasoning_item_from_thinking_blocks(blocks)
|
||||
return None if reasoning_item is None else dict(reasoning_item) # mutable-ok: API message payload
|
||||
if btype == "tool_use":
|
||||
return { # mutable-ok: API message payload
|
||||
"type": "function_call",
|
||||
"call_id": first.get("id", ""),
|
||||
"name": first.get("name", ""),
|
||||
"arguments": json.dumps(first.get("input", {})), # mutable-ok: API message payload
|
||||
}
|
||||
return None
|
||||
|
||||
def translate_messages_to_responses_input(
|
||||
self,
|
||||
messages: list[AllAnthropicPassThroughMessageValues],
|
||||
|
|
@ -113,6 +171,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
user image -> message(role=user, input_image)
|
||||
user tool_result -> function_call_output
|
||||
assistant text -> message(role=assistant, output_text)
|
||||
assistant thinking -> reasoning
|
||||
assistant tool_use -> function_call
|
||||
"""
|
||||
input_items: Final[list[dict[str, Any]]] = []
|
||||
|
|
@ -233,27 +292,17 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
}
|
||||
)
|
||||
elif isinstance(content, list):
|
||||
asst_parts: list[dict[str, Any]] = []
|
||||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
btype = block.get("type")
|
||||
if btype == "text":
|
||||
asst_parts.append({"type": "output_text", "text": block.get("text", "")})
|
||||
elif btype == "tool_use":
|
||||
# tool_use becomes a top-level function_call item
|
||||
input_items.append(
|
||||
{
|
||||
"type": "function_call",
|
||||
"call_id": block.get("id", ""),
|
||||
"name": block.get("name", ""),
|
||||
"arguments": json.dumps(block.get("input", {})),
|
||||
}
|
||||
)
|
||||
elif btype == "thinking":
|
||||
thinking_text = block.get("thinking", "")
|
||||
if thinking_text:
|
||||
asst_parts.append({"type": "output_text", "text": thinking_text})
|
||||
blocks = tuple(block for block in content if isinstance(block, dict))
|
||||
input_items.extend(
|
||||
item
|
||||
for _, group in groupby(enumerate(blocks), key=self._assistant_block_group_key)
|
||||
if (item := self._assistant_group_to_input_item(tuple(block for _, block in group))) is not None
|
||||
)
|
||||
asst_parts: list[dict[str, Any]] = [ # mutable-ok: API message payload
|
||||
{"type": "output_text", "text": block.get("text", "")} # mutable-ok: API message payload
|
||||
for block in blocks
|
||||
if block.get("type") == "text"
|
||||
]
|
||||
if asst_parts:
|
||||
input_items.append(
|
||||
{
|
||||
|
|
@ -514,16 +563,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
|
||||
for item in response.output:
|
||||
if isinstance(item, ResponseReasoningItem):
|
||||
for summary in item.summary:
|
||||
text = getattr(summary, "text", "")
|
||||
if text:
|
||||
content.append(
|
||||
AnthropicResponseContentBlockThinking(
|
||||
type="thinking",
|
||||
thinking=text,
|
||||
signature=None,
|
||||
).model_dump()
|
||||
)
|
||||
content.extend(self._thinking_blocks_from_reasoning_item(item.summary))
|
||||
|
||||
elif isinstance(item, ResponseOutputMessage):
|
||||
for part in item.content:
|
||||
|
|
@ -555,6 +595,12 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
content.append(
|
||||
AnthropicResponseContentBlockText(type="text", text=part.get("text", "")).model_dump()
|
||||
)
|
||||
elif item_type == "reasoning":
|
||||
content.extend(
|
||||
self._thinking_blocks_from_reasoning_item(
|
||||
cast(Iterable[object], item.get("summary") or ()), # cast-ok: untyped provider json
|
||||
)
|
||||
)
|
||||
elif item_type == "function_call":
|
||||
try:
|
||||
input_data = json.loads(item.get("arguments", "{}"))
|
||||
|
|
|
|||
|
|
@ -30,7 +30,12 @@ class AzureFoundryErrorStrings(str, enum.Enum):
|
|||
SET_EXTRA_PARAMETERS_TO_PASS_THROUGH = "Set extra-parameters to 'pass-through'"
|
||||
|
||||
|
||||
NON_OPENAI_SPEC_MESSAGE_FIELDS: Final = ("thinking_blocks", "provider_specific_fields", "cache_control")
|
||||
NON_OPENAI_SPEC_MESSAGE_FIELDS: Final = (
|
||||
"thinking_blocks",
|
||||
"reasoning_content",
|
||||
"provider_specific_fields",
|
||||
"cache_control",
|
||||
)
|
||||
|
||||
|
||||
class AzureAIStudioConfig(OpenAIConfig):
|
||||
|
|
@ -173,7 +178,8 @@ class AzureAIStudioConfig(OpenAIConfig):
|
|||
"""
|
||||
- Azure AI Studio doesn't support content as a list. This handles:
|
||||
1. Strips message fields that are not part of the OpenAI chat-completions
|
||||
schema (thinking_blocks, provider_specific_fields, cache_control).
|
||||
schema (thinking_blocks, reasoning_content, provider_specific_fields,
|
||||
cache_control).
|
||||
Azure AI Foundry backends set additionalProperties=false and reject
|
||||
these with "Extra inputs are not permitted", which breaks multi-turn
|
||||
Anthropic-format clients that echo thinking blocks back as history.
|
||||
|
|
|
|||
|
|
@ -3,10 +3,31 @@ Helper util for handling databricks-specific cost calculation
|
|||
- e.g.: handling 'dbrx-instruct-*'
|
||||
"""
|
||||
|
||||
from types import MappingProxyType
|
||||
from typing import Final
|
||||
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
|
||||
from litellm.types.utils import Usage
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
_LEGACY_ENDPOINT_NAMES: Final = MappingProxyType(
|
||||
{
|
||||
"dbrx-instruct": "databricks-dbrx-instruct",
|
||||
"meta-llama-3.1-70b-instruct": "databricks-meta-llama-3-1-70b-instruct",
|
||||
"meta-llama-3.1-405b-instruct": "databricks-meta-llama-3-1-405b-instruct",
|
||||
"mixtral-8x7b-instruct-v0.1": "databricks-mixtral-8x7b-instruct",
|
||||
"bge-large-en": "databricks-bge-large-en",
|
||||
"gte-large-en": "databricks-gte-large-en",
|
||||
"llama-2-70b-chat": "databricks-llama-2-70b-chat",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _registry_key(model: str) -> str:
|
||||
name: Final = model.removeprefix("databricks/")
|
||||
return next(
|
||||
(key for prefix, key in _LEGACY_ENDPOINT_NAMES.items() if name.startswith(prefix)),
|
||||
name,
|
||||
)
|
||||
|
||||
|
||||
def cost_per_token(model: str, usage: Usage) -> tuple[float, float]:
|
||||
|
|
@ -20,36 +41,8 @@ def cost_per_token(model: str, usage: Usage) -> tuple[float, float]:
|
|||
Returns:
|
||||
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
|
||||
"""
|
||||
base_model = model
|
||||
if model.startswith("databricks/dbrx-instruct") or model.startswith("dbrx-instruct"):
|
||||
base_model = "databricks-dbrx-instruct"
|
||||
elif model.startswith("databricks/meta-llama-3.1-70b-instruct") or model.startswith("meta-llama-3.1-70b-instruct"):
|
||||
base_model = "databricks-meta-llama-3-1-70b-instruct"
|
||||
elif model.startswith("databricks/meta-llama-3.1-405b-instruct") or model.startswith(
|
||||
"meta-llama-3.1-405b-instruct"
|
||||
):
|
||||
base_model = "databricks-meta-llama-3-1-405b-instruct"
|
||||
elif (
|
||||
model.startswith("databricks/mixtral-8x7b-instruct-v0.1")
|
||||
or model.startswith("mixtral-8x7b-instruct-v0.1")
|
||||
or model.startswith("databricks/mixtral-8x7b-instruct-v0.1")
|
||||
or model.startswith("mixtral-8x7b-instruct-v0.1")
|
||||
):
|
||||
base_model = "databricks-mixtral-8x7b-instruct"
|
||||
elif model.startswith("databricks/bge-large-en") or model.startswith("bge-large-en"):
|
||||
base_model = "databricks-bge-large-en"
|
||||
elif model.startswith("databricks/gte-large-en") or model.startswith("gte-large-en"):
|
||||
base_model = "databricks-gte-large-en"
|
||||
elif model.startswith("databricks/llama-2-70b-chat") or model.startswith("llama-2-70b-chat"):
|
||||
base_model = "databricks-llama-2-70b-chat"
|
||||
## GET MODEL INFO
|
||||
model_info: Final = get_model_info(model=base_model, custom_llm_provider="databricks")
|
||||
|
||||
## CALCULATE INPUT COST
|
||||
|
||||
prompt_cost: Final[float] = usage["prompt_tokens"] * model_info["input_cost_per_token"]
|
||||
|
||||
## CALCULATE OUTPUT COST
|
||||
completion_cost: Final = usage["completion_tokens"] * model_info["output_cost_per_token"]
|
||||
|
||||
return prompt_cost, completion_cost
|
||||
return generic_cost_per_token(
|
||||
model=_registry_key(model),
|
||||
usage=usage,
|
||||
custom_llm_provider="databricks",
|
||||
)
|
||||
|
|
|
|||
|
|
@ -504,6 +504,7 @@ class FireworksAIConfig(FireworksAIMixin, OpenAIGPTConfig):
|
|||
m = cast(dict, message)
|
||||
m.pop("provider_specific_fields", None)
|
||||
m.pop("thinking_blocks", None)
|
||||
m.pop("reasoning_content", None)
|
||||
|
||||
return messages
|
||||
|
||||
|
|
|
|||
|
|
@ -164,12 +164,13 @@ class HostedVLLMChatConfig(OpenAIGPTConfig):
|
|||
"""
|
||||
Support translating:
|
||||
- video files from file_id or file_data to video_url
|
||||
- thinking_blocks on assistant messages are removed, and content lists
|
||||
are converted to strings for vLLM compatibility
|
||||
- thinking_blocks and reasoning_content on assistant messages are removed,
|
||||
and content lists are converted to strings for vLLM compatibility
|
||||
"""
|
||||
for message in messages:
|
||||
if message["role"] == "assistant":
|
||||
message.pop("thinking_blocks", None)
|
||||
message.pop("reasoning_content", None)
|
||||
existing_content = message.get("content")
|
||||
if isinstance(existing_content, list):
|
||||
text_parts = []
|
||||
|
|
|
|||
|
|
@ -14551,6 +14551,8 @@
|
|||
]
|
||||
},
|
||||
"databricks/databricks-bge-large-en": {
|
||||
"cache_creation_input_token_cost": 1.0003e-07,
|
||||
"cache_read_input_token_cost": 1.0003e-07,
|
||||
"input_cost_per_token": 1.0003e-07,
|
||||
"input_dbu_cost_per_token": 1.429e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14566,6 +14568,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-claude-3-7-sonnet": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14581,10 +14585,41 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-fable-5": {
|
||||
"cache_creation_input_token_cost": 1.250004e-05,
|
||||
"cache_read_input_token_cost": 1.00002e-06,
|
||||
"input_cost_per_token": 1.000006e-05,
|
||||
"input_dbu_cost_per_token": 0.000142858,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 5.000002e-05,
|
||||
"output_dbu_cost_per_token": 0.000714286,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_mid_conversation_system": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false,
|
||||
"thinking_always_on": true
|
||||
},
|
||||
"databricks/databricks-claude-haiku-4-5": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.0003e-07,
|
||||
"input_cost_per_token": 1.00002e-06,
|
||||
"input_dbu_cost_per_token": 1.4286e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14600,10 +14635,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4": {
|
||||
"cache_creation_input_token_cost": 1.874999e-05,
|
||||
"cache_read_input_token_cost": 1.50003e-06,
|
||||
"input_cost_per_token": 1.5000020000000002e-05,
|
||||
"input_dbu_cost_per_token": 0.000214286,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14619,10 +14657,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-1": {
|
||||
"cache_creation_input_token_cost": 1.874999e-05,
|
||||
"cache_read_input_token_cost": 1.50003e-06,
|
||||
"input_cost_per_token": 1.5000020000000002e-05,
|
||||
"input_dbu_cost_per_token": 0.000214286,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14638,10 +14679,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-5": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14657,11 +14701,14 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_output_config": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-6": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14677,10 +14724,93 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.500001e-05,
|
||||
"output_dbu_cost_per_token": 0.000357143,
|
||||
"prompt_cache_min_tokens": 2048,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-8": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.500001e-05,
|
||||
"output_dbu_cost_per_token": 0.000357143,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_mid_conversation_system": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-5": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.500001e-05,
|
||||
"output_dbu_cost_per_token": 0.000357143,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_mid_conversation_system": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-4": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14696,10 +14826,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-4-1": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14715,10 +14848,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-4-5": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14734,10 +14870,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14753,10 +14892,40 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-5": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.99999e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields. Introductory launch rates of 28.571 input / 142.857 output / 35.714 cache write / 2.857 cache read DBU run through 2026-08-31; the standard rates are listed here because entries carry no expiry date."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.500002e-05,
|
||||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_mid_conversation_system": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-gemini-2-5-flash": {
|
||||
"cache_creation_input_token_cost": 3.0002e-07,
|
||||
"cache_read_input_token_cost": 3.0002e-08,
|
||||
"input_cost_per_token": 3.0001999999999996e-07,
|
||||
"input_dbu_cost_per_token": 4.285999999999999e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14771,9 +14940,12 @@
|
|||
"output_dbu_cost_per_token": 3.5714e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-2-5-pro": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.24999e-07,
|
||||
"input_cost_per_token": 1.24999e-06,
|
||||
"input_dbu_cost_per_token": 1.7857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14788,9 +14960,12 @@
|
|||
"output_dbu_cost_per_token": 0.000142857,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-3-1-flash-lite": {
|
||||
"cache_creation_input_token_cost": 3.1248e-07,
|
||||
"cache_read_input_token_cost": 3.122e-08,
|
||||
"input_cost_per_token": 3.1248e-07,
|
||||
"input_dbu_cost_per_token": 4.464e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14805,9 +14980,12 @@
|
|||
"output_dbu_cost_per_token": 2.6786e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-3-1-pro": {
|
||||
"cache_creation_input_token_cost": 2.49998e-06,
|
||||
"cache_read_input_token_cost": 2.4997e-07,
|
||||
"input_cost_per_token": 2.49998e-06,
|
||||
"input_dbu_cost_per_token": 3.5714e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14822,9 +15000,12 @@
|
|||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-3-flash": {
|
||||
"cache_creation_input_token_cost": 6.2503e-07,
|
||||
"cache_read_input_token_cost": 6.251e-08,
|
||||
"input_cost_per_token": 6.2503e-07,
|
||||
"input_dbu_cost_per_token": 8.929e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14839,9 +15020,12 @@
|
|||
"output_dbu_cost_per_token": 5.3571e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-3-pro": {
|
||||
"cache_creation_input_token_cost": 2.49998e-06,
|
||||
"cache_read_input_token_cost": 2.4997e-07,
|
||||
"input_cost_per_token": 2.49998e-06,
|
||||
"input_dbu_cost_per_token": 3.5714e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14856,9 +15040,12 @@
|
|||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemma-3-12b": {
|
||||
"cache_creation_input_token_cost": 1.5001e-07,
|
||||
"cache_read_input_token_cost": 1.5001e-07,
|
||||
"input_cost_per_token": 1.5000999999999998e-07,
|
||||
"input_dbu_cost_per_token": 2.1429999999999996e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14874,6 +15061,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-gpt-5": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.2502e-07,
|
||||
"input_cost_per_token": 1.24999e-06,
|
||||
"input_dbu_cost_per_token": 1.7857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14886,9 +15075,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 9.999990000000002e-06,
|
||||
"output_dbu_cost_per_token": 0.000142857,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-1": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.2502e-07,
|
||||
"input_cost_per_token": 1.24999e-06,
|
||||
"input_dbu_cost_per_token": 1.7857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14901,9 +15093,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 9.999990000000002e-06,
|
||||
"output_dbu_cost_per_token": 0.000142857,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-1-codex-max": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.2502e-07,
|
||||
"input_cost_per_token": 1.24999e-06,
|
||||
"input_dbu_cost_per_token": 1.7857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14916,9 +15111,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 9.999990000000002e-06,
|
||||
"output_dbu_cost_per_token": 0.000142857,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-1-codex-mini": {
|
||||
"cache_creation_input_token_cost": 2.4997e-07,
|
||||
"cache_read_input_token_cost": 2.499e-08,
|
||||
"input_cost_per_token": 2.4997e-07,
|
||||
"input_dbu_cost_per_token": 3.571e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14931,9 +15129,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.99997e-06,
|
||||
"output_dbu_cost_per_token": 2.8571e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-2": {
|
||||
"cache_creation_input_token_cost": 1.75e-06,
|
||||
"cache_read_input_token_cost": 1.75e-07,
|
||||
"input_cost_per_token": 1.75e-06,
|
||||
"input_dbu_cost_per_token": 2.5e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14946,9 +15147,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"output_dbu_cost_per_token": 0.0002,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-2-codex": {
|
||||
"cache_creation_input_token_cost": 1.75e-06,
|
||||
"cache_read_input_token_cost": 1.75e-07,
|
||||
"input_cost_per_token": 1.75e-06,
|
||||
"input_dbu_cost_per_token": 2.5e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14961,9 +15165,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"output_dbu_cost_per_token": 0.0002,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-3-codex": {
|
||||
"cache_creation_input_token_cost": 1.75e-06,
|
||||
"cache_read_input_token_cost": 1.75e-07,
|
||||
"input_cost_per_token": 1.75e-06,
|
||||
"input_dbu_cost_per_token": 2.5e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14976,9 +15183,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"output_dbu_cost_per_token": 0.0002,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-4": {
|
||||
"cache_creation_input_token_cost": 2.49998e-06,
|
||||
"cache_read_input_token_cost": 2.4997e-07,
|
||||
"input_cost_per_token": 2.49998e-06,
|
||||
"input_dbu_cost_per_token": 3.5714e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14991,9 +15201,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.5000020000000002e-05,
|
||||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-4-mini": {
|
||||
"cache_creation_input_token_cost": 7.4998e-07,
|
||||
"cache_read_input_token_cost": 7.497e-08,
|
||||
"input_cost_per_token": 7.4998e-07,
|
||||
"input_dbu_cost_per_token": 1.0714e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15006,9 +15219,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 4.50002e-06,
|
||||
"output_dbu_cost_per_token": 6.4286e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-4-nano": {
|
||||
"cache_creation_input_token_cost": 1.9999e-07,
|
||||
"cache_read_input_token_cost": 2.002e-08,
|
||||
"input_cost_per_token": 1.9999e-07,
|
||||
"input_dbu_cost_per_token": 2.857e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15021,9 +15237,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.24999e-06,
|
||||
"output_dbu_cost_per_token": 1.7857e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-mini": {
|
||||
"cache_creation_input_token_cost": 2.4997e-07,
|
||||
"cache_read_input_token_cost": 2.499e-08,
|
||||
"input_cost_per_token": 2.4997000000000006e-07,
|
||||
"input_dbu_cost_per_token": 3.571e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15036,9 +15255,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.9999700000000004e-06,
|
||||
"output_dbu_cost_per_token": 2.8571e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-nano": {
|
||||
"cache_creation_input_token_cost": 4.998e-08,
|
||||
"cache_read_input_token_cost": 4.97e-09,
|
||||
"input_cost_per_token": 4.998e-08,
|
||||
"input_dbu_cost_per_token": 7.14e-07,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15051,9 +15273,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.9998000000000007e-07,
|
||||
"output_dbu_cost_per_token": 5.714000000000001e-06,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-oss-120b": {
|
||||
"cache_creation_input_token_cost": 1.5001e-07,
|
||||
"cache_read_input_token_cost": 1.5001e-07,
|
||||
"input_cost_per_token": 1.5000999999999998e-07,
|
||||
"input_dbu_cost_per_token": 2.1429999999999996e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15069,6 +15294,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-gpt-oss-20b": {
|
||||
"cache_creation_input_token_cost": 7e-08,
|
||||
"cache_read_input_token_cost": 7e-08,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"input_dbu_cost_per_token": 1e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15084,6 +15311,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-gte-large-en": {
|
||||
"cache_creation_input_token_cost": 1.2999e-07,
|
||||
"cache_read_input_token_cost": 1.2999e-07,
|
||||
"input_cost_per_token": 1.2999000000000001e-07,
|
||||
"input_dbu_cost_per_token": 1.857e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15099,6 +15328,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-llama-2-70b-chat": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15115,6 +15346,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-llama-4-maverick": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15131,6 +15364,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-meta-llama-3-1-405b-instruct": {
|
||||
"cache_creation_input_token_cost": 5.00003e-06,
|
||||
"cache_read_input_token_cost": 5.00003e-06,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15147,6 +15382,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-meta-llama-3-1-8b-instruct": {
|
||||
"cache_creation_input_token_cost": 1.5001e-07,
|
||||
"cache_read_input_token_cost": 1.5001e-07,
|
||||
"input_cost_per_token": 1.5000999999999998e-07,
|
||||
"input_dbu_cost_per_token": 2.1429999999999996e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15162,6 +15399,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-meta-llama-3-3-70b-instruct": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15178,6 +15417,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-meta-llama-3-70b-instruct": {
|
||||
"cache_creation_input_token_cost": 1.00002e-06,
|
||||
"cache_read_input_token_cost": 1.00002e-06,
|
||||
"input_cost_per_token": 1.00002e-06,
|
||||
"input_dbu_cost_per_token": 1.4286e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15194,6 +15435,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-mixtral-8x7b-instruct": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15210,6 +15453,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-mpt-30b-instruct": {
|
||||
"cache_creation_input_token_cost": 1.00002e-06,
|
||||
"cache_read_input_token_cost": 1.00002e-06,
|
||||
"input_cost_per_token": 1.00002e-06,
|
||||
"input_dbu_cost_per_token": 1.4286e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15226,6 +15471,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-mpt-7b-instruct": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
|
|||
|
|
@ -6,21 +6,21 @@ Plugins are stored as metadata + git source references in LiteLLM database.
|
|||
Actual plugin files are hosted on GitHub/GitLab/Bitbucket.
|
||||
|
||||
Endpoints:
|
||||
/claude-code/marketplace.json - GET - List plugins for Claude Code discovery
|
||||
/claude-code/plugins - POST - Register a new plugin (create-only)
|
||||
/claude-code/plugins - GET - List plugins (admin)
|
||||
/claude-code/plugins/{name} - GET - Get plugin details
|
||||
/claude-code/plugins/{name} - PUT - Update an existing plugin
|
||||
/claude-code/plugins/{name}/enable - POST - Enable a plugin
|
||||
/claude-code/plugins/{name}/disable - POST - Disable a plugin
|
||||
/claude-code/plugins/{name} - DELETE - Delete a plugin
|
||||
/claude-code/marketplace.json - GET - List plugins for Claude Code discovery (unauthenticated)
|
||||
/claude-code/plugins - POST - Register a new plugin (create-only, proxy admin only)
|
||||
/claude-code/plugins - GET - List plugins (any authenticated key)
|
||||
/claude-code/plugins/{name} - GET - Get plugin details (any authenticated key)
|
||||
/claude-code/plugins/{name} - PUT - Update an existing plugin (proxy admin only)
|
||||
/claude-code/plugins/{name}/enable - POST - Enable a plugin (proxy admin only)
|
||||
/claude-code/plugins/{name}/disable - POST - Disable a plugin (proxy admin only)
|
||||
/claude-code/plugins/{name} - DELETE - Delete a plugin (proxy admin only)
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
from collections.abc import Mapping, Sequence
|
||||
from datetime import datetime, timezone
|
||||
from typing import Final, Protocol, TypedDict
|
||||
from typing import Annotated, Final, Protocol, TypedDict
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException
|
||||
from fastapi.responses import JSONResponse
|
||||
|
|
@ -28,6 +28,7 @@ from fastapi.responses import JSONResponse
|
|||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.common_utils.resource_ownership import is_proxy_admin
|
||||
from litellm.repositories.table_repositories import ClaudeCodePluginRepository
|
||||
from litellm.types.proxy.claude_code_endpoints import (
|
||||
ListPluginsResponse,
|
||||
|
|
@ -221,6 +222,18 @@ def _name_conflict_error(name: str) -> HTTPException:
|
|||
)
|
||||
|
||||
|
||||
def _require_proxy_admin(user_api_key_dict: UserAPIKeyAuth) -> None:
|
||||
"""Catalog mutations are restricted to proxy admins: marketplace.json is served
|
||||
unauthenticated and any registered/updated entry is immediately installable by
|
||||
every user, so a non-admin key must never be able to add or overwrite one.
|
||||
"""
|
||||
if not is_proxy_admin(user_api_key_dict):
|
||||
raise HTTPException(
|
||||
status_code=403,
|
||||
detail={"error": "Only proxy admins may modify the Claude Code plugin marketplace."},
|
||||
)
|
||||
|
||||
|
||||
@router.post(
|
||||
"/claude-code/plugins",
|
||||
tags=["Claude Code Marketplace"],
|
||||
|
|
@ -242,6 +255,8 @@ async def register_plugin(
|
|||
the same name already exists it returns 409 Conflict; use
|
||||
PUT /claude-code/plugins/{plugin_name} to update an existing plugin.
|
||||
|
||||
Requires a proxy admin API key.
|
||||
|
||||
Parameters:
|
||||
- name: Plugin name (kebab-case)
|
||||
- source: Git source reference (github, url, or git-subdir format)
|
||||
|
|
@ -271,6 +286,8 @@ async def register_plugin(
|
|||
from prisma.errors import UniqueViolationError
|
||||
|
||||
try:
|
||||
_require_proxy_admin(user_api_key_dict)
|
||||
|
||||
prisma_client: Final = await _get_prisma_client()
|
||||
|
||||
if not re.match(r"^[a-z0-9-]+$", request.name):
|
||||
|
|
@ -468,6 +485,7 @@ async def get_plugin(
|
|||
async def update_plugin(
|
||||
plugin_name: str,
|
||||
request: UpdatePluginRequest,
|
||||
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
|
||||
):
|
||||
"""
|
||||
Update an existing plugin in the LiteLLM marketplace.
|
||||
|
|
@ -481,6 +499,8 @@ async def update_plugin(
|
|||
Returns 404 if no plugin with the given name exists; use
|
||||
POST /claude-code/plugins to create a new plugin.
|
||||
|
||||
Requires a proxy admin API key.
|
||||
|
||||
Parameters:
|
||||
- plugin_name: Name of the plugin to update (path parameter)
|
||||
- source: Git source reference (github, url, or git-subdir format)
|
||||
|
|
@ -509,6 +529,8 @@ async def update_plugin(
|
|||
from prisma.errors import PrismaError
|
||||
|
||||
try:
|
||||
_require_proxy_admin(user_api_key_dict)
|
||||
|
||||
prisma_client: Final = await _get_prisma_client()
|
||||
|
||||
_validate_plugin_source(request.source)
|
||||
|
|
@ -566,10 +588,14 @@ async def enable_plugin(
|
|||
"""
|
||||
Enable a disabled plugin.
|
||||
|
||||
Requires a proxy admin API key.
|
||||
|
||||
Parameters:
|
||||
- plugin_name: The name of the plugin to enable
|
||||
"""
|
||||
try:
|
||||
_require_proxy_admin(user_api_key_dict)
|
||||
|
||||
prisma_client: Final = await _get_prisma_client()
|
||||
|
||||
plugin: Final[_PluginRecord | None] = await ClaudeCodePluginRepository(prisma_client).table.find_unique(
|
||||
|
|
@ -611,10 +637,14 @@ async def disable_plugin(
|
|||
"""
|
||||
Disable a plugin without deleting it.
|
||||
|
||||
Requires a proxy admin API key.
|
||||
|
||||
Parameters:
|
||||
- plugin_name: The name of the plugin to disable
|
||||
"""
|
||||
try:
|
||||
_require_proxy_admin(user_api_key_dict)
|
||||
|
||||
prisma_client: Final = await _get_prisma_client()
|
||||
|
||||
plugin: Final[_PluginRecord | None] = await ClaudeCodePluginRepository(prisma_client).table.find_unique(
|
||||
|
|
@ -656,10 +686,14 @@ async def delete_plugin(
|
|||
"""
|
||||
Delete a plugin from the marketplace.
|
||||
|
||||
Requires a proxy admin API key.
|
||||
|
||||
Parameters:
|
||||
- plugin_name: The name of the plugin to delete
|
||||
"""
|
||||
try:
|
||||
_require_proxy_admin(user_api_key_dict)
|
||||
|
||||
prisma_client: Final = await _get_prisma_client()
|
||||
|
||||
plugin: Final[_PluginRecord | None] = await ClaudeCodePluginRepository(prisma_client).table.find_unique(
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ from litellm.constants import (
|
|||
DEFAULT_HEALTH_CHECK_PROMPT,
|
||||
HEALTH_CHECK_TIMEOUT_SECONDS,
|
||||
)
|
||||
from litellm.router_utils.auto_router_model_naming import classify_strategy_router_model
|
||||
|
||||
ILLEGAL_DISPLAY_PARAMS: Final = [
|
||||
"messages",
|
||||
|
|
@ -182,30 +183,17 @@ async def run_with_timeout(task, timeout):
|
|||
return {"error": "Timeout exceeded", "exception": timeout_exception}
|
||||
|
||||
|
||||
def _is_semantic_auto_router_deployment(litellm_params: dict) -> bool:
|
||||
"""
|
||||
True for semantic auto_router deployments (auto_router/<name>) that are not
|
||||
sub-strategies (complexity_router, adaptive_router, quality_router).
|
||||
|
||||
These are meta-routers that select among real LLM deployments at request time;
|
||||
they have no LLM endpoint to health-check.
|
||||
"""
|
||||
def _is_strategy_router_deployment(litellm_params: dict) -> bool:
|
||||
"""True for strategy-router deployments."""
|
||||
model: Final[object] = litellm_params.get("model", "")
|
||||
if not isinstance(model, str):
|
||||
return False
|
||||
if not model.startswith("auto_router/"):
|
||||
return False
|
||||
for sub_strategy in ("complexity_router", "adaptive_router", "quality_router"):
|
||||
if model.startswith(f"auto_router/{sub_strategy}"):
|
||||
return False
|
||||
return True
|
||||
return isinstance(model, str) and classify_strategy_router_model(model) is not None
|
||||
|
||||
|
||||
async def _run_model_health_check(model: dict):
|
||||
litellm_params = model["litellm_params"]
|
||||
model_info: Final = model.get("model_info", {})
|
||||
|
||||
if _is_semantic_auto_router_deployment(litellm_params):
|
||||
if _is_strategy_router_deployment(litellm_params):
|
||||
return {}
|
||||
|
||||
mode: Final = _resolve_health_check_mode(
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ from typing import Any, Final, cast
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.constants import BACKGROUND_INTERACTION_COST_POLLING_ENABLED
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.litellm_core_utils.core_helpers import (
|
||||
_get_parent_otel_span_from_kwargs,
|
||||
|
|
@ -318,6 +319,21 @@ class _ProxyDBLogger(CustomLogger):
|
|||
elif budget_reservation is not None:
|
||||
await _release_budget_reservation(budget_reservation=budget_reservation)
|
||||
else:
|
||||
if _is_unbilled_interaction_response(completion_response):
|
||||
if BACKGROUND_INTERACTION_COST_POLLING_ENABLED and _is_unbilled_in_progress_interaction(
|
||||
completion_response
|
||||
):
|
||||
verbose_proxy_logger.debug(
|
||||
"Cost tracking deferred for in-progress background interaction; "
|
||||
"the budget reservation stays open until the poll task logs the final usage"
|
||||
)
|
||||
return
|
||||
await _release_budget_reservation(budget_reservation=budget_reservation)
|
||||
verbose_proxy_logger.debug(
|
||||
"Released the budget reservation for an interaction create with no usage "
|
||||
"that no poll task will settle"
|
||||
)
|
||||
return
|
||||
await _release_budget_reservation(budget_reservation=budget_reservation)
|
||||
# Non-model call types (health checks, afile_delete) have no model or standard_logging_object.
|
||||
# Use .get() for "stream" to avoid KeyError on health checks.
|
||||
|
|
@ -463,6 +479,24 @@ def _write_spend_metadata_to_kwargs(kwargs: dict, metadata: dict) -> None:
|
|||
bucket[key] = value
|
||||
|
||||
|
||||
def _is_unbilled_interaction_response(completion_response: object) -> bool:
|
||||
from litellm.interactions.background_cost_polling import missing_usage_is_expected
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
if not isinstance(completion_response, InteractionsAPIResponse):
|
||||
return False
|
||||
return completion_response.usage is None and missing_usage_is_expected(completion_response)
|
||||
|
||||
|
||||
def _is_unbilled_in_progress_interaction(completion_response: object) -> bool:
|
||||
from litellm.interactions.background_cost_polling import is_pollable_background_interaction
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
if not isinstance(completion_response, InteractionsAPIResponse):
|
||||
return False
|
||||
return completion_response.usage is None and is_pollable_background_interaction(completion_response)
|
||||
|
||||
|
||||
def _should_track_cost_callback(
|
||||
user_api_key: str | None,
|
||||
user_id: str | None,
|
||||
|
|
|
|||
|
|
@ -440,6 +440,12 @@ class CallTypes(str, Enum):
|
|||
query = "query"
|
||||
aquery = "aquery"
|
||||
|
||||
#########################################################
|
||||
# Google Interactions API Call Types
|
||||
#########################################################
|
||||
create_interaction = "create_interaction"
|
||||
acreate_interaction = "acreate_interaction"
|
||||
|
||||
#########################################################
|
||||
# Container Call Types
|
||||
#########################################################
|
||||
|
|
|
|||
|
|
@ -14551,6 +14551,8 @@
|
|||
]
|
||||
},
|
||||
"databricks/databricks-bge-large-en": {
|
||||
"cache_creation_input_token_cost": 1.0003e-07,
|
||||
"cache_read_input_token_cost": 1.0003e-07,
|
||||
"input_cost_per_token": 1.0003e-07,
|
||||
"input_dbu_cost_per_token": 1.429e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14566,6 +14568,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-claude-3-7-sonnet": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14581,10 +14585,41 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-fable-5": {
|
||||
"cache_creation_input_token_cost": 1.250004e-05,
|
||||
"cache_read_input_token_cost": 1.00002e-06,
|
||||
"input_cost_per_token": 1.000006e-05,
|
||||
"input_dbu_cost_per_token": 0.000142858,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 5.000002e-05,
|
||||
"output_dbu_cost_per_token": 0.000714286,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_mid_conversation_system": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false,
|
||||
"thinking_always_on": true
|
||||
},
|
||||
"databricks/databricks-claude-haiku-4-5": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.0003e-07,
|
||||
"input_cost_per_token": 1.00002e-06,
|
||||
"input_dbu_cost_per_token": 1.4286e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14600,10 +14635,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4": {
|
||||
"cache_creation_input_token_cost": 1.874999e-05,
|
||||
"cache_read_input_token_cost": 1.50003e-06,
|
||||
"input_cost_per_token": 1.5000020000000002e-05,
|
||||
"input_dbu_cost_per_token": 0.000214286,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14619,10 +14657,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-1": {
|
||||
"cache_creation_input_token_cost": 1.874999e-05,
|
||||
"cache_read_input_token_cost": 1.50003e-06,
|
||||
"input_cost_per_token": 1.5000020000000002e-05,
|
||||
"input_dbu_cost_per_token": 0.000214286,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14638,10 +14679,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-5": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14657,11 +14701,14 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_output_config": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-6": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14677,10 +14724,93 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-7": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.500001e-05,
|
||||
"output_dbu_cost_per_token": 0.000357143,
|
||||
"prompt_cache_min_tokens": 2048,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-4-8": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.500001e-05,
|
||||
"output_dbu_cost_per_token": 0.000357143,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_mid_conversation_system": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-claude-opus-5": {
|
||||
"cache_creation_input_token_cost": 6.25002e-06,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.500001e-05,
|
||||
"output_dbu_cost_per_token": 0.000357143,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_mid_conversation_system": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-4": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14696,10 +14826,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-4-1": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14715,10 +14848,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-4-5": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14734,10 +14870,13 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-4-6": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.9999900000000002e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14753,10 +14892,40 @@
|
|||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-claude-sonnet-5": {
|
||||
"cache_creation_input_token_cost": 3.74997e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.99999e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"metadata": {
|
||||
"notes": "Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields. Introductory launch rates of 28.571 input / 142.857 output / 35.714 cache write / 2.857 cache read DBU run through 2026-08-31; the standard rates are listed here because entries carry no expiry date."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.500002e-05,
|
||||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_mid_conversation_system": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_sampling_params": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-gemini-2-5-flash": {
|
||||
"cache_creation_input_token_cost": 3.0002e-07,
|
||||
"cache_read_input_token_cost": 3.0002e-08,
|
||||
"input_cost_per_token": 3.0001999999999996e-07,
|
||||
"input_dbu_cost_per_token": 4.285999999999999e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14771,9 +14940,12 @@
|
|||
"output_dbu_cost_per_token": 3.5714e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-2-5-pro": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.24999e-07,
|
||||
"input_cost_per_token": 1.24999e-06,
|
||||
"input_dbu_cost_per_token": 1.7857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14788,9 +14960,12 @@
|
|||
"output_dbu_cost_per_token": 0.000142857,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-3-1-flash-lite": {
|
||||
"cache_creation_input_token_cost": 3.1248e-07,
|
||||
"cache_read_input_token_cost": 3.122e-08,
|
||||
"input_cost_per_token": 3.1248e-07,
|
||||
"input_dbu_cost_per_token": 4.464e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14805,9 +14980,12 @@
|
|||
"output_dbu_cost_per_token": 2.6786e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-3-1-pro": {
|
||||
"cache_creation_input_token_cost": 2.49998e-06,
|
||||
"cache_read_input_token_cost": 2.4997e-07,
|
||||
"input_cost_per_token": 2.49998e-06,
|
||||
"input_dbu_cost_per_token": 3.5714e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14822,9 +15000,12 @@
|
|||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-3-flash": {
|
||||
"cache_creation_input_token_cost": 6.2503e-07,
|
||||
"cache_read_input_token_cost": 6.251e-08,
|
||||
"input_cost_per_token": 6.2503e-07,
|
||||
"input_dbu_cost_per_token": 8.929e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14839,9 +15020,12 @@
|
|||
"output_dbu_cost_per_token": 5.3571e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemini-3-pro": {
|
||||
"cache_creation_input_token_cost": 2.49998e-06,
|
||||
"cache_read_input_token_cost": 2.4997e-07,
|
||||
"input_cost_per_token": 2.49998e-06,
|
||||
"input_dbu_cost_per_token": 3.5714e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14856,9 +15040,12 @@
|
|||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-gemma-3-12b": {
|
||||
"cache_creation_input_token_cost": 1.5001e-07,
|
||||
"cache_read_input_token_cost": 1.5001e-07,
|
||||
"input_cost_per_token": 1.5000999999999998e-07,
|
||||
"input_dbu_cost_per_token": 2.1429999999999996e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14874,6 +15061,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-gpt-5": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.2502e-07,
|
||||
"input_cost_per_token": 1.24999e-06,
|
||||
"input_dbu_cost_per_token": 1.7857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14886,9 +15075,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 9.999990000000002e-06,
|
||||
"output_dbu_cost_per_token": 0.000142857,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-1": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.2502e-07,
|
||||
"input_cost_per_token": 1.24999e-06,
|
||||
"input_dbu_cost_per_token": 1.7857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14901,9 +15093,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 9.999990000000002e-06,
|
||||
"output_dbu_cost_per_token": 0.000142857,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-1-codex-max": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.2502e-07,
|
||||
"input_cost_per_token": 1.24999e-06,
|
||||
"input_dbu_cost_per_token": 1.7857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14916,9 +15111,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 9.999990000000002e-06,
|
||||
"output_dbu_cost_per_token": 0.000142857,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-1-codex-mini": {
|
||||
"cache_creation_input_token_cost": 2.4997e-07,
|
||||
"cache_read_input_token_cost": 2.499e-08,
|
||||
"input_cost_per_token": 2.4997e-07,
|
||||
"input_dbu_cost_per_token": 3.571e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14931,9 +15129,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.99997e-06,
|
||||
"output_dbu_cost_per_token": 2.8571e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-2": {
|
||||
"cache_creation_input_token_cost": 1.75e-06,
|
||||
"cache_read_input_token_cost": 1.75e-07,
|
||||
"input_cost_per_token": 1.75e-06,
|
||||
"input_dbu_cost_per_token": 2.5e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14946,9 +15147,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"output_dbu_cost_per_token": 0.0002,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-2-codex": {
|
||||
"cache_creation_input_token_cost": 1.75e-06,
|
||||
"cache_read_input_token_cost": 1.75e-07,
|
||||
"input_cost_per_token": 1.75e-06,
|
||||
"input_dbu_cost_per_token": 2.5e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14961,9 +15165,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"output_dbu_cost_per_token": 0.0002,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-3-codex": {
|
||||
"cache_creation_input_token_cost": 1.75e-06,
|
||||
"cache_read_input_token_cost": 1.75e-07,
|
||||
"input_cost_per_token": 1.75e-06,
|
||||
"input_dbu_cost_per_token": 2.5e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14976,9 +15183,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"output_dbu_cost_per_token": 0.0002,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-4": {
|
||||
"cache_creation_input_token_cost": 2.49998e-06,
|
||||
"cache_read_input_token_cost": 2.4997e-07,
|
||||
"input_cost_per_token": 2.49998e-06,
|
||||
"input_dbu_cost_per_token": 3.5714e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -14991,9 +15201,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.5000020000000002e-05,
|
||||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-4-mini": {
|
||||
"cache_creation_input_token_cost": 7.4998e-07,
|
||||
"cache_read_input_token_cost": 7.497e-08,
|
||||
"input_cost_per_token": 7.4998e-07,
|
||||
"input_dbu_cost_per_token": 1.0714e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15006,9 +15219,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 4.50002e-06,
|
||||
"output_dbu_cost_per_token": 6.4286e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-4-nano": {
|
||||
"cache_creation_input_token_cost": 1.9999e-07,
|
||||
"cache_read_input_token_cost": 2.002e-08,
|
||||
"input_cost_per_token": 1.9999e-07,
|
||||
"input_dbu_cost_per_token": 2.857e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15021,9 +15237,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.24999e-06,
|
||||
"output_dbu_cost_per_token": 1.7857e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-mini": {
|
||||
"cache_creation_input_token_cost": 2.4997e-07,
|
||||
"cache_read_input_token_cost": 2.499e-08,
|
||||
"input_cost_per_token": 2.4997000000000006e-07,
|
||||
"input_dbu_cost_per_token": 3.571e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15036,9 +15255,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.9999700000000004e-06,
|
||||
"output_dbu_cost_per_token": 2.8571e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-5-nano": {
|
||||
"cache_creation_input_token_cost": 4.998e-08,
|
||||
"cache_read_input_token_cost": 4.97e-09,
|
||||
"input_cost_per_token": 4.998e-08,
|
||||
"input_dbu_cost_per_token": 7.14e-07,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15051,9 +15273,12 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.9998000000000007e-07,
|
||||
"output_dbu_cost_per_token": 5.714000000000001e-06,
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving"
|
||||
"source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving",
|
||||
"supports_prompt_caching": true
|
||||
},
|
||||
"databricks/databricks-gpt-oss-120b": {
|
||||
"cache_creation_input_token_cost": 1.5001e-07,
|
||||
"cache_read_input_token_cost": 1.5001e-07,
|
||||
"input_cost_per_token": 1.5000999999999998e-07,
|
||||
"input_dbu_cost_per_token": 2.1429999999999996e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15069,6 +15294,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-gpt-oss-20b": {
|
||||
"cache_creation_input_token_cost": 7e-08,
|
||||
"cache_read_input_token_cost": 7e-08,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"input_dbu_cost_per_token": 1e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15084,6 +15311,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-gte-large-en": {
|
||||
"cache_creation_input_token_cost": 1.2999e-07,
|
||||
"cache_read_input_token_cost": 1.2999e-07,
|
||||
"input_cost_per_token": 1.2999000000000001e-07,
|
||||
"input_dbu_cost_per_token": 1.857e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15099,6 +15328,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-llama-2-70b-chat": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15115,6 +15346,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-llama-4-maverick": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15131,6 +15364,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-meta-llama-3-1-405b-instruct": {
|
||||
"cache_creation_input_token_cost": 5.00003e-06,
|
||||
"cache_read_input_token_cost": 5.00003e-06,
|
||||
"input_cost_per_token": 5.00003e-06,
|
||||
"input_dbu_cost_per_token": 7.1429e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15147,6 +15382,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-meta-llama-3-1-8b-instruct": {
|
||||
"cache_creation_input_token_cost": 1.5001e-07,
|
||||
"cache_read_input_token_cost": 1.5001e-07,
|
||||
"input_cost_per_token": 1.5000999999999998e-07,
|
||||
"input_dbu_cost_per_token": 2.1429999999999996e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15162,6 +15399,8 @@
|
|||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-meta-llama-3-3-70b-instruct": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15178,6 +15417,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-meta-llama-3-70b-instruct": {
|
||||
"cache_creation_input_token_cost": 1.00002e-06,
|
||||
"cache_read_input_token_cost": 1.00002e-06,
|
||||
"input_cost_per_token": 1.00002e-06,
|
||||
"input_dbu_cost_per_token": 1.4286e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15194,6 +15435,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-mixtral-8x7b-instruct": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15210,6 +15453,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-mpt-30b-instruct": {
|
||||
"cache_creation_input_token_cost": 1.00002e-06,
|
||||
"cache_read_input_token_cost": 1.00002e-06,
|
||||
"input_cost_per_token": 1.00002e-06,
|
||||
"input_dbu_cost_per_token": 1.4286e-05,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
@ -15226,6 +15471,8 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"databricks/databricks-mpt-7b-instruct": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
"input_cost_per_token": 5.0001e-07,
|
||||
"input_dbu_cost_per_token": 7.143e-06,
|
||||
"litellm_provider": "databricks",
|
||||
|
|
|
|||
|
|
@ -14,6 +14,14 @@ longer signal it.
|
|||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- **team**: `soft_budget`, `tags`, and `soft_budget_alerting_emails` attributes on `litellm_team`, matching what `/team/new` and `/team/update` already accept; `soft_budget_alerting_emails` is sent under `metadata`, where the proxy reads it
|
||||
|
||||
### Fixed
|
||||
|
||||
- **team**: Read now decodes the `team_info` envelope `/team/info` actually returns, so team attributes refresh from the proxy instead of always falling back to the prior state
|
||||
|
||||
### Changed
|
||||
|
||||
- **Versioning**: the provider is now published at the LiteLLM version, from the same commit as the proxy, on every LiteLLM release (dev, rc, stable). The `0.x` line ends at `0.4.0`; a `~> 0.4` constraint will not receive further releases, so re-pin to the LiteLLM version your proxy runs (for example `~> 1.99.0`). Existing `0.x` versions remain in the registry and keep verifying
|
||||
|
|
|
|||
|
|
@ -24,11 +24,18 @@ resource "litellm_team" "advanced_team" {
|
|||
|
||||
# Budget and rate limiting
|
||||
max_budget = 1000.0
|
||||
soft_budget = 800.0
|
||||
budget_duration = "1mo"
|
||||
tpm_limit = 500000
|
||||
rpm_limit = 5000
|
||||
blocked = false
|
||||
|
||||
# Who gets paged when spend crosses soft_budget
|
||||
soft_budget_alerting_emails = ["finops@example.com"]
|
||||
|
||||
# Tags for spend tracking and tag-based routing
|
||||
tags = ["team:ai-research", "environment:production"]
|
||||
|
||||
# Team member permissions
|
||||
team_member_permissions = [
|
||||
"create_key",
|
||||
|
|
@ -91,7 +98,9 @@ The following arguments are supported:
|
|||
|
||||
* `models` - (Optional) List of model names that this team can access.
|
||||
|
||||
* `metadata` - (Optional) A map of metadata key-value pairs associated with the team.
|
||||
* `metadata` - (Optional) A map of string metadata key-value pairs associated with the team. `tags` and `soft_budget_alerting_emails` are stored by the proxy under metadata but are managed through their own attributes below, not this map.
|
||||
|
||||
* `tags` - (Optional) List of tags applied to the team, used for [spend tracking](https://docs.litellm.ai/docs/proxy/enterprise#tracking-spend-for-custom-tags) and [tag-based routing](https://docs.litellm.ai/docs/proxy/tag_routing).
|
||||
|
||||
* `blocked` - (Optional) Whether the team is blocked from making requests. Default is `false`.
|
||||
|
||||
|
|
@ -101,6 +110,10 @@ The following arguments are supported:
|
|||
|
||||
* `max_budget` - (Optional) Maximum budget allocated to the team.
|
||||
|
||||
* `soft_budget` - (Optional) Spend threshold at which the proxy sends a soft budget alert without blocking requests.
|
||||
|
||||
* `soft_budget_alerting_emails` - (Optional) List of email addresses notified when the team's spend crosses `soft_budget`.
|
||||
|
||||
* `budget_duration` - (Optional) Duration for the budget cycle. Valid values are:
|
||||
* `daily`
|
||||
* `weekly`
|
||||
|
|
|
|||
|
|
@ -53,6 +53,11 @@ func ResourceLiteLLMTeam() *schema.Resource {
|
|||
Type: schema.TypeFloat,
|
||||
Optional: true,
|
||||
},
|
||||
"soft_budget": {
|
||||
Type: schema.TypeFloat,
|
||||
Optional: true,
|
||||
Description: "Spend threshold that triggers a soft budget alert without blocking requests",
|
||||
},
|
||||
"budget_duration": {
|
||||
Type: schema.TypeString,
|
||||
Optional: true,
|
||||
|
|
@ -72,6 +77,18 @@ func ResourceLiteLLMTeam() *schema.Resource {
|
|||
Elem: &schema.Schema{Type: schema.TypeString},
|
||||
Description: "List of permissions granted to team members",
|
||||
},
|
||||
"tags": {
|
||||
Type: schema.TypeList,
|
||||
Optional: true,
|
||||
Elem: &schema.Schema{Type: schema.TypeString},
|
||||
Description: "Tags for spend tracking and tag-based routing",
|
||||
},
|
||||
"soft_budget_alerting_emails": {
|
||||
Type: schema.TypeList,
|
||||
Optional: true,
|
||||
Elem: &schema.Schema{Type: schema.TypeString},
|
||||
Description: "Email addresses alerted when the team crosses soft_budget",
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
|
@ -117,21 +134,20 @@ func resourceLiteLLMTeamRead(d *schema.ResourceData, m interface{}) error {
|
|||
return nil
|
||||
}
|
||||
|
||||
var teamResp TeamResponse
|
||||
if err := json.NewDecoder(resp.Body).Decode(&teamResp); err != nil {
|
||||
var infoResp TeamInfoResponse
|
||||
if err := json.NewDecoder(resp.Body).Decode(&infoResp); err != nil {
|
||||
return fmt.Errorf("error decoding team info response: %w", err)
|
||||
}
|
||||
teamResp := infoResp.TeamInfo
|
||||
|
||||
// Update the state with values from the response or fall back to the data passed in during creation
|
||||
d.Set("team_alias", GetStringValue(teamResp.TeamAlias, d.Get("team_alias").(string)))
|
||||
d.Set("organization_id", GetStringValue(teamResp.OrganizationID, d.Get("organization_id").(string)))
|
||||
|
||||
// Handle metadata separately as it's a map
|
||||
if teamResp.Metadata != nil {
|
||||
d.Set("metadata", teamResp.Metadata)
|
||||
} else {
|
||||
d.Set("metadata", d.Get("metadata"))
|
||||
}
|
||||
metadata, tags, alertEmails := splitTeamMetadata(teamResp.Metadata)
|
||||
d.Set("metadata", metadata)
|
||||
d.Set("tags", tags)
|
||||
d.Set("soft_budget_alerting_emails", alertEmails)
|
||||
|
||||
if teamResp.TPMLimit != nil {
|
||||
d.Set("tpm_limit", *teamResp.TPMLimit)
|
||||
|
|
@ -142,6 +158,7 @@ func resourceLiteLLMTeamRead(d *schema.ResourceData, m interface{}) error {
|
|||
if teamResp.MaxBudget != nil {
|
||||
d.Set("max_budget", *teamResp.MaxBudget)
|
||||
}
|
||||
d.Set("soft_budget", teamResp.SoftBudget)
|
||||
d.Set("budget_duration", GetStringValue(teamResp.BudgetDuration, d.Get("budget_duration").(string)))
|
||||
|
||||
// Handle models separately as it's a list
|
||||
|
|
@ -240,15 +257,77 @@ func buildTeamData(d *schema.ResourceData, teamID string) map[string]interface{}
|
|||
"team_alias": d.Get("team_alias").(string),
|
||||
}
|
||||
|
||||
for _, key := range []string{"organization_id", "metadata", "tpm_limit", "rpm_limit", "max_budget", "budget_duration", "models", "blocked", "team_member_permissions"} {
|
||||
for _, key := range []string{"organization_id", "tpm_limit", "rpm_limit", "max_budget", "budget_duration", "models", "blocked", "team_member_permissions"} {
|
||||
if v, ok := d.GetOk(key); ok {
|
||||
teamData[key] = v
|
||||
}
|
||||
}
|
||||
|
||||
if v, ok := d.GetOk("soft_budget"); ok {
|
||||
teamData["soft_budget"] = v
|
||||
} else if d.HasChange("soft_budget") {
|
||||
teamData["soft_budget"] = nil
|
||||
}
|
||||
|
||||
if v, ok := d.GetOk("tags"); ok || d.HasChange("tags") {
|
||||
teamData["tags"] = v
|
||||
}
|
||||
|
||||
if metadata := buildTeamMetadata(d); metadata != nil {
|
||||
teamData["metadata"] = metadata
|
||||
}
|
||||
|
||||
return teamData
|
||||
}
|
||||
|
||||
// /team/update replaces metadata wholesale, so the full map must go out whenever either half changed.
|
||||
func buildTeamMetadata(d *schema.ResourceData) map[string]interface{} {
|
||||
metadata := map[string]interface{}{}
|
||||
for k, v := range d.Get("metadata").(map[string]interface{}) {
|
||||
metadata[k] = v
|
||||
}
|
||||
if v, ok := d.GetOk("soft_budget_alerting_emails"); ok {
|
||||
metadata["soft_budget_alerting_emails"] = v
|
||||
}
|
||||
if len(metadata) == 0 && !d.HasChange("metadata") && !d.HasChange("soft_budget_alerting_emails") {
|
||||
return nil
|
||||
}
|
||||
return metadata
|
||||
}
|
||||
|
||||
func splitTeamMetadata(raw map[string]interface{}) (map[string]string, []string, []string) {
|
||||
metadata := map[string]string{}
|
||||
var tags, alertEmails []string
|
||||
for k, v := range raw {
|
||||
switch k {
|
||||
case "tags":
|
||||
tags = toStringSlice(v)
|
||||
case "soft_budget_alerting_emails":
|
||||
alertEmails = toStringSlice(v)
|
||||
case "team_member_budget_id":
|
||||
default:
|
||||
if s, ok := v.(string); ok {
|
||||
metadata[k] = s
|
||||
}
|
||||
}
|
||||
}
|
||||
return metadata, tags, alertEmails
|
||||
}
|
||||
|
||||
func toStringSlice(v interface{}) []string {
|
||||
items, ok := v.([]interface{})
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
out := make([]string, 0, len(items))
|
||||
for _, item := range items {
|
||||
if s, ok := item.(string); ok {
|
||||
out = append(out, s)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func handleResponse(resp *http.Response, action string) error {
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
body, _ := io.ReadAll(resp.Body)
|
||||
|
|
|
|||
184
terraform/provider/litellm/resource_team_test.go
Normal file
184
terraform/provider/litellm/resource_team_test.go
Normal file
|
|
@ -0,0 +1,184 @@
|
|||
package litellm
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
"github.com/hashicorp/terraform-plugin-sdk/v2/helper/schema"
|
||||
"github.com/hashicorp/terraform-plugin-sdk/v2/terraform"
|
||||
)
|
||||
|
||||
func newTeamTestServer(t *testing.T, captured *map[string]interface{}, infoBody string) *httptest.Server {
|
||||
t.Helper()
|
||||
return httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
switch r.URL.Path {
|
||||
case endpointTeamNew, endpointTeamUpdate:
|
||||
body, _ := io.ReadAll(r.Body)
|
||||
json.Unmarshal(body, captured)
|
||||
w.Write([]byte(`{}`))
|
||||
case endpointTeamInfo:
|
||||
w.Write([]byte(infoBody))
|
||||
case endpointTeamPermissionsList:
|
||||
w.Write([]byte(`{"team_id":"team-1","team_member_permissions":[],"all_available_permissions":[]}`))
|
||||
default:
|
||||
w.WriteHeader(http.StatusNotFound)
|
||||
}
|
||||
}))
|
||||
}
|
||||
|
||||
const teamInfoWithSoftBudget = `{
|
||||
"team_id": "team-1",
|
||||
"team_info": {
|
||||
"team_id": "team-1",
|
||||
"team_alias": "insights",
|
||||
"max_budget": 750.0,
|
||||
"soft_budget": 600.0,
|
||||
"models": ["claude-haiku-4-5"],
|
||||
"metadata": {
|
||||
"department": "customer-insights",
|
||||
"tags": ["team:customer-insights", "environment:production"],
|
||||
"soft_budget_alerting_emails": ["finops@example.com"],
|
||||
"team_member_budget_id": "budget-1"
|
||||
}
|
||||
},
|
||||
"keys": [],
|
||||
"team_memberships": []
|
||||
}`
|
||||
|
||||
func TestTeamCreateSendsSoftBudgetTagsAndAlertEmails(t *testing.T) {
|
||||
var captured map[string]interface{}
|
||||
srv := newTeamTestServer(t, &captured, teamInfoWithSoftBudget)
|
||||
defer srv.Close()
|
||||
|
||||
d := schema.TestResourceDataRaw(t, ResourceLiteLLMTeam().Schema, map[string]interface{}{
|
||||
"team_alias": "insights",
|
||||
"max_budget": 750.0,
|
||||
"soft_budget": 600.0,
|
||||
"tags": []interface{}{"team:customer-insights", "environment:production"},
|
||||
"soft_budget_alerting_emails": []interface{}{"finops@example.com"},
|
||||
"metadata": map[string]interface{}{"department": "customer-insights"},
|
||||
})
|
||||
|
||||
if err := resourceLiteLLMTeamCreate(d, NewClient(srv.URL, "test-key", true)); err != nil {
|
||||
t.Fatalf("create failed: %v", err)
|
||||
}
|
||||
|
||||
if got := captured["soft_budget"]; got != 600.0 {
|
||||
t.Fatalf("payload soft_budget = %v, want 600", got)
|
||||
}
|
||||
wantTags := []interface{}{"team:customer-insights", "environment:production"}
|
||||
if got := captured["tags"]; !reflect.DeepEqual(got, wantTags) {
|
||||
t.Fatalf("payload tags = %v, want %v", got, wantTags)
|
||||
}
|
||||
wantMetadata := map[string]interface{}{
|
||||
"department": "customer-insights",
|
||||
"soft_budget_alerting_emails": []interface{}{"finops@example.com"},
|
||||
}
|
||||
if got := captured["metadata"]; !reflect.DeepEqual(got, wantMetadata) {
|
||||
t.Fatalf("payload metadata = %v, want %v", got, wantMetadata)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTeamReadMapsTeamInfoEnvelope(t *testing.T) {
|
||||
var captured map[string]interface{}
|
||||
srv := newTeamTestServer(t, &captured, teamInfoWithSoftBudget)
|
||||
defer srv.Close()
|
||||
|
||||
d := schema.TestResourceDataRaw(t, ResourceLiteLLMTeam().Schema, map[string]interface{}{})
|
||||
d.SetId("team-1")
|
||||
|
||||
if err := resourceLiteLLMTeamRead(d, NewClient(srv.URL, "test-key", true)); err != nil {
|
||||
t.Fatalf("read failed: %v", err)
|
||||
}
|
||||
|
||||
if got := d.Get("team_alias"); got != "insights" {
|
||||
t.Fatalf("team_alias = %v, want insights", got)
|
||||
}
|
||||
if got := d.Get("soft_budget"); got != 600.0 {
|
||||
t.Fatalf("soft_budget = %v, want 600", got)
|
||||
}
|
||||
if got := d.Get("max_budget"); got != 750.0 {
|
||||
t.Fatalf("max_budget = %v, want 750", got)
|
||||
}
|
||||
wantTags := []interface{}{"team:customer-insights", "environment:production"}
|
||||
if got := d.Get("tags"); !reflect.DeepEqual(got, wantTags) {
|
||||
t.Fatalf("tags = %v, want %v", got, wantTags)
|
||||
}
|
||||
wantEmails := []interface{}{"finops@example.com"}
|
||||
if got := d.Get("soft_budget_alerting_emails"); !reflect.DeepEqual(got, wantEmails) {
|
||||
t.Fatalf("soft_budget_alerting_emails = %v, want %v", got, wantEmails)
|
||||
}
|
||||
wantMetadata := map[string]interface{}{"department": "customer-insights"}
|
||||
if got := d.Get("metadata"); !reflect.DeepEqual(got, wantMetadata) {
|
||||
t.Fatalf("metadata = %v, want %v (server-managed team_member_budget_id dropped)", got, wantMetadata)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTeamUpdateClearsRemovedTagsAndSoftBudget(t *testing.T) {
|
||||
var captured map[string]interface{}
|
||||
srv := newTeamTestServer(t, &captured, `{"team_id":"team-1","team_info":{"team_id":"team-1","team_alias":"insights"},"keys":[],"team_memberships":[]}`)
|
||||
defer srv.Close()
|
||||
|
||||
res := ResourceLiteLLMTeam()
|
||||
priorData := schema.TestResourceDataRaw(t, res.Schema, map[string]interface{}{
|
||||
"team_alias": "insights",
|
||||
"soft_budget": 600.0,
|
||||
"tags": []interface{}{"team:to-be-removed"},
|
||||
"soft_budget_alerting_emails": []interface{}{"ops@example.com"},
|
||||
"metadata": map[string]interface{}{"department": "eng"},
|
||||
})
|
||||
priorData.SetId("team-1")
|
||||
prior := priorData.State()
|
||||
config := terraform.NewResourceConfigRaw(map[string]interface{}{
|
||||
"team_alias": "insights",
|
||||
"metadata": map[string]interface{}{"department": "eng"},
|
||||
})
|
||||
diff, err := res.Diff(context.Background(), prior, config, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("diff failed: %v", err)
|
||||
}
|
||||
d, err := schema.InternalMap(res.Schema).Data(prior, diff)
|
||||
if err != nil {
|
||||
t.Fatalf("data failed: %v", err)
|
||||
}
|
||||
|
||||
if err := resourceLiteLLMTeamUpdate(d, NewClient(srv.URL, "test-key", true)); err != nil {
|
||||
t.Fatalf("update failed: %v", err)
|
||||
}
|
||||
|
||||
if got, ok := captured["soft_budget"]; !ok || got != nil {
|
||||
t.Fatalf("payload soft_budget = %v (present=%v), want explicit null", got, ok)
|
||||
}
|
||||
if got := captured["tags"]; !reflect.DeepEqual(got, []interface{}{}) {
|
||||
t.Fatalf("payload tags = %v, want []", got)
|
||||
}
|
||||
if got := captured["metadata"]; !reflect.DeepEqual(got, map[string]interface{}{"department": "eng"}) {
|
||||
t.Fatalf("payload metadata = %v, want department only", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTeamReadClearsSoftBudgetWhenProxyReturnsNull(t *testing.T) {
|
||||
var captured map[string]interface{}
|
||||
srv := newTeamTestServer(t, &captured, `{"team_id":"team-1","team_info":{"team_id":"team-1","team_alias":"insights","soft_budget":null},"keys":[],"team_memberships":[]}`)
|
||||
defer srv.Close()
|
||||
|
||||
d := schema.TestResourceDataRaw(t, ResourceLiteLLMTeam().Schema, map[string]interface{}{
|
||||
"team_alias": "insights",
|
||||
"soft_budget": 600.0,
|
||||
})
|
||||
d.SetId("team-1")
|
||||
|
||||
if err := resourceLiteLLMTeamRead(d, NewClient(srv.URL, "test-key", true)); err != nil {
|
||||
t.Fatalf("read failed: %v", err)
|
||||
}
|
||||
|
||||
if got := d.Get("soft_budget"); got != 0.0 {
|
||||
t.Fatalf("soft_budget = %v, want cleared after the proxy returned null", got)
|
||||
}
|
||||
}
|
||||
|
|
@ -33,6 +33,11 @@ type ModelRequest struct {
|
|||
Additional map[string]interface{} `json:"additional"`
|
||||
}
|
||||
|
||||
type TeamInfoResponse struct {
|
||||
TeamID string `json:"team_id"`
|
||||
TeamInfo TeamResponse `json:"team_info"`
|
||||
}
|
||||
|
||||
// TeamResponse represents a response from the API containing team information.
|
||||
type TeamResponse struct {
|
||||
TeamID string `json:"team_id,omitempty"`
|
||||
|
|
@ -42,6 +47,7 @@ type TeamResponse struct {
|
|||
TPMLimit *int `json:"tpm_limit,omitempty"`
|
||||
RPMLimit *int `json:"rpm_limit,omitempty"`
|
||||
MaxBudget *float64 `json:"max_budget,omitempty"`
|
||||
SoftBudget *float64 `json:"soft_budget,omitempty"`
|
||||
BudgetDuration string `json:"budget_duration,omitempty"`
|
||||
Models []string `json:"models"`
|
||||
Blocked bool `json:"blocked,omitempty"`
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ from litellm.a2a_protocol.card_resolver import (
|
|||
LiteLLMA2ACardResolver,
|
||||
fix_agent_card_url,
|
||||
is_localhost_or_internal_url,
|
||||
normalize_agent_card_interfaces,
|
||||
set_agent_card_url,
|
||||
)
|
||||
|
||||
|
|
@ -114,3 +115,26 @@ def test_fix_agent_card_url_updates_interface_when_top_level_is_localhost():
|
|||
|
||||
assert result.url == "https://my-public-agent.example.com/"
|
||||
assert result.supported_interfaces[0].url == "https://my-public-agent.example.com/"
|
||||
|
||||
|
||||
def test_normalize_agent_card_interfaces_downgrades_miscased_interfaces_to_the_0_3_dialect():
|
||||
pb2 = pytest.importorskip("a2a.types.a2a_pb2")
|
||||
|
||||
card = pb2.AgentCard(
|
||||
name="langgraph",
|
||||
supported_interfaces=[
|
||||
pb2.AgentInterface(url="http://a/", protocol_binding="jsonrpc", protocol_version="1.0"),
|
||||
pb2.AgentInterface(url="http://b/", protocol_binding="JSONRPC", protocol_version="1.0"),
|
||||
pb2.AgentInterface(url="http://c/", protocol_binding="websocket", protocol_version="1.0"),
|
||||
],
|
||||
)
|
||||
|
||||
normalized = normalize_agent_card_interfaces(card)
|
||||
|
||||
assert [(i.protocol_binding, i.protocol_version) for i in normalized.supported_interfaces] == [
|
||||
("JSONRPC", "0.3"),
|
||||
("JSONRPC", "1.0"),
|
||||
("websocket", "1.0"),
|
||||
]
|
||||
assert card.supported_interfaces[0].protocol_binding == "jsonrpc"
|
||||
assert card.supported_interfaces[0].protocol_version == "1.0"
|
||||
|
|
|
|||
|
|
@ -176,10 +176,62 @@ _AGENT_A_HEADERS = {"x-agent-token": "token-for-a", "x-tenant": "tenant-a"}
|
|||
_AGENT_B_HEADERS = {"x-agent-token": "token-for-b", "x-tenant": "tenant-b"}
|
||||
|
||||
|
||||
_LANGGRAPH_TASK_REPLY = {
|
||||
"jsonrpc": "2.0",
|
||||
"id": "reply",
|
||||
"result": {
|
||||
"kind": "task",
|
||||
"id": "run-1:task-1",
|
||||
"contextId": "thread-1",
|
||||
"history": [
|
||||
{
|
||||
"kind": "message",
|
||||
"role": "user",
|
||||
"parts": [{"kind": "text", "text": "hi"}],
|
||||
"messageId": "m-user",
|
||||
"taskId": "run-1:task-1",
|
||||
"contextId": "thread-1",
|
||||
},
|
||||
{
|
||||
"kind": "message",
|
||||
"role": "agent",
|
||||
"parts": [{"kind": "text", "text": "langgraph echo: hi"}],
|
||||
"messageId": "m-agent",
|
||||
"taskId": "run-1:task-1",
|
||||
"contextId": "thread-1",
|
||||
},
|
||||
],
|
||||
"status": {"state": "completed", "timestamp": "2026-08-24T00:00:00+00:00"},
|
||||
"artifacts": [
|
||||
{
|
||||
"artifactId": "art-1",
|
||||
"name": "Assistant Response",
|
||||
"parts": [{"kind": "text", "text": "langgraph echo: hi"}],
|
||||
}
|
||||
],
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
_LOWERCASE_BINDING_CARD = {
|
||||
"name": "langgraph-agent",
|
||||
"version": "1.0.0",
|
||||
"capabilities": {"streaming": True},
|
||||
"defaultInputModes": ["text/plain"],
|
||||
"defaultOutputModes": ["text/plain"],
|
||||
"skills": [],
|
||||
"supportedInterfaces": [
|
||||
{"url": "http://127.0.0.1:9/", "protocolBinding": "jsonrpc", "protocolVersion": "1.0"}
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
class _RequestRecorder:
|
||||
"""Records the headers httpx put on the wire, per outbound request."""
|
||||
|
||||
def __init__(self):
|
||||
def __init__(self, card=_AGENT_CARD, rpc_reply=_RPC_REPLY):
|
||||
self.card = card
|
||||
self.rpc_reply = rpc_reply
|
||||
self.card_requests = []
|
||||
self.rpc_requests = []
|
||||
self.client = None
|
||||
|
|
@ -188,23 +240,23 @@ class _RequestRecorder:
|
|||
headers = {k.lower(): v for k, v in request.headers.items()}
|
||||
if request.method == "GET":
|
||||
self.card_requests.append(headers)
|
||||
return httpx.Response(200, json=_AGENT_CARD)
|
||||
return httpx.Response(200, json=self.card)
|
||||
self.rpc_requests.append(headers)
|
||||
return httpx.Response(200, json=_RPC_REPLY)
|
||||
return httpx.Response(200, json=self.rpc_reply)
|
||||
|
||||
|
||||
def _a2a_client_cache_key(timeout: float) -> str:
|
||||
return "async_httpx_client" + f"timeout_{timeout}" + httpxSpecialProvider.A2AProvider
|
||||
|
||||
|
||||
async def _seed_shared_a2a_client() -> _RequestRecorder:
|
||||
async def _seed_shared_a2a_client(card=_AGENT_CARD, rpc_reply=_RPC_REPLY) -> _RequestRecorder:
|
||||
"""Put the one A2A client the cache will hand out behind a mock transport.
|
||||
|
||||
Seeding has to happen on the test's own event loop, because the client cache keys on
|
||||
it. The injected client is a real httpx.AsyncClient, so the merge of per-request
|
||||
headers over client defaults, which is what these tests are about, stays real.
|
||||
"""
|
||||
recorder = _RequestRecorder()
|
||||
recorder = _RequestRecorder(card=card, rpc_reply=rpc_reply)
|
||||
handler = AsyncHTTPHandler(timeout=DEFAULT_A2A_AGENT_TIMEOUT)
|
||||
owned_client = handler.client
|
||||
handler.client = httpx.AsyncClient(transport=httpx.MockTransport(recorder))
|
||||
|
|
@ -311,6 +363,25 @@ async def test_streaming_send_carries_only_its_own_caller_headers(isolated_clien
|
|||
assert received["b"]["x-tenant"] == "tenant-b"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_lowercase_protocol_binding_card_round_trips_the_langgraph_dialect(isolated_client_cache):
|
||||
"""LangGraph Platform serves cards with protocolBinding "jsonrpc" and answers in the
|
||||
A2A 0.3 JSON dialect ("kind"-discriminated) while declaring protocolVersion "1.0".
|
||||
Without binding normalization client creation raises ValueError("no compatible
|
||||
transports found."); without the version downgrade the SDK's strict v1 transport
|
||||
rejects the reply with 'Message type "lf.a2a.v1.Task" has no field named "kind"'."""
|
||||
await _seed_shared_a2a_client(card=_LOWERCASE_BINDING_CARD, rpc_reply=_LANGGRAPH_TASK_REPLY)
|
||||
|
||||
a2a_client = await create_a2a_client(base_url="http://127.0.0.1:9")
|
||||
response = await _send_message(a2a_client, _send_request("lc"))
|
||||
|
||||
assert type(response.root.result).__name__ == "Task"
|
||||
assert response.root.result.artifacts[0].parts[0].root.text == "langgraph echo: hi"
|
||||
interface = a2a_client._litellm_agent_card.supported_interfaces[0]
|
||||
assert interface.protocol_binding == "JSONRPC"
|
||||
assert interface.protocol_version == "0.3"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_agent_card_fetch_carries_the_callers_headers(isolated_client_cache):
|
||||
"""Agent cards can sit behind the same auth as the agent, so the card fetch must stay
|
||||
|
|
|
|||
|
|
@ -3762,3 +3762,112 @@ def test_response_incomplete_stream_event_without_details_defaults_to_length():
|
|||
result = iterator.chunk_parser(chunk)
|
||||
|
||||
assert result.choices[0].finish_reason == "length"
|
||||
|
||||
|
||||
def test_assistant_message_with_tool_calls_keeps_its_content():
|
||||
"""Regression for https://github.com/BerriAI/litellm/issues/24985.
|
||||
|
||||
An assistant turn that both answered and called a tool used to lose its whole message:
|
||||
the branch handling tool_calls emitted the calls and dropped the text.
|
||||
"""
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
messages = [
|
||||
{"role": "user", "content": "What is the weather in Denver?"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "Let me look that up.",
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"type": "function",
|
||||
"function": {"name": "get_weather", "arguments": '{"city": "Denver"}'},
|
||||
}
|
||||
],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": "call_1", "content": "88F"},
|
||||
]
|
||||
|
||||
input_items, _ = handler.convert_chat_completion_messages_to_responses_api(messages)
|
||||
|
||||
assistant_message = next(
|
||||
item for item in input_items if item.get("type") == "message" and item.get("role") == "assistant"
|
||||
)
|
||||
assert assistant_message["content"] == [{"type": "output_text", "text": "Let me look that up."}]
|
||||
assert [item.get("type") for item in input_items] == [
|
||||
"message",
|
||||
"message",
|
||||
"function_call",
|
||||
"function_call_output",
|
||||
]
|
||||
|
||||
|
||||
def test_assistant_thinking_blocks_become_a_reasoning_input_item():
|
||||
"""Thinking blocks are how an Anthropic-shaped turn carries reasoning into this bridge."""
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
messages = [
|
||||
{"role": "user", "content": "What is the weather in Denver?"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "Denver is sunny.",
|
||||
"thinking_blocks": [
|
||||
{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "sig1"},
|
||||
{"type": "redacted_thinking", "data": "REDACTED"},
|
||||
],
|
||||
},
|
||||
{"role": "user", "content": "Why?"},
|
||||
]
|
||||
|
||||
input_items, _ = handler.convert_chat_completion_messages_to_responses_api(messages)
|
||||
|
||||
reasoning_item = next(item for item in input_items if item.get("type") == "reasoning")
|
||||
assert reasoning_item["summary"] == [{"type": "summary_text", "text": "August in Denver is dry."}]
|
||||
assert "id" not in reasoning_item
|
||||
|
||||
|
||||
def test_thinking_only_assistant_turn_still_sends_its_reasoning():
|
||||
"""An assistant turn can be pure reasoning, with no visible text and no tool call."""
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
messages = [
|
||||
{"role": "user", "content": "What is the weather in Denver?"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"thinking_blocks": [
|
||||
{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "sig1"}
|
||||
],
|
||||
},
|
||||
{"role": "user", "content": "Why?"},
|
||||
]
|
||||
|
||||
input_items, _ = handler.convert_chat_completion_messages_to_responses_api(messages)
|
||||
|
||||
reasoning_items = [item for item in input_items if item.get("type") == "reasoning"]
|
||||
assert len(reasoning_items) == 1
|
||||
assert reasoning_items[0]["summary"] == [{"type": "summary_text", "text": "August in Denver is dry."}]
|
||||
|
||||
|
||||
def test_stored_reasoning_items_win_over_thinking_blocks():
|
||||
"""A minted reasoning id beats a re-derived one, so the two must not both be sent."""
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
messages = [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "Denver is sunny.",
|
||||
"reasoning_items": [
|
||||
{
|
||||
"type": "reasoning",
|
||||
"id": "rs_real",
|
||||
"summary": [{"type": "summary_text", "text": "August in Denver is dry."}],
|
||||
}
|
||||
],
|
||||
"thinking_blocks": [
|
||||
{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "rs_real"}
|
||||
],
|
||||
},
|
||||
]
|
||||
|
||||
input_items, _ = handler.convert_chat_completion_messages_to_responses_api(messages)
|
||||
|
||||
reasoning_items = [item for item in input_items if item.get("type") == "reasoning"]
|
||||
assert len(reasoning_items) == 1
|
||||
assert reasoning_items[0]["id"] == "rs_real"
|
||||
|
|
|
|||
|
|
@ -1599,6 +1599,14 @@ class TestEnableAnthropicPromptCaching:
|
|||
assert supports_prompt_caching(model=model, custom_llm_provider=provider) is True
|
||||
assert self._points(model=model, provider=provider) == []
|
||||
|
||||
def test_databricks_claude_not_injected_despite_caching_support(self, monkeypatch, local_model_cost_map):
|
||||
from litellm.utils import supports_prompt_caching
|
||||
|
||||
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
|
||||
model = "databricks/databricks-claude-sonnet-4-5"
|
||||
assert supports_prompt_caching(model=model, custom_llm_provider="databricks") is True
|
||||
assert self._points(model=model, provider="databricks") == []
|
||||
|
||||
def test_model_without_caching_support_not_injected(self, monkeypatch):
|
||||
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
|
||||
assert self._points(model="anthropic.claude-3-5-sonnet-20240620-v1:0", provider="bedrock") == []
|
||||
|
|
|
|||
545
tests/test_litellm/interactions/test_background_cost_polling.py
Normal file
545
tests/test_litellm/interactions/test_background_cost_polling.py
Normal file
|
|
@ -0,0 +1,545 @@
|
|||
import asyncio
|
||||
import time
|
||||
from itertools import islice
|
||||
from typing import Optional
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.interactions.background_cost_polling import (
|
||||
_SETTLED_KEY,
|
||||
_poll_intervals,
|
||||
BackgroundInteractionPollContext,
|
||||
maybe_schedule_background_interaction_cost_polling,
|
||||
maybe_settle_background_interaction_before_delete,
|
||||
poll_and_log_background_interaction_cost,
|
||||
)
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LitellmLogging
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
USAGE_BLOCK = {
|
||||
"total_tokens": 175,
|
||||
"total_input_tokens": 100,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 100}],
|
||||
"total_cached_tokens": 0,
|
||||
"total_output_tokens": 50,
|
||||
"output_tokens_by_modality": [{"modality": "text", "tokens": 50}],
|
||||
"total_tool_use_tokens": 0,
|
||||
"total_thought_tokens": 25,
|
||||
}
|
||||
|
||||
|
||||
def _logging_obj(
|
||||
call_type: str = "acreate_interaction",
|
||||
litellm_params: Optional[dict] = None,
|
||||
) -> LitellmLogging:
|
||||
logging_obj = LitellmLogging(
|
||||
model="gemini-2.5-flash",
|
||||
messages=[],
|
||||
stream=False,
|
||||
call_type=call_type,
|
||||
start_time=time.time(),
|
||||
litellm_call_id="bg-interactions-call-id",
|
||||
function_id="bg-interactions-fn-id",
|
||||
)
|
||||
logging_obj.update_environment_variables(
|
||||
litellm_params=litellm_params or {},
|
||||
optional_params={},
|
||||
model="gemini-2.5-flash",
|
||||
custom_llm_provider="gemini",
|
||||
input="hi",
|
||||
)
|
||||
return logging_obj
|
||||
|
||||
|
||||
def _reservation() -> dict:
|
||||
return {"reserved_cost": 0.05, "entries": [], "finalized": False, "input_cost": 0.001}
|
||||
|
||||
|
||||
def _logging_obj_with_reservation(reservation: dict) -> LitellmLogging:
|
||||
return _logging_obj(litellm_params={"metadata": {"user_api_key_budget_reservation": reservation}})
|
||||
|
||||
|
||||
async def _raise_on_billing(result: InteractionsAPIResponse) -> None:
|
||||
raise RuntimeError("cost calculation failed for a settled background interaction")
|
||||
|
||||
|
||||
def _context(logging_obj: LitellmLogging, timeout_seconds: float = 1.0) -> BackgroundInteractionPollContext:
|
||||
return BackgroundInteractionPollContext(
|
||||
interaction_id="interactions/bg-abc",
|
||||
custom_llm_provider="gemini",
|
||||
logging_obj=logging_obj,
|
||||
initial_interval_seconds=0.001,
|
||||
max_interval_seconds=0.002,
|
||||
timeout_seconds=timeout_seconds,
|
||||
)
|
||||
|
||||
|
||||
def _response(status: str, with_usage: bool) -> InteractionsAPIResponse:
|
||||
return InteractionsAPIResponse(
|
||||
id="interactions/bg-abc",
|
||||
model="gemini-2.5-flash",
|
||||
status=status,
|
||||
steps=[],
|
||||
usage=dict(USAGE_BLOCK) if with_usage else None,
|
||||
)
|
||||
|
||||
|
||||
def _fetch_sequence(*responses):
|
||||
remaining = list(responses)
|
||||
calls = []
|
||||
|
||||
async def fetch(context):
|
||||
calls.append(context.interaction_id)
|
||||
item = remaining.pop(0) if len(remaining) > 1 else remaining[0]
|
||||
if isinstance(item, Exception):
|
||||
raise item
|
||||
return item
|
||||
|
||||
return fetch, calls
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"initial, maximum",
|
||||
[(0.0, 0.002), (0.001, 0.0), (-1.0, 0.002), (0.0, 0.0)],
|
||||
)
|
||||
def test_poll_intervals_stops_instead_of_looping_on_a_non_positive_interval(initial, maximum):
|
||||
intervals = list(islice(_poll_intervals(initial=initial, maximum=maximum, timeout=3600.0), 10))
|
||||
|
||||
assert len(intervals) < 10
|
||||
assert all(interval > 0 for interval in intervals)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_bills_once_when_interaction_completes():
|
||||
logging_obj = _logging_obj()
|
||||
fetch, calls = _fetch_sequence(
|
||||
_response("in_progress", with_usage=False),
|
||||
_response("completed", with_usage=True),
|
||||
)
|
||||
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert len(calls) == 2
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_bills_an_interaction_paused_for_a_tool_result():
|
||||
logging_obj = _logging_obj()
|
||||
fetch, calls = _fetch_sequence(
|
||||
_response("in_progress", with_usage=False),
|
||||
_response("requires_action", with_usage=True),
|
||||
)
|
||||
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert len(calls) == 2
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_does_not_pin_the_budget_for_an_interaction_paused_for_a_tool_result():
|
||||
reservation = _reservation()
|
||||
logging_obj = _logging_obj_with_reservation(reservation)
|
||||
fetch, _ = _fetch_sequence(_response("requires_action", with_usage=True))
|
||||
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
assert reservation["finalized"] is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_stops_without_billing_on_terminal_status_without_usage():
|
||||
logging_obj = _logging_obj()
|
||||
fetch, calls = _fetch_sequence(_response("failed", with_usage=False))
|
||||
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert len(calls) == 1
|
||||
assert logging_obj.model_call_details.get("response_cost") is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_gives_up_after_timeout_without_billing():
|
||||
logging_obj = _logging_obj()
|
||||
fetch, calls = _fetch_sequence(_response("in_progress", with_usage=False))
|
||||
|
||||
await poll_and_log_background_interaction_cost(
|
||||
_context(logging_obj, timeout_seconds=0.01),
|
||||
fetch_interaction=fetch,
|
||||
)
|
||||
|
||||
assert len(calls) >= 2
|
||||
assert logging_obj.model_call_details.get("response_cost") is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_releases_budget_reservation_when_interaction_ends_without_usage():
|
||||
reservation = _reservation()
|
||||
logging_obj = _logging_obj_with_reservation(reservation)
|
||||
fetch, _ = _fetch_sequence(_response("failed", with_usage=False))
|
||||
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_releases_budget_reservation_on_timeout_give_up():
|
||||
reservation = _reservation()
|
||||
logging_obj = _logging_obj_with_reservation(reservation)
|
||||
fetch, _ = _fetch_sequence(_response("in_progress", with_usage=False))
|
||||
|
||||
await poll_and_log_background_interaction_cost(
|
||||
_context(logging_obj, timeout_seconds=0.01),
|
||||
fetch_interaction=fetch,
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_releases_budget_reservation_when_billing_raises():
|
||||
reservation = _reservation()
|
||||
logging_obj = _logging_obj_with_reservation(reservation)
|
||||
fetch, _ = _fetch_sequence(_response("completed", with_usage=True))
|
||||
logging_obj.async_log_background_interaction_completion = _raise_on_billing
|
||||
|
||||
with pytest.raises(RuntimeError):
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_leaves_reservation_reconciliation_to_the_completion_event():
|
||||
reservation = _reservation()
|
||||
logging_obj = _logging_obj_with_reservation(reservation)
|
||||
fetch, _ = _fetch_sequence(
|
||||
_response("in_progress", with_usage=False),
|
||||
_response("completed", with_usage=True),
|
||||
)
|
||||
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
assert reservation["finalized"] is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_retries_after_fetch_error_and_still_bills():
|
||||
logging_obj = _logging_obj()
|
||||
fetch, calls = _fetch_sequence(
|
||||
RuntimeError("transient network error"),
|
||||
_response("completed", with_usage=True),
|
||||
)
|
||||
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert len(calls) == 2
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_schedule_creates_poll_task_for_in_progress_create():
|
||||
logging_obj = _logging_obj()
|
||||
task = maybe_schedule_background_interaction_cost_polling(
|
||||
response=_response("in_progress", with_usage=False),
|
||||
create_kwargs={"litellm_logging_obj": logging_obj},
|
||||
custom_llm_provider="gemini",
|
||||
)
|
||||
|
||||
assert isinstance(task, asyncio.Task)
|
||||
task.cancel()
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await task
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"response,create_kwargs",
|
||||
[
|
||||
(_response("completed", with_usage=True), {"litellm_logging_obj": "placeholder"}),
|
||||
(_response("in_progress", with_usage=False), {}),
|
||||
("not a response", {"litellm_logging_obj": "placeholder"}),
|
||||
],
|
||||
)
|
||||
async def test_schedule_skips_non_pollable_results(response, create_kwargs):
|
||||
if create_kwargs.get("litellm_logging_obj") == "placeholder":
|
||||
create_kwargs = {"litellm_logging_obj": _logging_obj()}
|
||||
|
||||
task = maybe_schedule_background_interaction_cost_polling(
|
||||
response=response,
|
||||
create_kwargs=create_kwargs,
|
||||
custom_llm_provider="gemini",
|
||||
)
|
||||
|
||||
assert task is None
|
||||
|
||||
|
||||
def _register_poll(logging_obj: LitellmLogging, poll_fetch=None) -> asyncio.Task:
|
||||
import litellm.interactions.background_cost_polling as bg
|
||||
|
||||
if poll_fetch is None:
|
||||
poll_fetch, _ = _fetch_sequence(_response("in_progress", with_usage=False))
|
||||
context = _context(logging_obj)
|
||||
task = asyncio.create_task(poll_and_log_background_interaction_cost(context, fetch_interaction=poll_fetch))
|
||||
bg._ACTIVE_POLLS[context.interaction_id] = bg._ActiveBackgroundPoll(task=task, context=context)
|
||||
task.add_done_callback(lambda finished: bg._discard_poll(context.interaction_id, finished))
|
||||
return task
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_settlement_bills_an_interaction_paused_for_a_tool_result():
|
||||
logging_obj = _logging_obj()
|
||||
task = _register_poll(logging_obj)
|
||||
fetch, calls = _fetch_sequence(_response("requires_action", with_usage=True))
|
||||
|
||||
await maybe_settle_background_interaction_before_delete(
|
||||
interaction_id="interactions/bg-abc",
|
||||
fetch_interaction=fetch,
|
||||
)
|
||||
|
||||
assert len(calls) == 1
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
await asyncio.wait_for(task, timeout=5)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_settlement_bills_pending_background_interaction():
|
||||
logging_obj = _logging_obj()
|
||||
task = _register_poll(logging_obj)
|
||||
fetch, calls = _fetch_sequence(_response("completed", with_usage=True))
|
||||
|
||||
await maybe_settle_background_interaction_before_delete(
|
||||
interaction_id="interactions/bg-abc",
|
||||
fetch_interaction=fetch,
|
||||
)
|
||||
|
||||
assert len(calls) == 1
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
|
||||
await asyncio.wait_for(task, timeout=5)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_settlement_releases_reservation_when_still_in_progress():
|
||||
reservation = _reservation()
|
||||
logging_obj = _logging_obj_with_reservation(reservation)
|
||||
task = _register_poll(logging_obj)
|
||||
fetch, _ = _fetch_sequence(_response("in_progress", with_usage=False))
|
||||
|
||||
await maybe_settle_background_interaction_before_delete(
|
||||
interaction_id="interactions/bg-abc",
|
||||
fetch_interaction=fetch,
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
assert logging_obj.model_call_details.get("response_cost") is None
|
||||
await asyncio.wait_for(task, timeout=5)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_settlement_releases_reservation_when_prefetch_fails():
|
||||
reservation = _reservation()
|
||||
logging_obj = _logging_obj_with_reservation(reservation)
|
||||
task = _register_poll(logging_obj)
|
||||
fetch, _ = _fetch_sequence(RuntimeError("interaction already deleted"))
|
||||
|
||||
await maybe_settle_background_interaction_before_delete(
|
||||
interaction_id="interactions/bg-abc",
|
||||
fetch_interaction=fetch,
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
assert logging_obj.model_call_details.get("response_cost") is None
|
||||
await asyncio.wait_for(task, timeout=5)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_settlement_releases_reservation_when_billing_raises():
|
||||
reservation = _reservation()
|
||||
logging_obj = _logging_obj_with_reservation(reservation)
|
||||
task = _register_poll(logging_obj)
|
||||
fetch, _ = _fetch_sequence(_response("completed", with_usage=True))
|
||||
logging_obj.async_log_background_interaction_completion = _raise_on_billing
|
||||
|
||||
with pytest.raises(RuntimeError):
|
||||
await maybe_settle_background_interaction_before_delete(
|
||||
interaction_id="interactions/bg-abc",
|
||||
fetch_interaction=fetch,
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
await asyncio.wait_for(task, timeout=5)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_settlement_ignores_interactions_without_pending_poll():
|
||||
fetch, calls = _fetch_sequence(_response("completed", with_usage=True))
|
||||
|
||||
await maybe_settle_background_interaction_before_delete(
|
||||
interaction_id="interactions/never-polled",
|
||||
fetch_interaction=fetch,
|
||||
)
|
||||
|
||||
assert calls == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_settlement_noop_after_poll_task_finished():
|
||||
logging_obj = _logging_obj()
|
||||
poll_fetch, _ = _fetch_sequence(_response("completed", with_usage=True))
|
||||
task = _register_poll(logging_obj, poll_fetch=poll_fetch)
|
||||
await asyncio.wait_for(task, timeout=5)
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
|
||||
settle_fetch, settle_calls = _fetch_sequence(_response("completed", with_usage=True))
|
||||
await maybe_settle_background_interaction_before_delete(
|
||||
interaction_id="interactions/bg-abc",
|
||||
fetch_interaction=settle_fetch,
|
||||
)
|
||||
|
||||
assert settle_calls == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_settlement_does_not_rebill_when_gate_already_claimed():
|
||||
logging_obj = _logging_obj()
|
||||
logging_obj.model_call_details[_SETTLED_KEY] = True
|
||||
task = _register_poll(logging_obj)
|
||||
fetch, calls = _fetch_sequence(_response("completed", with_usage=True))
|
||||
|
||||
await maybe_settle_background_interaction_before_delete(
|
||||
interaction_id="interactions/bg-abc",
|
||||
fetch_interaction=fetch,
|
||||
)
|
||||
|
||||
assert len(calls) == 1
|
||||
assert logging_obj.model_call_details.get("response_cost") is None
|
||||
await asyncio.wait_for(task, timeout=5)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_exits_without_billing_once_settled_elsewhere():
|
||||
logging_obj = _logging_obj()
|
||||
logging_obj.model_call_details[_SETTLED_KEY] = True
|
||||
fetch, calls = _fetch_sequence(_response("completed", with_usage=True))
|
||||
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert calls == []
|
||||
assert logging_obj.model_call_details.get("response_cost") is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_schedule_respects_kill_switch(monkeypatch):
|
||||
import litellm.interactions.background_cost_polling as module
|
||||
|
||||
monkeypatch.setattr(module, "BACKGROUND_INTERACTION_COST_POLLING_ENABLED", False)
|
||||
|
||||
task = maybe_schedule_background_interaction_cost_polling(
|
||||
response=_response("in_progress", with_usage=False),
|
||||
create_kwargs={"litellm_logging_obj": _logging_obj()},
|
||||
custom_llm_provider="gemini",
|
||||
)
|
||||
|
||||
assert task is None
|
||||
|
||||
|
||||
def test_every_status_the_api_can_return_is_either_pollable_or_terminal():
|
||||
"""
|
||||
The proxy bills a usage-less create in exactly two ways: it polls the
|
||||
interaction until it settles, or it recognises the status as terminal and
|
||||
settles immediately. A status in neither set is billed by nobody, alerts
|
||||
nobody, and releases its budget reservation, which is the zero-spend bug
|
||||
this whole module exists to fix.
|
||||
|
||||
Pinned against the generated spec enum rather than a hand-written list, so
|
||||
a status Google adds later breaks this test instead of silently shipping
|
||||
another unbilled path.
|
||||
"""
|
||||
from litellm.interactions.background_cost_polling import _POLLABLE_STATUSES, _TERMINAL_STATUSES
|
||||
from litellm.types.interactions.generated import Status1
|
||||
|
||||
spec_statuses = {member.value for member in Status1}
|
||||
handled = _POLLABLE_STATUSES | _TERMINAL_STATUSES
|
||||
|
||||
assert spec_statuses - handled == set()
|
||||
assert handled - spec_statuses == set()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_schedule_creates_poll_task_for_queued_create():
|
||||
"""
|
||||
``queued`` is the API's not-started-yet state. It carries no usage, so the
|
||||
create cannot bill it, and it is not terminal, so nothing settles it:
|
||||
without a poll task it is never charged at all.
|
||||
"""
|
||||
logging_obj = _logging_obj()
|
||||
task = maybe_schedule_background_interaction_cost_polling(
|
||||
response=_response("queued", with_usage=False),
|
||||
create_kwargs={"litellm_logging_obj": logging_obj},
|
||||
custom_llm_provider="gemini",
|
||||
)
|
||||
|
||||
assert isinstance(task, asyncio.Task)
|
||||
task.cancel()
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await task
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_poller_bills_an_interaction_that_started_out_queued():
|
||||
logging_obj = _logging_obj()
|
||||
fetch, calls = _fetch_sequence(
|
||||
_response("queued", with_usage=False),
|
||||
_response("in_progress", with_usage=False),
|
||||
_response("completed", with_usage=True),
|
||||
)
|
||||
|
||||
await poll_and_log_background_interaction_cost(_context(logging_obj), fetch_interaction=fetch)
|
||||
|
||||
assert len(calls) == 3
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
|
||||
|
||||
|
||||
def test_poll_intervals_double_up_to_the_cap_and_stay_inside_the_timeout():
|
||||
"""
|
||||
The degenerate cases are covered above; this pins the shape the proxy
|
||||
actually ships, so an off-by-one in the doubling or in the remaining-budget
|
||||
check cannot pass green.
|
||||
"""
|
||||
intervals = list(_poll_intervals(initial=5.0, maximum=60.0, timeout=3600.0))
|
||||
|
||||
assert intervals[:6] == [5.0, 10.0, 20.0, 40.0, 60.0, 60.0]
|
||||
assert max(intervals) == 60.0
|
||||
assert sum(intervals) <= 3600.0
|
||||
assert sum(intervals) + 60.0 > 3600.0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_giving_up_on_an_unrecognized_status_says_which_status_it_was(monkeypatch):
|
||||
"""
|
||||
A status outside both sets polls for the full timeout and then gives up.
|
||||
The give-up line is the only trace it leaves, so it has to name the status
|
||||
rather than reporting it as an interaction that was merely still running.
|
||||
"""
|
||||
import litellm.interactions.background_cost_polling as bg
|
||||
|
||||
errors = []
|
||||
monkeypatch.setattr(bg.verbose_logger, "error", lambda *args, **kwargs: errors.append(args))
|
||||
|
||||
logging_obj = _logging_obj()
|
||||
fetch, _ = _fetch_sequence(_response("halted_for_review", with_usage=False))
|
||||
|
||||
await poll_and_log_background_interaction_cost(
|
||||
_context(logging_obj, timeout_seconds=0.01), fetch_interaction=fetch
|
||||
)
|
||||
|
||||
assert len(errors) == 1
|
||||
assert "halted_for_review" in errors[0]
|
||||
|
|
@ -0,0 +1,150 @@
|
|||
from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import (
|
||||
InteractionsUsageObjectTransformation,
|
||||
)
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
OMNI_VIDEO_USAGE = {
|
||||
"total_tokens": 18247,
|
||||
"total_input_tokens": 16,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 16}],
|
||||
"total_cached_tokens": 0,
|
||||
"total_output_tokens": 17937,
|
||||
"output_tokens_by_modality": [{"modality": "video", "tokens": 17376}],
|
||||
"total_tool_use_tokens": 0,
|
||||
"total_thought_tokens": 294,
|
||||
}
|
||||
|
||||
|
||||
def test_detects_interactions_usage_object():
|
||||
assert InteractionsUsageObjectTransformation.is_interactions_usage_object(OMNI_VIDEO_USAGE) is True
|
||||
|
||||
|
||||
def test_rejects_chat_and_responses_api_usage_objects():
|
||||
chat_usage = {"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30}
|
||||
responses_api_usage = {"input_tokens": 10, "output_tokens": 20, "total_tokens": 30}
|
||||
assert InteractionsUsageObjectTransformation.is_interactions_usage_object(chat_usage) is False
|
||||
assert InteractionsUsageObjectTransformation.is_interactions_usage_object(responses_api_usage) is False
|
||||
assert InteractionsUsageObjectTransformation.is_interactions_usage_object(None) is False
|
||||
assert InteractionsUsageObjectTransformation.is_interactions_usage_object("usage") is False
|
||||
|
||||
|
||||
def test_transforms_real_omni_video_usage_block():
|
||||
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(OMNI_VIDEO_USAGE)
|
||||
|
||||
assert isinstance(usage, Usage)
|
||||
assert usage.prompt_tokens == 16
|
||||
assert usage.completion_tokens == 17937 + 294
|
||||
assert usage.total_tokens == 18247
|
||||
assert usage.prompt_tokens_details is not None
|
||||
assert usage.prompt_tokens_details.text_tokens == 16
|
||||
assert usage.completion_tokens_details is not None
|
||||
assert usage.completion_tokens_details.video_tokens == 17376
|
||||
assert usage.completion_tokens_details.reasoning_tokens == 294
|
||||
|
||||
|
||||
def test_transforms_reasoning_tokens_spec_field_name():
|
||||
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
|
||||
{
|
||||
"total_input_tokens": 10,
|
||||
"total_output_tokens": 20,
|
||||
"total_reasoning_tokens": 5,
|
||||
}
|
||||
)
|
||||
assert usage.completion_tokens == 25
|
||||
assert usage.completion_tokens_details is not None
|
||||
assert usage.completion_tokens_details.reasoning_tokens == 5
|
||||
assert usage.total_tokens == 35
|
||||
|
||||
|
||||
def test_cached_tokens_subtracted_from_text_input():
|
||||
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
|
||||
{
|
||||
"total_input_tokens": 1000,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 1000}],
|
||||
"total_cached_tokens": 400,
|
||||
"total_output_tokens": 50,
|
||||
}
|
||||
)
|
||||
assert usage.prompt_tokens == 1000
|
||||
assert usage.prompt_tokens_details is not None
|
||||
assert usage.prompt_tokens_details.text_tokens == 600
|
||||
assert usage.prompt_tokens_details.cached_tokens == 400
|
||||
assert usage._cache_read_input_tokens == 400
|
||||
|
||||
|
||||
def test_cached_tokens_subtracted_per_modality_when_breakdown_present():
|
||||
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
|
||||
{
|
||||
"total_input_tokens": 1500,
|
||||
"input_tokens_by_modality": [
|
||||
{"modality": "text", "tokens": 1000},
|
||||
{"modality": "audio", "tokens": 500},
|
||||
],
|
||||
"total_cached_tokens": 300,
|
||||
"cached_tokens_by_modality": [{"modality": "audio", "tokens": 300}],
|
||||
"total_output_tokens": 50,
|
||||
}
|
||||
)
|
||||
assert usage.prompt_tokens_details is not None
|
||||
assert usage.prompt_tokens_details.text_tokens == 1000
|
||||
assert usage.prompt_tokens_details.audio_tokens == 200
|
||||
assert usage.prompt_tokens_details.cached_tokens == 300
|
||||
|
||||
|
||||
def test_tool_use_tokens_billed_as_input():
|
||||
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
|
||||
{
|
||||
"total_input_tokens": 100,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 100}],
|
||||
"total_tool_use_tokens": 40,
|
||||
"tool_use_tokens_by_modality": [{"modality": "text", "tokens": 40}],
|
||||
"total_output_tokens": 10,
|
||||
}
|
||||
)
|
||||
assert usage.prompt_tokens == 140
|
||||
assert usage.prompt_tokens_details is not None
|
||||
assert usage.prompt_tokens_details.text_tokens == 140
|
||||
|
||||
|
||||
def test_google_search_grounding_count_maps_to_web_search_requests():
|
||||
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
|
||||
{
|
||||
"total_input_tokens": 103,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 103}],
|
||||
"total_output_tokens": 226,
|
||||
"total_thought_tokens": 351,
|
||||
"grounding_tool_count": [
|
||||
{"type": "google_search", "count": 3},
|
||||
{"type": "url_context", "count": 2},
|
||||
],
|
||||
}
|
||||
)
|
||||
assert usage.prompt_tokens_details is not None
|
||||
assert usage.prompt_tokens_details.web_search_requests == 3
|
||||
|
||||
|
||||
def test_no_grounding_leaves_web_search_requests_unset():
|
||||
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
|
||||
{
|
||||
"total_input_tokens": 10,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 10}],
|
||||
"total_output_tokens": 5,
|
||||
}
|
||||
)
|
||||
assert usage.prompt_tokens_details is not None
|
||||
assert getattr(usage.prompt_tokens_details, "web_search_requests", None) is None
|
||||
|
||||
|
||||
def test_document_modality_folds_into_text():
|
||||
usage = InteractionsUsageObjectTransformation.transform_interactions_usage_object(
|
||||
{
|
||||
"total_input_tokens": 80,
|
||||
"input_tokens_by_modality": [
|
||||
{"modality": "text", "tokens": 30},
|
||||
{"modality": "document", "tokens": 50},
|
||||
],
|
||||
"total_output_tokens": 10,
|
||||
}
|
||||
)
|
||||
assert usage.prompt_tokens_details is not None
|
||||
assert usage.prompt_tokens_details.text_tokens == 80
|
||||
|
|
@ -378,7 +378,7 @@ def test_shipped_rules_flag_unmapped_fable_as_always_on_thinking(shipped_cost_ma
|
|||
"model,provider",
|
||||
[
|
||||
("claude-opus-4-9@20260101", "vertex_ai"),
|
||||
("databricks-claude-opus-5-1", "databricks"),
|
||||
("databricks-claude-haiku-5-1", "databricks"),
|
||||
],
|
||||
)
|
||||
def test_shipped_rules_are_provider_neutral_for_unmapped_ids(shipped_cost_map, model, provider):
|
||||
|
|
@ -388,6 +388,8 @@ def test_shipped_rules_are_provider_neutral_for_unmapped_ids(shipped_cost_map, m
|
|||
assert info["supports_adaptive_thinking"] is True
|
||||
assert info["supports_mid_conversation_system"] is True
|
||||
assert info["supports_function_calling"] is True
|
||||
assert not info.get("input_cost_per_token")
|
||||
assert not info.get("output_cost_per_token")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
|
|||
|
|
@ -4546,6 +4546,323 @@ def test_zero_token_video_usage_preserves_duration_seconds(logging_obj):
|
|||
assert payload["completion_tokens"] == 0
|
||||
|
||||
|
||||
INTERACTIONS_USAGE_BLOCK = {
|
||||
"total_tokens": 175,
|
||||
"total_input_tokens": 100,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 100}],
|
||||
"total_cached_tokens": 0,
|
||||
"total_output_tokens": 50,
|
||||
"output_tokens_by_modality": [{"modality": "text", "tokens": 50}],
|
||||
"total_tool_use_tokens": 0,
|
||||
"total_thought_tokens": 25,
|
||||
}
|
||||
|
||||
|
||||
def _interactions_logging_obj(stream: bool, call_type: str = "acreate"):
|
||||
logging_obj = LitellmLogging(
|
||||
model="gemini-2.5-flash",
|
||||
messages=[],
|
||||
stream=stream,
|
||||
call_type=call_type,
|
||||
start_time=time.time(),
|
||||
litellm_call_id="interactions-call-id",
|
||||
function_id="interactions-fn-id",
|
||||
)
|
||||
logging_obj.update_environment_variables(
|
||||
litellm_params={},
|
||||
optional_params={},
|
||||
model="gemini-2.5-flash",
|
||||
custom_llm_provider="gemini",
|
||||
input="hi",
|
||||
)
|
||||
return logging_obj
|
||||
|
||||
|
||||
@pytest.mark.parametrize("call_type", ["create", "acreate", "create_interaction", "acreate_interaction"])
|
||||
def test_interactions_response_is_recognized_for_logging(call_type):
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logging_obj = _interactions_logging_obj(stream=False, call_type=call_type)
|
||||
response = InteractionsAPIResponse(
|
||||
id="interactions/abc",
|
||||
model="gemini-2.5-flash",
|
||||
status="completed",
|
||||
usage=dict(INTERACTIONS_USAGE_BLOCK),
|
||||
)
|
||||
assert logging_obj._is_recognized_call_type_for_logging(logging_result=response) is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("call_type", ["acreate", "acreate_interaction"])
|
||||
def test_in_progress_background_create_is_not_billed(call_type):
|
||||
import datetime as dt
|
||||
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logging_obj = _interactions_logging_obj(stream=False, call_type=call_type)
|
||||
response = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
|
||||
|
||||
assert logging_obj._is_recognized_call_type_for_logging(logging_result=response) is False
|
||||
|
||||
logging_obj._success_handler_helper_fn(
|
||||
result=response,
|
||||
start_time=dt.datetime.now(),
|
||||
end_time=dt.datetime.now(),
|
||||
cache_hit=False,
|
||||
)
|
||||
|
||||
assert logging_obj.model_call_details.get("response_cost") is None
|
||||
assert logging_obj.model_call_details.get("standard_logging_object") is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_background_interaction_completion_rebills_after_in_progress_success():
|
||||
import datetime as dt
|
||||
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logging_obj = _interactions_logging_obj(stream=False)
|
||||
in_progress = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
|
||||
await logging_obj.async_success_handler(
|
||||
result=in_progress,
|
||||
start_time=dt.datetime.now(),
|
||||
end_time=dt.datetime.now(),
|
||||
)
|
||||
|
||||
assert logging_obj.model_call_details.get("response_cost") is None
|
||||
assert logging_obj.should_run_logging(event_type="async_success") is False
|
||||
|
||||
completed = InteractionsAPIResponse(
|
||||
id="interactions/abc",
|
||||
model="gemini-2.5-flash",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage=dict(INTERACTIONS_USAGE_BLOCK),
|
||||
)
|
||||
await logging_obj.async_log_background_interaction_completion(result=completed)
|
||||
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_background_interaction_completion_prices_the_settled_body_itself():
|
||||
"""
|
||||
The poll fetches the settled body through its own client call, which
|
||||
prices it against a throwaway logging object holding none of this
|
||||
request's deployment context. Adopting that price would bill a
|
||||
custom-priced deployment at the wrong rate, and it would also satisfy the
|
||||
"already calculated" shortcut and skip repricing, leaving the breakdown at
|
||||
the zeros the usage-less create stamped and writing those to the spend log.
|
||||
"""
|
||||
import datetime as dt
|
||||
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logging_obj = _interactions_logging_obj(stream=False)
|
||||
in_progress = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
|
||||
await logging_obj.async_success_handler(
|
||||
result=in_progress,
|
||||
start_time=dt.datetime.now(),
|
||||
end_time=dt.datetime.now(),
|
||||
)
|
||||
|
||||
completed = InteractionsAPIResponse(
|
||||
id="interactions/abc",
|
||||
model="gemini-2.5-flash",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage=dict(INTERACTIONS_USAGE_BLOCK),
|
||||
)
|
||||
completed._hidden_params = {"response_cost": 99.0}
|
||||
|
||||
await logging_obj.async_log_background_interaction_completion(result=completed)
|
||||
|
||||
response_cost = logging_obj.model_call_details["response_cost"]
|
||||
assert response_cost != 99.0
|
||||
assert response_cost > 0
|
||||
|
||||
cost_breakdown = logging_obj.model_call_details["standard_logging_object"]["cost_breakdown"]
|
||||
assert cost_breakdown["total_cost"] == response_cost
|
||||
assert cost_breakdown["input_cost"] > 0
|
||||
assert cost_breakdown["output_cost"] > 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_background_interaction_completion_lets_otel_emit_the_cost_span():
|
||||
"""
|
||||
OTEL, and every integration that derives from it, dedupes span emission on
|
||||
a marker kept in the request's own metadata. The in-progress create claims
|
||||
that marker, so without clearing it the settled completion, the only event
|
||||
carrying usage and cost, is discarded as a duplicate and every
|
||||
OTEL-family backend shows the interaction as a span with no cost at all.
|
||||
"""
|
||||
import datetime as dt
|
||||
|
||||
from litellm.integrations.opentelemetry import OpenTelemetry, OpenTelemetryConfig
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
otel = OpenTelemetry(config=OpenTelemetryConfig(exporter="console"))
|
||||
logging_obj = _interactions_logging_obj(stream=False)
|
||||
in_progress = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
|
||||
await logging_obj.async_success_handler(
|
||||
result=in_progress,
|
||||
start_time=dt.datetime.now(),
|
||||
end_time=dt.datetime.now(),
|
||||
)
|
||||
|
||||
assert otel._emit_once(logging_obj.model_call_details, "success") is True
|
||||
assert otel._emit_once(logging_obj.model_call_details, "success") is False
|
||||
|
||||
completed = InteractionsAPIResponse(
|
||||
id="interactions/abc",
|
||||
model="gemini-2.5-flash",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage=dict(INTERACTIONS_USAGE_BLOCK),
|
||||
)
|
||||
await logging_obj.async_log_background_interaction_completion(result=completed)
|
||||
|
||||
assert otel._emit_once(logging_obj.model_call_details, "success") is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"call_type",
|
||||
["aget", "get", "aget_interaction", "adelete_interaction", "acancel_interaction"],
|
||||
)
|
||||
def test_interactions_get_poll_is_not_billed(call_type):
|
||||
import datetime as dt
|
||||
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logging_obj = _interactions_logging_obj(stream=False, call_type=call_type)
|
||||
response = InteractionsAPIResponse(
|
||||
id="interactions/abc",
|
||||
model="gemini-2.5-flash",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage=dict(INTERACTIONS_USAGE_BLOCK),
|
||||
)
|
||||
|
||||
assert logging_obj._is_recognized_call_type_for_logging(logging_result=response) is False
|
||||
|
||||
logging_obj._success_handler_helper_fn(
|
||||
result=response,
|
||||
start_time=dt.datetime.now(),
|
||||
end_time=dt.datetime.now(),
|
||||
cache_hit=False,
|
||||
)
|
||||
|
||||
assert logging_obj.model_call_details.get("response_cost") is None
|
||||
assert logging_obj.model_call_details.get("standard_logging_object") is None
|
||||
|
||||
|
||||
def test_non_streaming_interactions_success_sets_response_cost_and_usage():
|
||||
import datetime as dt
|
||||
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logging_obj = _interactions_logging_obj(stream=False)
|
||||
response = InteractionsAPIResponse(
|
||||
id="interactions/abc",
|
||||
model="gemini-2.5-flash",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage=dict(INTERACTIONS_USAGE_BLOCK),
|
||||
)
|
||||
|
||||
logging_obj._success_handler_helper_fn(
|
||||
result=response,
|
||||
start_time=dt.datetime.now(),
|
||||
end_time=dt.datetime.now(),
|
||||
cache_hit=False,
|
||||
)
|
||||
|
||||
assert logging_obj.model_call_details["response_cost"] > 0
|
||||
standard_logging_object = logging_obj.model_call_details["standard_logging_object"]
|
||||
assert standard_logging_object["prompt_tokens"] == 100
|
||||
assert standard_logging_object["completion_tokens"] == 75
|
||||
assert standard_logging_object["total_tokens"] == 175
|
||||
assert standard_logging_object["response_cost"] == logging_obj.model_call_details["response_cost"]
|
||||
|
||||
|
||||
def test_assembled_streaming_response_from_completed_interaction_event():
|
||||
import datetime as dt
|
||||
|
||||
from litellm.types.interactions import (
|
||||
InteractionsAPIResponse,
|
||||
InteractionsAPIStreamingResponse,
|
||||
)
|
||||
|
||||
logging_obj = _interactions_logging_obj(stream=True)
|
||||
completed_event = InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.completed",
|
||||
interaction={
|
||||
"id": "interactions/abc",
|
||||
"model": "gemini-2.5-flash",
|
||||
"status": "completed",
|
||||
"steps": [],
|
||||
"usage": dict(INTERACTIONS_USAGE_BLOCK),
|
||||
},
|
||||
)
|
||||
|
||||
assembled = logging_obj._get_assembled_streaming_response(
|
||||
result=completed_event,
|
||||
start_time=dt.datetime.now(),
|
||||
end_time=dt.datetime.now(),
|
||||
is_async=True,
|
||||
streaming_chunks=[],
|
||||
)
|
||||
|
||||
assert isinstance(assembled, InteractionsAPIResponse)
|
||||
assert assembled.usage == INTERACTIONS_USAGE_BLOCK
|
||||
|
||||
in_progress_event = InteractionsAPIStreamingResponse(event_type="interaction.in_progress")
|
||||
assert (
|
||||
logging_obj._get_assembled_streaming_response(
|
||||
result=in_progress_event,
|
||||
start_time=dt.datetime.now(),
|
||||
end_time=dt.datetime.now(),
|
||||
is_async=True,
|
||||
streaming_chunks=[],
|
||||
)
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
def test_assembled_streaming_response_from_legacy_completed_chunk():
|
||||
from litellm.types.interactions import (
|
||||
InteractionsAPIResponse,
|
||||
InteractionsAPIStreamingResponse,
|
||||
)
|
||||
|
||||
legacy_chunk = InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.complete",
|
||||
id="interactions/legacy",
|
||||
model="gemini-2.5-flash",
|
||||
status="completed",
|
||||
outputs=[],
|
||||
usage=dict(INTERACTIONS_USAGE_BLOCK),
|
||||
)
|
||||
|
||||
assembled = LitellmLogging._assemble_completed_interaction_response(legacy_chunk)
|
||||
|
||||
assert isinstance(assembled, InteractionsAPIResponse)
|
||||
assert assembled.id == "interactions/legacy"
|
||||
assert assembled.usage == INTERACTIONS_USAGE_BLOCK
|
||||
|
||||
|
||||
def test_standard_logging_payload_maps_interactions_usage():
|
||||
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
|
||||
|
||||
usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(
|
||||
response_obj={"usage": dict(INTERACTIONS_USAGE_BLOCK)}
|
||||
)
|
||||
|
||||
assert usage.prompt_tokens == 100
|
||||
assert usage.completion_tokens == 75
|
||||
assert usage.total_tokens == 175
|
||||
|
||||
|
||||
def test_pre_call_does_not_pin_request_in_module_state(logging_obj):
|
||||
"""
|
||||
pre_call/post_call must not stash their locals (full messages, the Logging
|
||||
|
|
|
|||
|
|
@ -359,6 +359,48 @@ def test_translate_anthropic_messages_to_openai_thinking_blocks():
|
|||
assert result[1]["tool_calls"][0]["id"] == "toolu_01234"
|
||||
|
||||
|
||||
def test_translate_anthropic_messages_to_openai_sets_reasoning_content():
|
||||
"""Reasoning-aware chat providers read reasoning_content, so thinking text must land there.
|
||||
|
||||
Without it Moonshot and DeepSeek fill in a single-space placeholder and the model gets
|
||||
a blank where its own prior reasoning belongs.
|
||||
"""
|
||||
|
||||
anthropic_messages = [
|
||||
AnthropicMessagesUserMessageParam(
|
||||
role="user",
|
||||
content=[{"type": "text", "text": "Which city is best for a picnic?"}],
|
||||
),
|
||||
AnthopicMessagesAssistantMessageParam(
|
||||
role="assistant",
|
||||
content=[
|
||||
{"type": "thinking", "thinking": "Denver is dry in August.", "signature": "sig1"},
|
||||
{"type": "thinking", "thinking": "San Francisco is foggy.", "signature": "sig2"},
|
||||
{"type": "redacted_thinking", "data": "REDACTED"},
|
||||
{"type": "text", "text": "Denver."},
|
||||
],
|
||||
),
|
||||
]
|
||||
|
||||
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(messages=anthropic_messages)
|
||||
|
||||
assert result[1]["reasoning_content"] == "Denver is dry in August.\nSan Francisco is foggy."
|
||||
assert result[1]["content"] == "Denver."
|
||||
|
||||
|
||||
def test_translate_anthropic_messages_to_openai_sets_no_reasoning_content_without_thinking():
|
||||
anthropic_messages = [
|
||||
AnthopicMessagesAssistantMessageParam(
|
||||
role="assistant",
|
||||
content=[{"type": "text", "text": "Denver."}],
|
||||
),
|
||||
]
|
||||
|
||||
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(messages=anthropic_messages)
|
||||
|
||||
assert "reasoning_content" not in result[0]
|
||||
|
||||
|
||||
def test_translate_anthropic_messages_to_openai_tool_message_placement():
|
||||
"""Test that tool result messages are placed before user messages in the conversation order."""
|
||||
|
||||
|
|
|
|||
|
|
@ -144,9 +144,15 @@ class TestReasoningItemWithoutSummaryText:
|
|||
("content_block_delta", 1),
|
||||
("content_block_stop", 1),
|
||||
]
|
||||
assert chunks[1]["content_block"] == {"type": "thinking", "thinking": ""}
|
||||
assert chunks[1]["content_block"] == {"type": "thinking", "thinking": "", "signature": ""}
|
||||
assert "".join(c["delta"]["thinking"] for c in chunks[2:4]) == "Weighing options"
|
||||
|
||||
def test_the_reasoning_item_id_is_never_streamed_as_a_signature(self):
|
||||
"""A stand-in signature would be replayed as a real one, so none is ever sent."""
|
||||
chunks = _drain_async(self._gpt_turn(reasoning_summary_deltas=["Weighing options"]))
|
||||
|
||||
assert not [c for c in chunks if c.get("delta", {}).get("type") == "signature_delta"]
|
||||
|
||||
|
||||
class TestToolUseBlockClosedExactlyOnce:
|
||||
"""Regression for https://github.com/BerriAI/litellm/issues/37273.
|
||||
|
|
|
|||
|
|
@ -486,8 +486,8 @@ class TestTranslateMessagesToResponsesInput:
|
|||
}
|
||||
]
|
||||
|
||||
def test_assistant_thinking_block_becomes_output_text(self):
|
||||
"""Assistant thinking block text is included as output_text."""
|
||||
def test_assistant_thinking_block_becomes_reasoning_item(self):
|
||||
"""Assistant thinking block becomes a reasoning item, never visible assistant prose."""
|
||||
messages = [
|
||||
{
|
||||
"role": "assistant",
|
||||
|
|
@ -495,7 +495,77 @@ class TestTranslateMessagesToResponsesInput:
|
|||
}
|
||||
]
|
||||
result = _translate_messages(messages)
|
||||
assert result[0]["content"] == [{"type": "output_text", "text": "Let me reason step by step."}]
|
||||
assert result == [
|
||||
{
|
||||
"type": "reasoning",
|
||||
"summary": [{"type": "summary_text", "text": "Let me reason step by step."}],
|
||||
}
|
||||
]
|
||||
|
||||
def test_reasoning_item_carries_no_id(self):
|
||||
"""A fabricated reasoning id 404s upstream, so the item must go out without one."""
|
||||
messages = [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [{"type": "thinking", "thinking": "Private reasoning.", "signature": "rs_abc123"}],
|
||||
}
|
||||
]
|
||||
result = _translate_messages(messages)
|
||||
assert "id" not in result[0]
|
||||
|
||||
def test_consecutive_thinking_blocks_become_one_reasoning_item(self):
|
||||
"""Summary parts of one upstream reasoning item are regrouped into that item."""
|
||||
messages = [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "thinking", "thinking": "First part."},
|
||||
{"type": "thinking", "thinking": "Second part."},
|
||||
],
|
||||
}
|
||||
]
|
||||
result = _translate_messages(messages)
|
||||
assert result == [
|
||||
{
|
||||
"type": "reasoning",
|
||||
"summary": [
|
||||
{"type": "summary_text", "text": "First part."},
|
||||
{"type": "summary_text", "text": "Second part."},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
def test_a_tool_call_splits_the_reasoning_items_around_it(self):
|
||||
"""Thinking on either side of a tool call belongs to two different reasoning items."""
|
||||
messages = [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "thinking", "thinking": "Before the call."},
|
||||
{"type": "tool_use", "id": "call_1", "name": "get_weather", "input": {"city": "Denver"}},
|
||||
{"type": "thinking", "thinking": "After the call."},
|
||||
],
|
||||
}
|
||||
]
|
||||
result = _translate_messages(messages)
|
||||
assert [item["type"] for item in result] == ["reasoning", "function_call", "reasoning"]
|
||||
assert result[0]["summary"] == [{"type": "summary_text", "text": "Before the call."}]
|
||||
assert result[2]["summary"] == [{"type": "summary_text", "text": "After the call."}]
|
||||
|
||||
def test_thinking_and_text_stay_separate(self):
|
||||
"""The visible answer stays the only thing in the assistant message."""
|
||||
messages = [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "thinking", "thinking": "The user wants Denver."},
|
||||
{"type": "text", "text": "Denver is the best pick."},
|
||||
],
|
||||
}
|
||||
]
|
||||
result = _translate_messages(messages)
|
||||
assert [item["type"] for item in result] == ["reasoning", "message"]
|
||||
assert result[1]["content"] == [{"type": "output_text", "text": "Denver is the best pick."}]
|
||||
|
||||
def test_assistant_empty_thinking_block_skipped(self):
|
||||
"""Assistant thinking block with empty thinking text is skipped."""
|
||||
|
|
@ -1094,7 +1164,7 @@ def _make_function_call_item(call_id: str, name: str, arguments: str) -> MagicMo
|
|||
return item
|
||||
|
||||
|
||||
def _make_reasoning_item(summaries: List[str]) -> MagicMock:
|
||||
def _make_reasoning_item(summaries: List[str], item_id: str = "rs_test_1") -> MagicMock:
|
||||
"""Build a mock ResponseReasoningItem."""
|
||||
from openai.types.responses import ResponseReasoningItem # type: ignore[import]
|
||||
|
||||
|
|
@ -1105,6 +1175,7 @@ def _make_reasoning_item(summaries: List[str]) -> MagicMock:
|
|||
summary_mocks.append(s)
|
||||
|
||||
item = MagicMock(spec=ResponseReasoningItem)
|
||||
item.id = item_id
|
||||
item.summary = summary_mocks
|
||||
return item
|
||||
|
||||
|
|
@ -1178,6 +1249,53 @@ class TestTranslateResponse:
|
|||
result: Any = _ADAPTER.translate_response(response)
|
||||
assert result["content"] == []
|
||||
|
||||
def test_null_summary_text_skipped_rather_than_stringified(self):
|
||||
"""A summary part whose text is null must not reach the client as the word "None"."""
|
||||
response = _make_mock_response(
|
||||
output=[
|
||||
{
|
||||
"type": "reasoning",
|
||||
"id": "rs_null_1",
|
||||
"summary": [{"type": "summary_text", "text": None}],
|
||||
}
|
||||
]
|
||||
)
|
||||
result: Any = _ADAPTER.translate_response(response)
|
||||
assert result["content"] == []
|
||||
|
||||
def test_reasoning_item_id_never_becomes_a_thinking_signature(self):
|
||||
"""Only Anthropic can sign a thinking block, so a stand-in signature is never invented."""
|
||||
reasoning = _make_reasoning_item(["Part one.", "Part two."], item_id="rs_abc123")
|
||||
response = _make_mock_response(output=[reasoning])
|
||||
result: Any = _ADAPTER.translate_response(response)
|
||||
assert [block["signature"] for block in result["content"]] == [None, None]
|
||||
|
||||
def test_dict_reasoning_item_becomes_thinking_block(self):
|
||||
"""A reasoning item arriving as a plain dict is kept, not dropped."""
|
||||
response = _make_mock_response(
|
||||
output=[
|
||||
{
|
||||
"type": "reasoning",
|
||||
"id": "rs_dict_1",
|
||||
"summary": [{"type": "summary_text", "text": "Weighing the options."}],
|
||||
}
|
||||
]
|
||||
)
|
||||
result: Any = _ADAPTER.translate_response(response)
|
||||
assert result["content"] == [
|
||||
{"type": "thinking", "thinking": "Weighing the options.", "signature": None}
|
||||
]
|
||||
|
||||
def test_thinking_blocks_are_dropped_when_replayed_to_anthropic(self):
|
||||
"""Replaying this turn to an Anthropic model must not send a signature it cannot verify."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
_drop_unsignable_thinking_blocks,
|
||||
)
|
||||
|
||||
response = _make_mock_response(output=[_make_reasoning_item(["Part one."], item_id="rs_abc123")])
|
||||
result: Any = _ADAPTER.translate_response(response)
|
||||
assert _drop_unsignable_thinking_blocks(result["content"]) == []
|
||||
|
||||
def test_usage_mapped_correctly(self):
|
||||
"""Input/output tokens from ResponseAPIUsage are mapped to AnthropicUsage."""
|
||||
response = _make_mock_response(
|
||||
|
|
|
|||
|
|
@ -413,6 +413,7 @@ def test_select_azure_base_url_called(setup_mocks):
|
|||
"avector_store_create",
|
||||
"avector_store_search",
|
||||
"acreate_skill",
|
||||
"acreate_interaction",
|
||||
]
|
||||
],
|
||||
)
|
||||
|
|
|
|||
|
|
@ -300,6 +300,7 @@ def test_azure_ai_strips_non_openai_spec_message_fields():
|
|||
"cache_control": {"type": "ephemeral"},
|
||||
}
|
||||
],
|
||||
"reasoning_content": "The user wants me to read a file.",
|
||||
"provider_specific_fields": {"thought_signature": "sig-top"},
|
||||
"tool_calls": [
|
||||
{
|
||||
|
|
@ -327,6 +328,7 @@ def test_azure_ai_strips_non_openai_spec_message_fields():
|
|||
transformed_messages = request["messages"]
|
||||
|
||||
assert not _find_key_anywhere(transformed_messages, "thinking_blocks")
|
||||
assert not _find_key_anywhere(transformed_messages, "reasoning_content")
|
||||
assert not _find_key_anywhere(transformed_messages, "provider_specific_fields")
|
||||
assert not _find_key_anywhere(transformed_messages, "cache_control")
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,264 @@
|
|||
import json
|
||||
from decimal import Decimal
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.databricks.cost_calculator import cost_per_token
|
||||
from litellm.types.utils import ModelInfo, Usage
|
||||
|
||||
REPO_ROOT: Final = Path(__file__).parents[4]
|
||||
MAIN_PRICES: Final = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
BACKUP_PRICES: Final = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
NEW_MODELS: Final = (
|
||||
"databricks/databricks-claude-opus-4-7",
|
||||
"databricks/databricks-claude-opus-4-8",
|
||||
"databricks/databricks-claude-opus-5",
|
||||
"databricks/databricks-claude-sonnet-5",
|
||||
"databricks/databricks-claude-fable-5",
|
||||
)
|
||||
|
||||
DOLLARS_PER_DBU: Final = Decimal("0.070")
|
||||
PRICE_FIELDS: Final = (
|
||||
"input_cost_per_token",
|
||||
"output_cost_per_token",
|
||||
"cache_creation_input_token_cost",
|
||||
"cache_read_input_token_cost",
|
||||
)
|
||||
PUBLISHED_DBU_PER_MILLION: Final = {
|
||||
"databricks/databricks-claude-fable-5": ("142.858", "714.286", "178.572", "14.286"),
|
||||
"databricks/databricks-claude-opus-5": ("71.429", "357.143", "89.286", "7.143"),
|
||||
"databricks/databricks-claude-opus-4-8": ("71.429", "357.143", "89.286", "7.143"),
|
||||
"databricks/databricks-claude-opus-4-7": ("71.429", "357.143", "89.286", "7.143"),
|
||||
"databricks/databricks-claude-opus-4-6": ("71.429", "357.143", "89.286", "7.143"),
|
||||
"databricks/databricks-claude-opus-4-5": ("71.429", "357.143", "89.286", "7.143"),
|
||||
"databricks/databricks-claude-opus-4-1": ("214.286", "1071.429", "267.857", "21.429"),
|
||||
"databricks/databricks-claude-opus-4": ("214.286", "1071.429", "267.857", "21.429"),
|
||||
"databricks/databricks-claude-sonnet-5": ("42.857", "214.286", "53.571", "4.286"),
|
||||
"databricks/databricks-claude-sonnet-4-6": ("42.857", "214.286", "53.571", "4.286"),
|
||||
"databricks/databricks-claude-sonnet-4-5": ("42.857", "214.286", "53.571", "4.286"),
|
||||
"databricks/databricks-claude-sonnet-4-1": ("42.857", "214.286", "53.571", "4.286"),
|
||||
"databricks/databricks-claude-sonnet-4": ("42.857", "214.286", "53.571", "4.286"),
|
||||
"databricks/databricks-claude-3-7-sonnet": ("42.857", "214.286", "53.571", "4.286"),
|
||||
"databricks/databricks-claude-haiku-4-5": ("14.286", "71.429", "17.857", "1.429"),
|
||||
"databricks/databricks-gpt-5": ("17.857", "142.857", "17.857", "1.786"),
|
||||
"databricks/databricks-gpt-5-1": ("17.857", "142.857", "17.857", "1.786"),
|
||||
"databricks/databricks-gpt-5-1-codex-max": ("17.857", "142.857", "17.857", "1.786"),
|
||||
"databricks/databricks-gpt-5-1-codex-mini": ("3.571", "28.571", "3.571", "0.357"),
|
||||
"databricks/databricks-gpt-5-mini": ("3.571", "28.571", "3.571", "0.357"),
|
||||
"databricks/databricks-gpt-5-nano": ("0.714", "5.714", "0.714", "0.071"),
|
||||
"databricks/databricks-gpt-5-2": ("25.000", "200.000", "25.000", "2.500"),
|
||||
"databricks/databricks-gpt-5-2-codex": ("25.000", "200.000", "25.000", "2.500"),
|
||||
"databricks/databricks-gpt-5-3-codex": ("25.000", "200.000", "25.000", "2.500"),
|
||||
"databricks/databricks-gpt-5-4": ("35.714", "214.286", "35.714", "3.571"),
|
||||
"databricks/databricks-gpt-5-4-mini": ("10.714", "64.286", "10.714", "1.071"),
|
||||
"databricks/databricks-gpt-5-4-nano": ("2.857", "17.857", "2.857", "0.286"),
|
||||
"databricks/databricks-gemini-3-1-pro": ("35.714", "214.286", "35.714", "3.571"),
|
||||
"databricks/databricks-gemini-3-pro": ("35.714", "214.286", "35.714", "3.571"),
|
||||
"databricks/databricks-gemini-3-flash": ("8.929", "53.571", "8.929", "0.893"),
|
||||
"databricks/databricks-gemini-3-1-flash-lite": ("4.464", "26.786", "4.464", "0.446"),
|
||||
"databricks/databricks-gemini-2-5-pro": ("22.321", "178.571", "22.321", "2.232"),
|
||||
"databricks/databricks-gemini-2-5-flash": ("5.357", "44.643", "5.357", "0.536"),
|
||||
}
|
||||
PROMOTIONAL_DISCOUNT: Final = 0.80
|
||||
PROMOTION_EXPIRES: Final = "2027-01-31"
|
||||
ENTRIES_STORING_PROMOTIONAL_RATE: Final = (
|
||||
"databricks/databricks-gemini-2-5-pro",
|
||||
"databricks/databricks-gemini-2-5-flash",
|
||||
)
|
||||
ENTRIES_STORING_LIST_RATE_DESPITE_PROMOTION: Final = (
|
||||
"databricks/databricks-gemini-3-1-pro",
|
||||
"databricks/databricks-gemini-3-pro",
|
||||
"databricks/databricks-gemini-3-flash",
|
||||
"databricks/databricks-gemini-3-1-flash-lite",
|
||||
)
|
||||
CACHE_FIELDS: Final = ("cache_creation_input_token_cost", "cache_read_input_token_cost")
|
||||
|
||||
|
||||
def _model_info(model: str) -> ModelInfo:
|
||||
return litellm.get_model_info(model=model, custom_llm_provider="databricks")
|
||||
|
||||
|
||||
def _dollars_per_token(dbu_per_million: str) -> float:
|
||||
return float(Decimal(dbu_per_million) * DOLLARS_PER_DBU / Decimal(10) ** 6)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
"databricks/databricks-claude-opus-4-8",
|
||||
"databricks/databricks-claude-opus-5",
|
||||
"databricks/databricks-claude-sonnet-5",
|
||||
],
|
||||
)
|
||||
def test_cached_tokens_bill_at_cache_rates(local_model_cost_map: None, model: str) -> None:
|
||||
info: Final = _model_info(model)
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=11000,
|
||||
completion_tokens=500,
|
||||
total_tokens=11500,
|
||||
cache_creation_input_tokens=2000,
|
||||
cache_read_input_tokens=8000,
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
|
||||
|
||||
assert prompt_cost == pytest.approx(
|
||||
1000 * info["input_cost_per_token"]
|
||||
+ 2000 * info["cache_creation_input_token_cost"]
|
||||
+ 8000 * info["cache_read_input_token_cost"]
|
||||
)
|
||||
assert completion_cost == pytest.approx(500 * info["output_cost_per_token"])
|
||||
assert prompt_cost < 11000 * info["input_cost_per_token"]
|
||||
|
||||
|
||||
def test_uncached_request_bills_every_prompt_token_at_the_input_rate(local_model_cost_map: None) -> None:
|
||||
model: Final = "databricks/databricks-claude-sonnet-5"
|
||||
info: Final = _model_info(model)
|
||||
usage: Final = Usage(prompt_tokens=1000, completion_tokens=200, total_tokens=1200)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
|
||||
|
||||
assert prompt_cost == pytest.approx(1000 * info["input_cost_per_token"])
|
||||
assert completion_cost == pytest.approx(200 * info["output_cost_per_token"])
|
||||
|
||||
|
||||
def test_legacy_endpoint_names_still_resolve(local_model_cost_map: None) -> None:
|
||||
info: Final = _model_info("databricks/databricks-mixtral-8x7b-instruct")
|
||||
usage: Final = Usage(prompt_tokens=100, completion_tokens=100, total_tokens=200)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="databricks/mixtral-8x7b-instruct-v0.1", usage=usage)
|
||||
|
||||
assert prompt_cost == pytest.approx(100 * info["input_cost_per_token"])
|
||||
assert completion_cost == pytest.approx(100 * info["output_cost_per_token"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", NEW_MODELS)
|
||||
def test_new_models_price_at_published_dbu_rates(local_model_cost_map: None, model: str) -> None:
|
||||
info: Final = _model_info(model)
|
||||
|
||||
for field, dbu_per_million in zip(PRICE_FIELDS, PUBLISHED_DBU_PER_MILLION[model]):
|
||||
assert info[field] == _dollars_per_token(dbu_per_million), field
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", sorted(set(PUBLISHED_DBU_PER_MILLION) - set(ENTRIES_STORING_PROMOTIONAL_RATE)))
|
||||
def test_cache_rates_derive_from_published_cache_dbu(local_model_cost_map: None, model: str) -> None:
|
||||
info: Final = _model_info(model)
|
||||
cache_dbu_per_million: Final = PUBLISHED_DBU_PER_MILLION[model][2:]
|
||||
|
||||
for field, dbu_per_million in zip(CACHE_FIELDS, cache_dbu_per_million):
|
||||
assert info[field] == _dollars_per_token(dbu_per_million), field
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", NEW_MODELS)
|
||||
def test_new_models_carry_cache_pricing(local_model_cost_map: None, model: str) -> None:
|
||||
info: Final = _model_info(model)
|
||||
|
||||
assert info["input_cost_per_token"] > 0
|
||||
assert info["output_cost_per_token"] > 0
|
||||
assert info["cache_creation_input_token_cost"] > info["input_cost_per_token"]
|
||||
assert info["cache_read_input_token_cost"] < info["input_cost_per_token"]
|
||||
assert info["supports_prompt_caching"] is True
|
||||
|
||||
|
||||
def test_every_priced_databricks_model_declares_cache_rates(local_model_cost_map: None) -> None:
|
||||
undeclared: Final = [
|
||||
model
|
||||
for model, info in litellm.model_cost.items()
|
||||
if model.startswith("databricks/")
|
||||
and info.get("input_cost_per_token") is not None
|
||||
and any(info.get(field) is None for field in CACHE_FIELDS)
|
||||
]
|
||||
|
||||
assert undeclared == []
|
||||
|
||||
|
||||
def test_models_without_a_cache_discount_bill_cache_tokens_at_the_input_rate(
|
||||
local_model_cost_map: None,
|
||||
) -> None:
|
||||
model: Final = "databricks/databricks-meta-llama-3-3-70b-instruct"
|
||||
info: Final = _model_info(model)
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=10000,
|
||||
completion_tokens=100,
|
||||
total_tokens=10100,
|
||||
cache_read_input_tokens=8000,
|
||||
)
|
||||
|
||||
prompt_cost, _ = cost_per_token(model=model, usage=usage)
|
||||
|
||||
assert prompt_cost == pytest.approx(10000 * info["input_cost_per_token"])
|
||||
assert prompt_cost > 8000 * info["input_cost_per_token"]
|
||||
|
||||
|
||||
def test_every_model_without_published_cache_dbu_bills_cache_at_its_own_input_rate(
|
||||
local_model_cost_map: None,
|
||||
) -> None:
|
||||
without_published_rates: Final = [
|
||||
model
|
||||
for model, info in litellm.model_cost.items()
|
||||
if model.startswith("databricks/")
|
||||
and info.get("input_cost_per_token")
|
||||
and model not in PUBLISHED_DBU_PER_MILLION
|
||||
]
|
||||
|
||||
assert len(without_published_rates) == 14
|
||||
for model in without_published_rates:
|
||||
info = _model_info(model)
|
||||
for field in CACHE_FIELDS:
|
||||
assert info[field] == pytest.approx(info["input_cost_per_token"]), (model, field)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", NEW_MODELS)
|
||||
def test_backup_price_map_matches_main(model: str) -> None:
|
||||
main_cost: Final = json.loads(MAIN_PRICES.read_text())
|
||||
backup_cost: Final = json.loads(BACKUP_PRICES.read_text())
|
||||
|
||||
assert model in main_cost
|
||||
assert model in backup_cost
|
||||
assert backup_cost[model] == main_cost[model]
|
||||
|
||||
|
||||
def test_sonnet_5_ships_standard_rates_not_introductory(local_model_cost_map: None) -> None:
|
||||
sonnet_5: Final = _model_info("databricks/databricks-claude-sonnet-5")
|
||||
sonnet_4_6: Final = _model_info("databricks/databricks-claude-sonnet-4-6")
|
||||
|
||||
for field in PRICE_FIELDS:
|
||||
assert sonnet_5[field] == pytest.approx(sonnet_4_6[field]), field
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ENTRIES_STORING_PROMOTIONAL_RATE)
|
||||
def test_entries_storing_the_promotional_rate_price_below_the_published_table(
|
||||
local_model_cost_map: None,
|
||||
model: str,
|
||||
) -> None:
|
||||
info: Final = _model_info(model)
|
||||
input_dbu, output_dbu, _, _ = PUBLISHED_DBU_PER_MILLION[model]
|
||||
expiry_hint: Final = f"the gemini promotion expires {PROMOTION_EXPIRES}, after which the list rate applies"
|
||||
|
||||
assert info["input_cost_per_token"] == pytest.approx(
|
||||
_dollars_per_token(input_dbu) * PROMOTIONAL_DISCOUNT, rel=2e-4
|
||||
), expiry_hint
|
||||
assert info["output_cost_per_token"] == pytest.approx(
|
||||
_dollars_per_token(output_dbu) * PROMOTIONAL_DISCOUNT, rel=2e-4
|
||||
), expiry_hint
|
||||
assert info["cache_creation_input_token_cost"] == pytest.approx(info["input_cost_per_token"])
|
||||
assert info["cache_read_input_token_cost"] == pytest.approx(0.1 * info["input_cost_per_token"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ENTRIES_STORING_LIST_RATE_DESPITE_PROMOTION)
|
||||
def test_entries_storing_the_list_rate_bill_above_the_promotional_price(
|
||||
local_model_cost_map: None,
|
||||
model: str,
|
||||
) -> None:
|
||||
info: Final = _model_info(model)
|
||||
input_dbu, _, _, _ = PUBLISHED_DBU_PER_MILLION[model]
|
||||
list_rate: Final = _dollars_per_token(input_dbu)
|
||||
|
||||
assert info["input_cost_per_token"] == pytest.approx(list_rate, rel=2e-4), (
|
||||
f"{model} moved off the list rate; if it now stores the discount that runs to "
|
||||
f"{PROMOTION_EXPIRES}, move it into ENTRIES_STORING_PROMOTIONAL_RATE"
|
||||
)
|
||||
assert info["cache_creation_input_token_cost"] == pytest.approx(info["input_cost_per_token"])
|
||||
|
|
@ -473,12 +473,14 @@ def test_transform_messages_helper_strips_thinking_blocks():
|
|||
"thinking_blocks": [
|
||||
{"type": "thinking", "thinking": "internal", "signature": ""}
|
||||
],
|
||||
"reasoning_content": "internal",
|
||||
},
|
||||
]
|
||||
out = config._transform_messages_helper(
|
||||
messages, model="accounts/fireworks/models/glm-5p1", litellm_params={}
|
||||
)
|
||||
assert "thinking_blocks" not in out[1]
|
||||
assert "reasoning_content" not in out[1]
|
||||
assert out[1]["content"] == "I can help."
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -200,6 +200,7 @@ def test_hosted_vllm_thinking_blocks_prepended_to_assistant_content():
|
|||
"signature": "abc123",
|
||||
}
|
||||
],
|
||||
"reasoning_content": "Let me reason about this...",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -218,6 +219,7 @@ def test_hosted_vllm_thinking_blocks_prepended_to_assistant_content():
|
|||
assert isinstance(assistant_msg["content"], str)
|
||||
assert assistant_msg["content"] == "Here is my answer."
|
||||
assert "thinking_blocks" not in assistant_msg
|
||||
assert "reasoning_content" not in assistant_msg
|
||||
|
||||
|
||||
def test_hosted_vllm_thinking_blocks_with_list_content():
|
||||
|
|
|
|||
|
|
@ -245,13 +245,21 @@ def _fake_get_async_httpx_client_factory(captured_calls: list):
|
|||
return _fake_get_async_httpx_client
|
||||
|
||||
|
||||
async def _fake_create_client(base_url, client_config=None, **kwargs):
|
||||
async def _fake_create_client(agent_card, client_config=None, **kwargs):
|
||||
client = MagicMock()
|
||||
if client_config is not None:
|
||||
client._litellm_httpx_client = client_config.httpx_client
|
||||
return client
|
||||
|
||||
|
||||
def _fake_card_resolver(httpx_client, base_url, **kwargs):
|
||||
resolver = MagicMock()
|
||||
card = MagicMock()
|
||||
card.supported_interfaces = ()
|
||||
resolver.get_agent_card = AsyncMock(return_value=card)
|
||||
return resolver
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_a2a_client_leaves_the_shared_client_untouched():
|
||||
"""
|
||||
|
|
@ -276,6 +284,10 @@ async def test_create_a2a_client_leaves_the_shared_client_untouched():
|
|||
"litellm.a2a_protocol.main.create_client",
|
||||
new=AsyncMock(side_effect=_fake_create_client),
|
||||
),
|
||||
patch(
|
||||
"litellm.a2a_protocol.main.A2ACardResolver",
|
||||
side_effect=_fake_card_resolver,
|
||||
),
|
||||
):
|
||||
await create_a2a_client(
|
||||
base_url="http://agent-a:9999",
|
||||
|
|
@ -321,6 +333,10 @@ async def test_create_a2a_client_default_timeout_matches_constant():
|
|||
"litellm.a2a_protocol.main.create_client",
|
||||
new=AsyncMock(side_effect=_fake_create_client),
|
||||
),
|
||||
patch(
|
||||
"litellm.a2a_protocol.main.A2ACardResolver",
|
||||
side_effect=_fake_card_resolver,
|
||||
),
|
||||
):
|
||||
await create_a2a_client(base_url="http://127.0.0.1:9")
|
||||
|
||||
|
|
@ -352,6 +368,10 @@ async def test_create_a2a_client_explicit_timeout_overrides_default():
|
|||
"litellm.a2a_protocol.main.create_client",
|
||||
new=AsyncMock(side_effect=_fake_create_client),
|
||||
),
|
||||
patch(
|
||||
"litellm.a2a_protocol.main.A2ACardResolver",
|
||||
side_effect=_fake_card_resolver,
|
||||
),
|
||||
):
|
||||
await create_a2a_client(base_url="http://127.0.0.1:9", timeout=42.5)
|
||||
|
||||
|
|
|
|||
|
|
@ -18,6 +18,9 @@ from litellm.types.proxy.claude_code_endpoints import (
|
|||
UpdatePluginRequest,
|
||||
)
|
||||
from litellm.proxy.anthropic_endpoints.claude_code_endpoints.claude_code_marketplace import (
|
||||
delete_plugin,
|
||||
disable_plugin,
|
||||
enable_plugin,
|
||||
get_marketplace,
|
||||
register_plugin,
|
||||
update_plugin,
|
||||
|
|
@ -72,6 +75,12 @@ _USER = UserAPIKeyAuth(
|
|||
user_id="test-user",
|
||||
)
|
||||
|
||||
_NON_ADMIN_USER = UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.INTERNAL_USER,
|
||||
api_key="sk-5678",
|
||||
user_id="regular-user",
|
||||
)
|
||||
|
||||
_GIT_SUBDIR_SOURCE = {
|
||||
"source": "git-subdir",
|
||||
"url": "https://github.com/org/monorepo.git",
|
||||
|
|
@ -151,6 +160,7 @@ async def test_update_plugin_replaces_existing_source():
|
|||
response = await update_plugin(
|
||||
plugin_name=name,
|
||||
request=UpdatePluginRequest(source=new_source, version="2.0.0", description="updated"),
|
||||
user_api_key_dict=_USER,
|
||||
)
|
||||
|
||||
assert response.status == "success"
|
||||
|
|
@ -170,6 +180,7 @@ async def test_update_plugin_not_found():
|
|||
await update_plugin(
|
||||
plugin_name="does-not-exist",
|
||||
request=UpdatePluginRequest(source=_GIT_SUBDIR_SOURCE),
|
||||
user_api_key_dict=_USER,
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 404
|
||||
|
|
@ -213,6 +224,7 @@ async def test_update_plugin_db_error_maps_to_structured_500():
|
|||
await update_plugin(
|
||||
plugin_name=name,
|
||||
request=UpdatePluginRequest(source={"source": "github", "repo": "org/replacement"}),
|
||||
user_api_key_dict=_USER,
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 500
|
||||
|
|
@ -341,3 +353,62 @@ async def test_register_plugin_unknown_source_type():
|
|||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "git-subdir" in exc_info.value.detail["error"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_register_plugin_rejects_non_admin():
|
||||
"""A non-admin key cannot add an entry to the marketplace catalog."""
|
||||
request = RegisterPluginRequest(name="attacker-plugin", source=_GIT_SUBDIR_SOURCE)
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await register_plugin(request=request, user_api_key_dict=_NON_ADMIN_USER)
|
||||
|
||||
assert exc_info.value.status_code == 403
|
||||
|
||||
table = litellm.proxy.proxy_server.prisma_client.db.litellm_claudecodeplugintable
|
||||
assert await table.find_unique(where={"name": "attacker-plugin"}) is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_plugin_rejects_non_admin_overwrite():
|
||||
"""A non-admin key cannot overwrite an existing plugin's source."""
|
||||
name = "trusted-plugin"
|
||||
await register_plugin(
|
||||
request=RegisterPluginRequest(name=name, source=_GIT_SUBDIR_SOURCE, version="1.0.0"),
|
||||
user_api_key_dict=_USER,
|
||||
)
|
||||
|
||||
malicious_source = {"source": "github", "repo": "attacker/malicious-repo"}
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await update_plugin(
|
||||
plugin_name=name,
|
||||
request=UpdatePluginRequest(source=malicious_source),
|
||||
user_api_key_dict=_NON_ADMIN_USER,
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 403
|
||||
|
||||
stored = await _read_stored_manifest(name)
|
||||
assert stored["source"] == _GIT_SUBDIR_SOURCE
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_enable_disable_delete_plugin_reject_non_admin():
|
||||
"""Non-admin keys cannot enable, disable, or delete catalog entries."""
|
||||
name = "trusted-plugin-2"
|
||||
await register_plugin(
|
||||
request=RegisterPluginRequest(name=name, source=_GIT_SUBDIR_SOURCE, version="1.0.0"),
|
||||
user_api_key_dict=_USER,
|
||||
)
|
||||
|
||||
for coro in (
|
||||
enable_plugin(plugin_name=name, user_api_key_dict=_NON_ADMIN_USER),
|
||||
disable_plugin(plugin_name=name, user_api_key_dict=_NON_ADMIN_USER),
|
||||
delete_plugin(plugin_name=name, user_api_key_dict=_NON_ADMIN_USER),
|
||||
):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await coro
|
||||
assert exc_info.value.status_code == 403
|
||||
|
||||
table = litellm.proxy.proxy_server.prisma_client.db.litellm_claudecodeplugintable
|
||||
assert (await table.find_unique(where={"name": name})).enabled is True
|
||||
|
|
|
|||
|
|
@ -731,6 +731,318 @@ async def test_track_cost_callback_skips_when_no_standard_logging_object():
|
|||
mock_proxy_logging.failed_tracking_alert.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_track_cost_callback_defers_in_progress_background_interaction(): # test-quality-ok: writing no spend row and raising no alert is the whole observable contract of the deferral path
|
||||
"""
|
||||
A background=true interaction create returns in_progress with no usage
|
||||
block, so its success event has a model but no standard_logging_object.
|
||||
The callback must skip quietly (billing happens later via the background
|
||||
poll task) instead of raising 'Cost tracking failed' and alerting.
|
||||
"""
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logger = _ProxyDBLogger()
|
||||
|
||||
kwargs = {
|
||||
"call_type": "acreate_interaction",
|
||||
"model": "gemini/gemini-3-flash-preview",
|
||||
"litellm_call_id": "test-call-id",
|
||||
"litellm_params": {},
|
||||
"stream": False,
|
||||
}
|
||||
in_progress_response = InteractionsAPIResponse(
|
||||
id="interactions/bg-abc",
|
||||
model="gemini-3-flash-preview",
|
||||
status="in_progress",
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
) as mock_proxy_logging:
|
||||
mock_proxy_logging.failed_tracking_alert = AsyncMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=kwargs,
|
||||
completion_response=in_progress_response,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
mock_proxy_logging.db_spend_update_writer.update_database.assert_not_called()
|
||||
mock_proxy_logging.failed_tracking_alert.assert_not_called()
|
||||
|
||||
|
||||
def _in_progress_interaction_kwargs(reservation: dict) -> dict:
|
||||
return {
|
||||
"call_type": "acreate_interaction",
|
||||
"model": "gemini/gemini-3-flash-preview",
|
||||
"litellm_call_id": "test-call-id",
|
||||
"litellm_params": {"metadata": {"user_api_key_budget_reservation": reservation}},
|
||||
"stream": False,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("status", ["in_progress", "queued"])
|
||||
async def test_track_cost_callback_keeps_reservation_open_for_in_progress_background_interaction(status):
|
||||
"""
|
||||
The pre-call budget reservation must stay open while a background
|
||||
interaction is in flight, so concurrent creates cannot stack past the
|
||||
budget; the poll task's completion event reconciles it to the actual cost.
|
||||
|
||||
``queued`` is in flight for the same reason ``in_progress`` is: it has not
|
||||
reached a terminal status, so releasing its reservation here would drop the
|
||||
estimate off the spend counters while the interaction is still going to run
|
||||
and still going to cost money.
|
||||
"""
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logger = _ProxyDBLogger()
|
||||
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
|
||||
in_progress_response = InteractionsAPIResponse(
|
||||
id="interactions/bg-abc",
|
||||
model="gemini-3-flash-preview",
|
||||
status=status,
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
) as mock_proxy_logging:
|
||||
mock_proxy_logging.failed_tracking_alert = AsyncMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=_in_progress_interaction_kwargs(reservation),
|
||||
completion_response=in_progress_response,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is False
|
||||
mock_proxy_logging.failed_tracking_alert.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_track_cost_callback_releases_reservation_for_in_progress_interaction_when_polling_disabled(
|
||||
monkeypatch,
|
||||
):
|
||||
"""
|
||||
With the poll task kill switch off nothing will ever reconcile the
|
||||
reservation, so the callback must release it or the spend counters stay
|
||||
pinned at the estimated cost forever.
|
||||
"""
|
||||
import litellm.proxy.hooks.proxy_track_cost_callback as callback_module
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
monkeypatch.setattr(callback_module, "BACKGROUND_INTERACTION_COST_POLLING_ENABLED", False)
|
||||
|
||||
logger = _ProxyDBLogger()
|
||||
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
|
||||
in_progress_response = InteractionsAPIResponse(
|
||||
id="interactions/bg-abc",
|
||||
model="gemini-3-flash-preview",
|
||||
status="in_progress",
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
) as mock_proxy_logging:
|
||||
mock_proxy_logging.failed_tracking_alert = AsyncMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=_in_progress_interaction_kwargs(reservation),
|
||||
completion_response=in_progress_response,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
mock_proxy_logging.failed_tracking_alert.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"status",
|
||||
["failed", "cancelled", "incomplete", "budget_exceeded"],
|
||||
)
|
||||
async def test_track_cost_callback_releases_reservation_for_unpollable_interaction(status):
|
||||
"""
|
||||
Only an in-progress create gets a poll task, so a create that comes back
|
||||
terminal with no usage has nobody left to reconcile its reservation. The
|
||||
callback must release it there and then, or the pre-call estimate stays
|
||||
added to the key, user, team and org spend counters and starts refusing
|
||||
traffic against budget that was never actually spent.
|
||||
|
||||
None of these statuses produced output, so their missing usage is a normal
|
||||
outcome rather than a cost-tracking failure, and the callback must not fire
|
||||
``failed_tracking_alert``: doing so would flood operators with false alerts
|
||||
and mask real cost-tracking failures.
|
||||
"""
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logger = _ProxyDBLogger()
|
||||
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
|
||||
terminal_response = InteractionsAPIResponse(
|
||||
id="interactions/bg-abc",
|
||||
model="gemini-3-flash-preview",
|
||||
status=status,
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
) as mock_proxy_logging:
|
||||
mock_proxy_logging.failed_tracking_alert = AsyncMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=_in_progress_interaction_kwargs(reservation),
|
||||
completion_response=terminal_response,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
mock_proxy_logging.failed_tracking_alert.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("status", ["completed", "requires_action"])
|
||||
async def test_track_cost_callback_alerts_when_an_interaction_that_produced_output_has_no_usage(status):
|
||||
"""
|
||||
``completed`` and ``requires_action`` both mean the model produced output,
|
||||
so a usage block is always expected with them. One arriving without it
|
||||
means the charge for real work was lost, which is exactly what the
|
||||
cost-tracking alert is for: silencing it here would let an operator's
|
||||
interactions bill nothing with no signal that anything went wrong.
|
||||
|
||||
The reservation still has to be released, since suppressing the alert was
|
||||
never what freed it.
|
||||
"""
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logger = _ProxyDBLogger()
|
||||
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
|
||||
usageless_response = InteractionsAPIResponse(
|
||||
id="interactions/bg-abc",
|
||||
model="gemini-3-flash-preview",
|
||||
status=status,
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
) as mock_proxy_logging:
|
||||
mock_proxy_logging.failed_tracking_alert = AsyncMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=_in_progress_interaction_kwargs(reservation),
|
||||
completion_response=usageless_response,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
mock_proxy_logging.failed_tracking_alert.assert_called_once()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_track_cost_callback_releases_reservation_for_interaction_without_an_id():
|
||||
"""
|
||||
The scheduler also refuses a response with no id, since it has nothing to
|
||||
poll for, so the callback must not defer to a poll task that will never
|
||||
exist, and it must not fire ``failed_tracking_alert`` for what is a
|
||||
legitimate no-usage response rather than a cost-tracking failure.
|
||||
"""
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logger = _ProxyDBLogger()
|
||||
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
|
||||
idless_response = InteractionsAPIResponse(
|
||||
id="",
|
||||
model="gemini-3-flash-preview",
|
||||
status="in_progress",
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
) as mock_proxy_logging:
|
||||
mock_proxy_logging.failed_tracking_alert = AsyncMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=_in_progress_interaction_kwargs(reservation),
|
||||
completion_response=idless_response,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
mock_proxy_logging.failed_tracking_alert.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_callback_handles_every_status_the_interactions_api_can_return():
|
||||
"""
|
||||
Whatever status a usage-less create comes back with, exactly one of two
|
||||
things has to happen to its budget reservation: the callback holds it open
|
||||
for a poll task that will settle it, or it releases it on the spot. A
|
||||
status that falls through both leaves the pre-call estimate pinned to the
|
||||
key, user, team and org spend counters forever, refusing traffic against
|
||||
budget nobody spent.
|
||||
|
||||
Driven off the generated spec enum so a status Google adds later fails here
|
||||
instead of quietly leaking reservations in production.
|
||||
"""
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
from litellm.types.interactions.generated import Status1
|
||||
|
||||
deferred = set()
|
||||
released = set()
|
||||
|
||||
for status in sorted(member.value for member in Status1):
|
||||
logger = _ProxyDBLogger()
|
||||
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
|
||||
response = InteractionsAPIResponse(
|
||||
id="interactions/bg-abc",
|
||||
model="gemini-3-flash-preview",
|
||||
status=status,
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
) as mock_proxy_logging:
|
||||
mock_proxy_logging.failed_tracking_alert = AsyncMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=_in_progress_interaction_kwargs(reservation),
|
||||
completion_response=response,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
(deferred if reservation["finalized"] is False else released).add(status)
|
||||
|
||||
assert deferred == {"in_progress", "queued"}
|
||||
assert released == {
|
||||
"completed",
|
||||
"requires_action",
|
||||
"failed",
|
||||
"cancelled",
|
||||
"incomplete",
|
||||
"budget_exceeded",
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_post_call_failure_hook_propagates_trace_id_from_logging_obj():
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ import litellm
|
|||
from litellm.litellm_core_utils.health_check_helpers import HealthCheckHelpers
|
||||
from litellm.proxy import health_check as hc_module
|
||||
from litellm.proxy.health_check import (
|
||||
_is_semantic_auto_router_deployment,
|
||||
_is_strategy_router_deployment,
|
||||
_resolve_health_check_max_tokens,
|
||||
_resolve_health_check_mode,
|
||||
_update_litellm_params_for_health_check,
|
||||
|
|
@ -499,33 +499,22 @@ def test_autodetected_embedding_skips_reasoning_effort():
|
|||
assert "max_tokens" not in updated
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# auto_router (semantic router) deployments must be skipped by health checks.
|
||||
#
|
||||
# These are meta-routers that select among real LLM deployments at request
|
||||
# time. They have no LLM endpoint to probe. Before this fix, the health check
|
||||
# passed model="auto_router/router_1" to get_llm_provider(), which raised
|
||||
# BadRequestError: "Unmapped LLM provider for this endpoint" because
|
||||
# auto_router is not a real LLM provider.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, expected",
|
||||
[
|
||||
("auto_router/router_1", True),
|
||||
("auto_router/my_router", True),
|
||||
("auto_router/complexity_router", False),
|
||||
("auto_router/adaptive_router", False),
|
||||
("auto_router/quality_router", False),
|
||||
("auto_router/adaptive_router/subpath", False),
|
||||
("auto_router/complexity_router", True),
|
||||
("auto_router/adaptive_router", True),
|
||||
("auto_router/quality_router", True),
|
||||
("auto_router/adaptive_router/subpath", True),
|
||||
("gpt-4", False),
|
||||
("openai/gpt-4", False),
|
||||
("bedrock/claude", False),
|
||||
],
|
||||
)
|
||||
def test_is_semantic_auto_router_deployment(model, expected):
|
||||
assert _is_semantic_auto_router_deployment({"model": model}) == expected
|
||||
def test_is_strategy_router_deployment(model, expected):
|
||||
assert _is_strategy_router_deployment({"model": model}) == expected
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -676,3 +665,22 @@ async def test_bedrock_invoke_body_has_no_media_source_without_health_check_para
|
|||
body = await _pegasus_health_check_request_body({"mode": "chat"}, monkeypatch)
|
||||
|
||||
assert "mediaSource" not in body
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_run_model_health_check_skips_complexity_router_deployment():
|
||||
fake_ahealth_check = AsyncMock(return_value={})
|
||||
model = {
|
||||
"litellm_params": {
|
||||
"model": "auto_router/complexity_router",
|
||||
"complexity_router_config": {"tiers": {"simple": "gpt-4o-mini"}},
|
||||
"complexity_router_default_model": "gpt-4o-mini",
|
||||
},
|
||||
"model_info": {},
|
||||
}
|
||||
|
||||
with patch.object(hc_module.litellm, "ahealth_check", fake_ahealth_check):
|
||||
result = await hc_module._run_model_health_check(model)
|
||||
|
||||
fake_ahealth_check.assert_not_called()
|
||||
assert result == {}
|
||||
|
|
|
|||
|
|
@ -114,3 +114,46 @@ def test_assistant_message_after_tool_call_is_folded_into_it():
|
|||
tool_call_idx = next(i for i, m in enumerate(msgs) if isinstance(m, dict) and m.get("tool_calls"))
|
||||
assert msgs[tool_call_idx].get("role") == "assistant"
|
||||
assert msgs[tool_call_idx + 1].get("role") == "tool"
|
||||
|
||||
|
||||
def test_assistant_message_before_function_call_keeps_one_assistant_turn():
|
||||
"""The chat->responses bridge emits an assistant message ahead of its function_call.
|
||||
|
||||
Round-tripping that order back to chat must fold both into a single assistant
|
||||
turn, so the tool result still follows the message that made the call.
|
||||
"""
|
||||
msgs = LiteLLMCompletionResponsesConfig._transform_response_input_param_to_chat_completion_message(
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"type": "message",
|
||||
"content": [{"type": "input_text", "text": "What is the weather?"}],
|
||||
},
|
||||
{
|
||||
"role": "assistant",
|
||||
"type": "message",
|
||||
"content": [{"type": "output_text", "text": "Let me check."}],
|
||||
},
|
||||
{
|
||||
"type": "function_call",
|
||||
"name": "get_weather",
|
||||
"call_id": "call_1",
|
||||
"arguments": "{}",
|
||||
},
|
||||
{
|
||||
"type": "function_call_output",
|
||||
"call_id": "call_1",
|
||||
"output": "sunny",
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
assistant_msgs = [m for m in msgs if isinstance(m, dict) and m.get("role") == "assistant"]
|
||||
assert len(assistant_msgs) == 1
|
||||
assistant = assistant_msgs[0]
|
||||
assert assistant["content"] == [{"type": "text", "text": "Let me check."}]
|
||||
assert [tc["function"]["name"] for tc in assistant["tool_calls"]] == ["get_weather"]
|
||||
|
||||
assistant_idx = msgs.index(assistant)
|
||||
assert msgs[assistant_idx + 1].get("role") == "tool"
|
||||
assert msgs[assistant_idx + 1].get("tool_call_id") == "call_1"
|
||||
|
|
|
|||
|
|
@ -3584,6 +3584,104 @@ def test_batch_cost_calculator_cache_creation_falls_back_to_input_rate():
|
|||
assert prompt_cost == pytest.approx((1000 * 3e-6 + 8000 * 3e-7 + 2000 * 3e-6) / 2)
|
||||
|
||||
|
||||
def test_completion_cost_bills_interactions_api_response():
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
model_info = litellm.get_model_info(model="gemini-2.5-flash", custom_llm_provider="gemini")
|
||||
response = InteractionsAPIResponse(
|
||||
id="interactions/abc123",
|
||||
model="gemini-2.5-flash",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage={
|
||||
"total_tokens": 175,
|
||||
"total_input_tokens": 100,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 100}],
|
||||
"total_cached_tokens": 0,
|
||||
"total_output_tokens": 50,
|
||||
"output_tokens_by_modality": [{"modality": "text", "tokens": 50}],
|
||||
"total_tool_use_tokens": 0,
|
||||
"total_thought_tokens": 25,
|
||||
},
|
||||
)
|
||||
|
||||
cost = completion_cost(completion_response=response, custom_llm_provider="gemini")
|
||||
|
||||
reasoning_rate = model_info.get("output_cost_per_reasoning_token") or model_info["output_cost_per_token"]
|
||||
expected = (
|
||||
100 * model_info["input_cost_per_token"]
|
||||
+ 50 * model_info["output_cost_per_token"]
|
||||
+ 25 * reasoning_rate
|
||||
)
|
||||
assert cost == pytest.approx(expected)
|
||||
assert cost > 0
|
||||
|
||||
|
||||
def test_completion_cost_bills_interactions_google_search_per_query():
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
model_info = litellm.get_model_info(model="gemini-3-flash-preview", custom_llm_provider="gemini")
|
||||
response = InteractionsAPIResponse(
|
||||
id="interactions/search123",
|
||||
model="gemini-3-flash-preview",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage={
|
||||
"total_tokens": 680,
|
||||
"total_input_tokens": 103,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 103}],
|
||||
"total_cached_tokens": 0,
|
||||
"total_output_tokens": 226,
|
||||
"total_tool_use_tokens": 0,
|
||||
"total_thought_tokens": 351,
|
||||
"grounding_tool_count": [{"type": "google_search", "count": 3}],
|
||||
},
|
||||
)
|
||||
|
||||
cost = completion_cost(completion_response=response, custom_llm_provider="gemini")
|
||||
|
||||
per_query_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"]
|
||||
reasoning_rate = model_info.get("output_cost_per_reasoning_token") or model_info["output_cost_per_token"]
|
||||
expected = (
|
||||
103 * model_info["input_cost_per_token"]
|
||||
+ 226 * model_info["output_cost_per_token"]
|
||||
+ 351 * reasoning_rate
|
||||
+ 3 * per_query_cost
|
||||
)
|
||||
assert model_info.get("web_search_billing_unit") == "per_query"
|
||||
assert cost == pytest.approx(expected)
|
||||
assert cost > 3 * per_query_cost
|
||||
|
||||
|
||||
def test_completion_cost_bills_interactions_video_output_at_video_rate():
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
model_info = litellm.get_model_info(model="gemini-omni-flash-preview", custom_llm_provider="gemini")
|
||||
video_tokens = 5792 * 8
|
||||
response = InteractionsAPIResponse(
|
||||
id="interactions/video123",
|
||||
model="gemini-omni-flash-preview",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage={
|
||||
"total_tokens": 10 + video_tokens,
|
||||
"total_input_tokens": 10,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 10}],
|
||||
"total_cached_tokens": 0,
|
||||
"total_output_tokens": video_tokens,
|
||||
"output_tokens_by_modality": [{"modality": "video", "tokens": video_tokens}],
|
||||
"total_tool_use_tokens": 0,
|
||||
"total_thought_tokens": 0,
|
||||
},
|
||||
)
|
||||
|
||||
cost = completion_cost(completion_response=response, custom_llm_provider="gemini")
|
||||
|
||||
expected = 10 * model_info["input_cost_per_token"] + video_tokens * model_info["output_cost_per_video_token"]
|
||||
assert model_info["output_cost_per_video_token"] != model_info["output_cost_per_token"]
|
||||
assert cost == pytest.approx(expected)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"batch_rate,expected_prompt,expected_completion",
|
||||
[
|
||||
|
|
|
|||
|
|
@ -165,6 +165,20 @@ describe("OrganizationsTable", () => {
|
|||
expect(screen.getByText("RPM: Unlimited")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("renders a tpm/rpm limit of 0 as 0, never as Unlimited", () => {
|
||||
render(
|
||||
<OrganizationsTable
|
||||
{...baseProps}
|
||||
organizations={[makeOrganization({ litellm_budget_table: { max_budget: null, tpm_limit: 0, rpm_limit: 0 } })]}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(screen.getByText("TPM: 0")).toBeInTheDocument();
|
||||
expect(screen.getByText("RPM: 0")).toBeInTheDocument();
|
||||
expect(screen.queryByText("TPM: Unlimited")).not.toBeInTheDocument();
|
||||
expect(screen.queryByText("RPM: Unlimited")).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("renders loading skeletons instead of rows while loading", () => {
|
||||
render(
|
||||
<OrganizationsTable
|
||||
|
|
|
|||
|
|
@ -28,8 +28,8 @@ function OrganizationLimitsCell({ organization }: { organization: Organization }
|
|||
const { tpm_limit, rpm_limit } = getOrganizationBudget(organization);
|
||||
return (
|
||||
<div className="flex flex-col text-xs text-muted-foreground">
|
||||
<span>TPM: {tpm_limit ? tpm_limit : "Unlimited"}</span>
|
||||
<span>RPM: {rpm_limit ? rpm_limit : "Unlimited"}</span>
|
||||
<span>TPM: {tpm_limit ?? "Unlimited"}</span>
|
||||
<span>RPM: {rpm_limit ?? "Unlimited"}</span>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -87,6 +87,21 @@ describe("ChatMessageBubble", () => {
|
|||
expect(screen.getByText("Hi there")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it.each([
|
||||
{ role: "user" as const, bubble: ["bg-info/10", "border-info/20"], avatar: "bg-info/20" },
|
||||
{ role: "assistant" as const, bubble: ["bg-card", "border-border"], avatar: "bg-muted" },
|
||||
])("should paint the $role surface from theme tokens, not fixed colours", ({ role, bubble, avatar }) => {
|
||||
render(<ChatMessageBubble {...defaultProps} message={{ role, content: "Hello" }} />);
|
||||
|
||||
const header = screen.getByText(role).closest("div") as HTMLElement;
|
||||
const surface = header.parentElement as HTMLElement;
|
||||
|
||||
expect(surface).toHaveClass(...bubble);
|
||||
expect(surface).not.toHaveAttribute("style");
|
||||
expect(header.firstElementChild).toHaveClass(avatar);
|
||||
expect(header.firstElementChild).not.toHaveAttribute("style");
|
||||
});
|
||||
|
||||
it("should show model badge for assistant messages when model is provided", () => {
|
||||
render(<ChatMessageBubble {...defaultProps} message={{ role: "assistant", content: "Reply", model: "gpt-4" }} />);
|
||||
|
||||
|
|
|
|||
|
|
@ -46,20 +46,16 @@ function ChatMessageBubble({
|
|||
return (
|
||||
<div className={`mb-4 min-w-0 ${isUser ? "text-right" : "text-left"}`}>
|
||||
<div
|
||||
className="inline-block min-w-0 max-w-[92%] overflow-hidden rounded-lg p-3 shadow-xs sm:max-w-[85%] sm:px-4"
|
||||
style={{
|
||||
backgroundColor: isUser ? "#f0f8ff" : "#ffffff",
|
||||
border: isUser ? "1px solid #e6f0fa" : "1px solid #f0f0f0",
|
||||
textAlign: "left",
|
||||
}}
|
||||
className={`inline-block min-w-0 max-w-[92%] overflow-hidden rounded-lg border p-3 text-left text-card-foreground shadow-xs sm:max-w-[85%] sm:px-4 ${
|
||||
isUser ? "border-info/20 bg-info/10" : "border-border bg-card"
|
||||
}`}
|
||||
>
|
||||
{/* Header: role icon + name + model badge */}
|
||||
<div className="mb-1.5 flex min-w-0 items-center gap-2">
|
||||
<div
|
||||
className="flex items-center justify-center w-6 h-6 rounded-full mr-1"
|
||||
style={{
|
||||
backgroundColor: isUser ? "#e6f0fa" : "#f5f5f5",
|
||||
}}
|
||||
className={`flex items-center justify-center w-6 h-6 rounded-full mr-1 ${
|
||||
isUser ? "bg-info/20" : "bg-muted"
|
||||
}`}
|
||||
>
|
||||
{isUser ? (
|
||||
<User className="size-3 text-info" aria-hidden="true" />
|
||||
|
|
|
|||
|
|
@ -1784,19 +1784,9 @@ const ChatUI: React.FC<ChatUIProps> = ({
|
|||
chatHistory.length > 0 &&
|
||||
chatHistory[chatHistory.length - 1].role === "user" && (
|
||||
<div className="mb-4 text-left">
|
||||
<div
|
||||
className="inline-block max-w-[80%] rounded-lg p-3.5 px-4 shadow-xs"
|
||||
style={{
|
||||
backgroundColor: "#ffffff",
|
||||
border: "1px solid #f0f0f0",
|
||||
textAlign: "left",
|
||||
}}
|
||||
>
|
||||
<div className="inline-block max-w-[80%] rounded-lg border border-border bg-card p-3.5 px-4 text-left text-card-foreground shadow-xs">
|
||||
<div className="mb-1.5 flex items-center gap-2">
|
||||
<div
|
||||
className="mr-1 flex h-6 w-6 items-center justify-center rounded-full"
|
||||
style={{ backgroundColor: "#f5f5f5" }}
|
||||
>
|
||||
<div className="mr-1 flex h-6 w-6 items-center justify-center rounded-full bg-muted">
|
||||
<Bot className="size-3 text-muted-foreground" aria-hidden="true" />
|
||||
</div>
|
||||
<strong className="text-sm capitalize">Assistant</strong>
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ import { KeyResponse, Team } from "../key_team_helpers/key_list";
|
|||
import { useKeyInfo } from "@/app/(dashboard)/hooks/keys/useKeyInfo";
|
||||
import { KeysResponse, useKeys } from "@/app/(dashboard)/hooks/keys/useKeys";
|
||||
import useTeams from "@/app/(dashboard)/hooks/useTeams";
|
||||
import { regenerateKeyCall } from "../networking";
|
||||
|
||||
// Resolve debounced values synchronously so an applied filter lands in the useKeys query within the test tick.
|
||||
vi.mock("@tanstack/react-pacer/debouncer", async () => {
|
||||
|
|
@ -25,6 +26,11 @@ vi.mock("@tanstack/react-pacer/debouncer", async () => {
|
|||
|
||||
vi.mock("next/navigation", () => ({ useRouter: () => ({ push: vi.fn() }) }));
|
||||
|
||||
vi.mock("../networking", async (importOriginal) => ({
|
||||
...(await importOriginal<typeof import("../networking")>()),
|
||||
regenerateKeyCall: vi.fn(),
|
||||
}));
|
||||
|
||||
vi.mock("@/app/(dashboard)/hooks/useAuthorized", () => ({
|
||||
default: vi.fn(() => ({
|
||||
accessToken: "test-token",
|
||||
|
|
@ -389,6 +395,28 @@ it("renders KeyInfoView when the URL has ?key= for a key on the current page, wi
|
|||
expect(screen.getByTestId("pagination-range")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("repoints ?key= to the rotated hash once the regenerate dialog is dismissed", async () => {
|
||||
const user = userEvent.setup();
|
||||
vi.mocked(regenerateKeyCall).mockResolvedValue({
|
||||
key: "sk-rotated-plaintext",
|
||||
token: null,
|
||||
token_id: "rotated-hash-456",
|
||||
});
|
||||
const onUrlUpdate = vi.fn<OnUrlUpdateFunction>();
|
||||
renderWithProviders(<VirtualKeysTable />, { searchParams: { key: mockKey.token }, onUrlUpdate });
|
||||
|
||||
await user.click(await screen.findByRole("button", { name: /regenerate key/i }));
|
||||
await user.click(await screen.findByRole("button", { name: /^Regenerate$/ }));
|
||||
expect(await screen.findAllByText("sk-rotated-plaintext")).not.toHaveLength(0);
|
||||
expect(lastKeyParam(onUrlUpdate)).toBeUndefined();
|
||||
|
||||
await user.click(screen.getAllByRole("button", { name: "Close" })[0]);
|
||||
|
||||
await waitFor(() => {
|
||||
expect(lastKeyParam(onUrlUpdate)).toBe("rotated-hash-456");
|
||||
});
|
||||
});
|
||||
|
||||
it("fetches the key by id when the URL has ?key= for a key not in the loaded page", async () => {
|
||||
mockUseKeyInfo.mockReturnValue(
|
||||
keyInfoResult({ ...mockKey, token: "other-key-hash", key_alias: "Fetched Key Alias" }),
|
||||
|
|
|
|||
|
|
@ -20,7 +20,7 @@ import { KeyRound } from "lucide-react";
|
|||
import { parseAsString, useQueryState } from "nuqs";
|
||||
import React, { useCallback, useMemo, useState } from "react";
|
||||
|
||||
import { Team } from "../key_team_helpers/key_list";
|
||||
import { KeyResponse, Team } from "../key_team_helpers/key_list";
|
||||
import KeyInfoView from "../templates/key_info_view";
|
||||
import { getKeyTableColumns, KEY_TABLE_HIDDEN_COLUMNS } from "./keyTableColumns";
|
||||
|
||||
|
|
@ -139,6 +139,16 @@ export function VirtualKeysTable({ headerActions }: VirtualKeysTableProps) {
|
|||
[organizations],
|
||||
);
|
||||
|
||||
const handleSelectedKeyDataUpdate = useCallback(
|
||||
(updated: Partial<KeyResponse>) => {
|
||||
const rotatedToken = updated.token ?? updated.token_id;
|
||||
if (!rotatedToken || rotatedToken === selectedKeyId) return;
|
||||
void setSelectedKeyId(rotatedToken);
|
||||
void refetch();
|
||||
},
|
||||
[refetch, selectedKeyId, setSelectedKeyId],
|
||||
);
|
||||
|
||||
const formatFilterValue = useCallback(
|
||||
(columnId: string, value: unknown): string => {
|
||||
const raw = String(value);
|
||||
|
|
@ -165,6 +175,7 @@ export function VirtualKeysTable({ headerActions }: VirtualKeysTableProps) {
|
|||
keyData={selectedKey}
|
||||
teams={allTeams}
|
||||
onDelete={refetch}
|
||||
onKeyDataUpdate={handleSelectedKeyDataUpdate}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
|
|
|
|||
|
|
@ -34,6 +34,8 @@ const modelMappingsRule = {
|
|||
},
|
||||
};
|
||||
|
||||
const tooltipCodeClassName = "rounded-sm bg-background/20 px-1 py-0.5 font-mono text-xs";
|
||||
|
||||
const ConditionalPublicModelName: React.FC = () => {
|
||||
const form = useFormContext<MountedFormValues>();
|
||||
|
||||
|
|
@ -123,22 +125,22 @@ const ConditionalPublicModelName: React.FC = () => {
|
|||
if (!showPublicModelName) return null;
|
||||
|
||||
const publicNameTooltipContent = (
|
||||
<>
|
||||
<div className="mb-2 font-normal">The name you specify in your API calls to LiteLLM Proxy</div>
|
||||
<div className="mb-2 font-normal">
|
||||
<div className="flex flex-col gap-2 text-left font-normal">
|
||||
<div>The name you specify in your API calls to LiteLLM Proxy</div>
|
||||
<div>
|
||||
<strong>Example:</strong> If you name your public model{" "}
|
||||
<code className="bg-muted px-1 py-0.5 rounded-sm text-xs">example-name</code>, and choose{" "}
|
||||
<code className="bg-muted px-1 py-0.5 rounded-sm text-xs">openai/qwen-plus-latest</code> as the LiteLLM model
|
||||
<code className={tooltipCodeClassName}>example-name</code>, and choose{" "}
|
||||
<code className={tooltipCodeClassName}>openai/qwen-plus-latest</code> as the LiteLLM model
|
||||
</div>
|
||||
<div className="mb-2 font-normal">
|
||||
<div>
|
||||
<strong>Usage:</strong> You make an API call to the LiteLLM proxy with{" "}
|
||||
<code className="bg-muted px-1 py-0.5 rounded-sm text-xs">model = "example-name"</code>
|
||||
<code className={tooltipCodeClassName}>model = "example-name"</code>
|
||||
</div>
|
||||
<div className="font-normal">
|
||||
<strong>Result:</strong> LiteLLM sends{" "}
|
||||
<code className="bg-muted px-1 py-0.5 rounded-sm text-xs">qwen-plus-latest</code> to the provider
|
||||
<div>
|
||||
<strong>Result:</strong> LiteLLM sends <code className={tooltipCodeClassName}>qwen-plus-latest</code> to the
|
||||
provider
|
||||
</div>
|
||||
</>
|
||||
</div>
|
||||
);
|
||||
|
||||
const liteLLMModelTooltipContent = <div>The model name LiteLLM will send to the LLM API</div>;
|
||||
|
|
|
|||
|
|
@ -19,13 +19,13 @@ import {
|
|||
isValidSubPath,
|
||||
buildMarketplaceSettingsSnippet,
|
||||
} from "./helpers";
|
||||
import { MarketplacePluginEntry, PluginSource } from "./types";
|
||||
import { MarketplacePluginEntry } from "./types";
|
||||
|
||||
describe("buildMarketplaceSettingsSnippet", () => {
|
||||
it("nests the url under a source object so Claude Code accepts the marketplace", () => {
|
||||
expect(JSON.parse(buildMarketplaceSettingsSnippet("https://proxy.example.com"))).toEqual({
|
||||
extraKnownMarketplaces: {
|
||||
"my-org": {
|
||||
litellm: {
|
||||
source: {
|
||||
source: "url",
|
||||
url: "https://proxy.example.com/claude-code/marketplace.json",
|
||||
|
|
@ -37,28 +37,12 @@ describe("buildMarketplaceSettingsSnippet", () => {
|
|||
});
|
||||
|
||||
describe("formatInstallCommand", () => {
|
||||
it("formats github source with repo", () => {
|
||||
const source: PluginSource = { source: "github", repo: "org/repo" };
|
||||
expect(formatInstallCommand({ name: "my-plugin", source })).toBe("/plugin marketplace add org/repo");
|
||||
it("produces a /plugin install command scoped to the litellm marketplace", () => {
|
||||
expect(formatInstallCommand({ name: "my-plugin" })).toBe("/plugin install my-plugin@litellm");
|
||||
});
|
||||
|
||||
it("formats url source", () => {
|
||||
const source: PluginSource = { source: "url", url: "https://example.com/plugin" };
|
||||
expect(formatInstallCommand({ name: "my-plugin", source })).toBe(
|
||||
"/plugin marketplace add https://example.com/plugin",
|
||||
);
|
||||
});
|
||||
|
||||
it("formats git-subdir source using its url", () => {
|
||||
const source: PluginSource = { source: "git-subdir", url: "https://github.com/org/repo", path: "plugins/x" };
|
||||
expect(formatInstallCommand({ name: "my-plugin", source })).toBe(
|
||||
"/plugin marketplace add https://github.com/org/repo",
|
||||
);
|
||||
});
|
||||
|
||||
it("falls back to plugin name when no repo or url", () => {
|
||||
const source: PluginSource = { source: "github" };
|
||||
expect(formatInstallCommand({ name: "my-plugin", source })).toBe("/plugin marketplace add my-plugin");
|
||||
it("uses the plugin name as the identifier", () => {
|
||||
expect(formatInstallCommand({ name: "code-review" })).toBe("/plugin install code-review@litellm");
|
||||
});
|
||||
});
|
||||
|
||||
|
|
|
|||
|
|
@ -179,13 +179,14 @@ export const parseSkillSource = (rawUrl: string, subPath?: string): SkillSourceP
|
|||
/**
|
||||
* Build the `~/.claude/settings.json` snippet that registers the proxy as a marketplace.
|
||||
* Claude Code expects `extraKnownMarketplaces.<name>.source` to be a source object, not a
|
||||
* bare `"url"` string, so the url/source pair is nested one level deeper.
|
||||
* bare `"url"` string, so the url/source pair is nested one level deeper. The key must be
|
||||
* "litellm" to match the name the proxy returns in marketplace.json.
|
||||
*/
|
||||
export const buildMarketplaceSettingsSnippet = (proxyOrigin: string): string =>
|
||||
JSON.stringify(
|
||||
{
|
||||
extraKnownMarketplaces: {
|
||||
"my-org": {
|
||||
litellm: {
|
||||
source: {
|
||||
source: "url",
|
||||
url: `${proxyOrigin}/claude-code/marketplace.json`,
|
||||
|
|
@ -198,20 +199,10 @@ export const buildMarketplaceSettingsSnippet = (proxyOrigin: string): string =>
|
|||
);
|
||||
|
||||
/**
|
||||
* Generate install command for Claude Code CLI
|
||||
* Format: /plugin marketplace add org/repo OR /plugin marketplace add url
|
||||
* Generate install command for Claude Code CLI.
|
||||
* Installs the named plugin from the "litellm" marketplace registered in settings.json.
|
||||
*/
|
||||
export const formatInstallCommand = (plugin: { name: string; source: PluginSource }): string => {
|
||||
const { source } = plugin;
|
||||
if (source.source === "github" && source.repo) {
|
||||
return `/plugin marketplace add ${source.repo}`;
|
||||
}
|
||||
if ((source.source === "url" || source.source === "git-subdir") && source.url) {
|
||||
return `/plugin marketplace add ${source.url}`;
|
||||
}
|
||||
// Fallback to plugin name
|
||||
return `/plugin marketplace add ${plugin.name}`;
|
||||
};
|
||||
export const formatInstallCommand = (plugin: { name: string }): string => `/plugin install ${plugin.name}@litellm`;
|
||||
|
||||
/**
|
||||
* Extract unique categories from plugins list
|
||||
|
|
|
|||
|
|
@ -261,6 +261,32 @@ const SkillDetail: React.FC<SkillDetailProps> = ({ skill, onBack }) => {
|
|||
</pre>
|
||||
</div>
|
||||
|
||||
{/* Shown when the marketplace catalog is stale and the plugin isn't found yet */}
|
||||
<div
|
||||
style={{
|
||||
border: "1px solid #fce8b2",
|
||||
borderRadius: 8,
|
||||
padding: "12px 16px",
|
||||
backgroundColor: "#fefce8",
|
||||
marginBottom: 16,
|
||||
}}
|
||||
>
|
||||
<p style={{ fontSize: 13, color: "#5f6368", lineHeight: 1.6, margin: "0 0 8px 0" }}>
|
||||
If you see "Plugin {skill.name} not found in marketplace", update the catalog first:
|
||||
</p>
|
||||
<pre
|
||||
style={{
|
||||
margin: 0,
|
||||
fontSize: 13,
|
||||
fontFamily: "monospace",
|
||||
color: "#202124",
|
||||
backgroundColor: "transparent",
|
||||
}}
|
||||
>
|
||||
/plugin marketplace update litellm
|
||||
</pre>
|
||||
</div>
|
||||
|
||||
<p style={{ fontSize: 13, color: "#5f6368", lineHeight: 1.6, margin: 0 }}>
|
||||
Don't have the marketplace configured yet?{" "}
|
||||
<span onClick={() => setActiveTab("setup")} style={{ color: "#1a73e8", cursor: "pointer" }}>
|
||||
|
|
@ -276,12 +302,73 @@ const SkillDetail: React.FC<SkillDetailProps> = ({ skill, onBack }) => {
|
|||
<h2 style={{ fontSize: 18, fontWeight: 400, color: "#202124", margin: "0 0 8px 0" }}>
|
||||
One-time marketplace setup
|
||||
</h2>
|
||||
<p style={{ fontSize: 14, color: "#5f6368", margin: "0 0 24px 0", lineHeight: 1.6 }}>
|
||||
Add this to{" "}
|
||||
|
||||
{/* Option 1: single command — fastest path for most users */}
|
||||
<p style={{ fontSize: 14, color: "#5f6368", margin: "0 0 12px 0", lineHeight: 1.6 }}>
|
||||
Run this command in Claude Code to register the marketplace:
|
||||
</p>
|
||||
<div
|
||||
style={{
|
||||
border: "1px solid #dadce0",
|
||||
borderRadius: 8,
|
||||
overflow: "hidden",
|
||||
marginBottom: 24,
|
||||
}}
|
||||
>
|
||||
<div
|
||||
style={{
|
||||
display: "flex",
|
||||
alignItems: "center",
|
||||
justifyContent: "space-between",
|
||||
padding: "10px 16px",
|
||||
backgroundColor: "#f8f9fa",
|
||||
borderBottom: "1px solid #dadce0",
|
||||
}}
|
||||
>
|
||||
<span style={{ fontSize: 13, color: "#3c4043", fontWeight: 500 }}>Run in Claude Code</span>
|
||||
<button
|
||||
onClick={() => {
|
||||
const origin = typeof window !== "undefined" ? window.location.origin : "";
|
||||
copyToClipboard(`/plugin marketplace add ${origin}/claude-code/marketplace.json`, "marketplace-cmd");
|
||||
}}
|
||||
style={{
|
||||
display: "flex",
|
||||
alignItems: "center",
|
||||
gap: 4,
|
||||
fontSize: 12,
|
||||
color: copiedKey === "marketplace-cmd" ? "#137333" : "#1a73e8",
|
||||
background: "none",
|
||||
border: "none",
|
||||
cursor: "pointer",
|
||||
padding: 0,
|
||||
}}
|
||||
>
|
||||
{copiedKey === "marketplace-cmd" ? <CheckOutlined /> : <CopyOutlined />}
|
||||
{copiedKey === "marketplace-cmd" ? "Copied" : "Copy"}
|
||||
</button>
|
||||
</div>
|
||||
<pre
|
||||
style={{
|
||||
margin: 0,
|
||||
padding: "14px 16px",
|
||||
fontSize: 13,
|
||||
fontFamily: "monospace",
|
||||
color: "#202124",
|
||||
backgroundColor: "#fff",
|
||||
}}
|
||||
>
|
||||
{`/plugin marketplace add ${typeof window !== "undefined" ? window.location.origin : "<proxy-url>"}/claude-code/marketplace.json`}
|
||||
</pre>
|
||||
</div>
|
||||
|
||||
{/* Option 2: settings.json — for persistent config or managed deployments.
|
||||
extraKnownMarketplaces requires source to be a nested object, not a flat string. */}
|
||||
<p style={{ fontSize: 14, color: "#5f6368", margin: "0 0 12px 0", lineHeight: 1.6 }}>
|
||||
Or add this to{" "}
|
||||
<code style={{ fontSize: 13, backgroundColor: "#f1f3f4", padding: "1px 6px", borderRadius: 4 }}>
|
||||
~/.claude/settings.json
|
||||
</code>{" "}
|
||||
to point Claude Code at your proxy:
|
||||
for a persistent configuration:
|
||||
</p>
|
||||
<div
|
||||
style={{
|
||||
|
|
|
|||
|
|
@ -427,4 +427,21 @@ describe("RegenerateKeyModal", () => {
|
|||
);
|
||||
});
|
||||
});
|
||||
|
||||
it("should report the rotated hash from token_id when the API leaves token null", async () => {
|
||||
const user = userEvent.setup();
|
||||
mockRegenerateKeyCall.mockResolvedValue({
|
||||
key: "sk-new-regenerated-key",
|
||||
token: null,
|
||||
token_id: "rotated-hash-456",
|
||||
});
|
||||
|
||||
renderWithProviders(<RegenerateKeyModal {...defaultProps} />);
|
||||
await user.click(screen.getByRole("button", { name: /Regenerate/ }));
|
||||
|
||||
await waitFor(() => {
|
||||
expect(mockOnKeyUpdate).toHaveBeenCalledOnce();
|
||||
});
|
||||
expect(mockOnKeyUpdate.mock.calls[0][0].token).toBe("rotated-hash-456");
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -107,7 +107,7 @@ export function RegenerateKeyModal({ selectedToken, visible, onClose, onKeyUpdat
|
|||
// formatted preview, otherwise downstream expiry parsing breaks.
|
||||
const updatedKeyData: Partial<KeyResponse> = {
|
||||
...response,
|
||||
token: response.token || response.key_id || selectedToken.token,
|
||||
token: response.token_id || response.token || selectedToken.token,
|
||||
key_name: response.key,
|
||||
max_budget: formValues.max_budget,
|
||||
tpm_limit: formValues.tpm_limit,
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
import React from "react";
|
||||
import { fireEvent, screen, waitFor } from "@testing-library/react";
|
||||
import { fireEvent, screen, waitFor, within } from "@testing-library/react";
|
||||
import userEvent from "@testing-library/user-event";
|
||||
import { vi, test, expect, beforeEach } from "vitest";
|
||||
import { renderWithProviders } from "../../../tests/test-utils";
|
||||
|
|
@ -290,3 +290,37 @@ test("should keep unsaved settings edits when switching tabs and back", async ()
|
|||
|
||||
expect(screen.getByLabelText(/Organization Name/i)).toHaveValue("Renamed Org");
|
||||
});
|
||||
|
||||
test("renders a tpm/rpm limit of 0 as 0 in the overview and settings tabs, never as Unlimited", async () => {
|
||||
const zeroLimitOrg = {
|
||||
...mockOrg,
|
||||
litellm_budget_table: { ...mockOrg.litellm_budget_table, tpm_limit: 0, rpm_limit: 0 },
|
||||
};
|
||||
mockUseOrganization.mockReturnValue({ data: zeroLimitOrg, isLoading: false } as unknown as ReturnType<
|
||||
typeof useOrganization
|
||||
>);
|
||||
|
||||
const user = userEvent.setup();
|
||||
renderWithProviders(
|
||||
<OrganizationInfoView
|
||||
organizationId="org_123"
|
||||
onClose={() => {}}
|
||||
accessToken="test-token"
|
||||
is_org_admin={false}
|
||||
is_proxy_admin={true}
|
||||
userModels={[]}
|
||||
editOrg={false}
|
||||
/>,
|
||||
);
|
||||
|
||||
const overview = await screen.findByRole("tabpanel", { name: "Overview" });
|
||||
expect(within(overview).getByText("TPM: 0")).toBeInTheDocument();
|
||||
expect(within(overview).getByText("RPM: 0")).toBeInTheDocument();
|
||||
|
||||
await user.click(screen.getByRole("tab", { name: "Settings" }));
|
||||
const settings = await screen.findByRole("tabpanel", { name: "Settings" });
|
||||
expect(within(settings).getByText("TPM: 0")).toBeInTheDocument();
|
||||
expect(within(settings).getByText("RPM: 0")).toBeInTheDocument();
|
||||
expect(screen.queryByText("TPM: Unlimited")).not.toBeInTheDocument();
|
||||
expect(screen.queryByText("RPM: Unlimited")).not.toBeInTheDocument();
|
||||
});
|
||||
|
|
|
|||
|
|
@ -206,8 +206,8 @@ const OrganizationInfoView: React.FC<OrganizationInfoProps> = ({
|
|||
<CardContent>
|
||||
<p className="text-sm text-muted-foreground">Rate Limits</p>
|
||||
<div className="mt-2 text-sm text-foreground">
|
||||
<p>TPM: {orgData.litellm_budget_table.tpm_limit || "Unlimited"}</p>
|
||||
<p>RPM: {orgData.litellm_budget_table.rpm_limit || "Unlimited"}</p>
|
||||
<p>TPM: {orgData.litellm_budget_table.tpm_limit ?? "Unlimited"}</p>
|
||||
<p>RPM: {orgData.litellm_budget_table.rpm_limit ?? "Unlimited"}</p>
|
||||
{orgData.litellm_budget_table.max_parallel_requests && (
|
||||
<p>Max Parallel Requests: {orgData.litellm_budget_table.max_parallel_requests}</p>
|
||||
)}
|
||||
|
|
@ -311,8 +311,8 @@ const OrganizationInfoView: React.FC<OrganizationInfoProps> = ({
|
|||
</div>
|
||||
<div>
|
||||
<p className="font-medium text-foreground">Rate Limits</p>
|
||||
<div>TPM: {orgData.litellm_budget_table.tpm_limit || "Unlimited"}</div>
|
||||
<div>RPM: {orgData.litellm_budget_table.rpm_limit || "Unlimited"}</div>
|
||||
<div>TPM: {orgData.litellm_budget_table.tpm_limit ?? "Unlimited"}</div>
|
||||
<div>RPM: {orgData.litellm_budget_table.rpm_limit ?? "Unlimited"}</div>
|
||||
</div>
|
||||
<div>
|
||||
<p className="font-medium text-foreground">Budget</p>
|
||||
|
|
|
|||
|
|
@ -19,6 +19,15 @@ describe("CreatedKeyDisplay", () => {
|
|||
expect(screen.getByText("sk-test-123")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("should theme the key box with tokens instead of a hardcoded light background", () => {
|
||||
render(<CreatedKeyDisplay apiKey="sk-test-123" />);
|
||||
|
||||
const keyBox = screen.getByText("sk-test-123").parentElement as HTMLElement;
|
||||
|
||||
expect(keyBox).not.toHaveAttribute("style");
|
||||
expect(keyBox).toHaveClass("bg-muted");
|
||||
});
|
||||
|
||||
it("should display the security warning", () => {
|
||||
render(<CreatedKeyDisplay apiKey="sk-test-123" />);
|
||||
expect(screen.getByText(/you will not be able to view it again/i)).toBeInTheDocument();
|
||||
|
|
|
|||
|
|
@ -29,15 +29,8 @@ const CreatedKeyDisplay: React.FC<CreatedKeyDisplayProps> = ({ apiKey }) => {
|
|||
</p>
|
||||
|
||||
<p className="text-sm text-muted-foreground mt-3 mb-1">Virtual Key:</p>
|
||||
<div
|
||||
style={{
|
||||
background: "#f8f8f8",
|
||||
padding: "10px",
|
||||
borderRadius: "5px",
|
||||
marginBottom: "10px",
|
||||
}}
|
||||
>
|
||||
<pre style={{ wordWrap: "break-word", whiteSpace: "normal", margin: 0 }}>{apiKey}</pre>
|
||||
<div className="bg-muted rounded-md p-2.5 mb-2.5">
|
||||
<pre className="m-0 whitespace-normal break-words text-foreground">{apiKey}</pre>
|
||||
</div>
|
||||
|
||||
<CopyToClipboard text={apiKey} onCopy={handleCopy}>
|
||||
|
|
|
|||
|
|
@ -110,7 +110,7 @@ describe("EditMembership submit payload", () => {
|
|||
expect(submitted()).toStrictEqual(expected);
|
||||
});
|
||||
|
||||
it("collapses falsy budget and limit values to null and a missing model list to an empty array", async () => {
|
||||
it("keeps stored 0 budget and limits as 0 on an untouched save, collapsing only empty strings and a missing model list", async () => {
|
||||
renderEdit(teamMemberConfig, {
|
||||
user_id: "u1",
|
||||
user_email: "a@b.com",
|
||||
|
|
@ -128,10 +128,10 @@ describe("EditMembership submit payload", () => {
|
|||
user_email: "a@b.com",
|
||||
user_id: "u1",
|
||||
role: "user",
|
||||
max_budget_in_team: null,
|
||||
max_budget_in_team: 0,
|
||||
budget_duration: null,
|
||||
tpm_limit: null,
|
||||
rpm_limit: null,
|
||||
tpm_limit: 0,
|
||||
rpm_limit: 0,
|
||||
allowed_models: [],
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -319,6 +319,34 @@ describe("TeamInfoView", () => {
|
|||
expect(screen.getByText(/of \$1,000\.00/)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("renders a tpm/rpm/budget limit of 0 as 0 in the overview and settings tabs, never as Unlimited or No Limit", async () => {
|
||||
vi.mocked(networking.teamInfoCall).mockResolvedValue(
|
||||
createMockTeamData({
|
||||
tpm_limit: 0,
|
||||
rpm_limit: 0,
|
||||
team_member_budget_table: { max_budget: 0, budget_duration: null, tpm_limit: 0, rpm_limit: 0 },
|
||||
}),
|
||||
);
|
||||
|
||||
renderWithProviders(<TeamInfoView {...defaultProps} />);
|
||||
|
||||
const overview = await screen.findByRole("tabpanel", { name: "Overview" });
|
||||
expect(within(overview).getByText("TPM: 0")).toBeInTheDocument();
|
||||
expect(within(overview).getByText("RPM: 0")).toBeInTheDocument();
|
||||
|
||||
await userEvent.setup({ delay: null }).click(screen.getByRole("tab", { name: "Settings" }));
|
||||
const settings = await screen.findByRole("tabpanel", { name: "Settings" });
|
||||
expect(within(settings).getByText("TPM: 0")).toBeInTheDocument();
|
||||
expect(within(settings).getByText("RPM: 0")).toBeInTheDocument();
|
||||
expect(within(settings).getByText("TPM Limit: 0")).toBeInTheDocument();
|
||||
expect(within(settings).getByText("RPM Limit: 0")).toBeInTheDocument();
|
||||
expect(within(settings).getByText("Max Budget: 0")).toBeInTheDocument();
|
||||
expect(screen.queryByText("TPM: Unlimited")).not.toBeInTheDocument();
|
||||
expect(screen.queryByText("RPM: Unlimited")).not.toBeInTheDocument();
|
||||
expect(screen.queryByText("TPM Limit: No Limit")).not.toBeInTheDocument();
|
||||
expect(screen.queryByText("RPM Limit: No Limit")).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("should display guardrails in overview when present", async () => {
|
||||
vi.mocked(networking.teamInfoCall).mockResolvedValue(
|
||||
createMockTeamData({
|
||||
|
|
|
|||
|
|
@ -971,8 +971,8 @@ const TeamInfoView: React.FC<TeamInfoProps> = ({
|
|||
<Card className="block p-6">
|
||||
<p>Rate Limits</p>
|
||||
<div className="mt-2">
|
||||
<p>TPM: {info.tpm_limit || "Unlimited"}</p>
|
||||
<p>RPM: {info.rpm_limit || "Unlimited"}</p>
|
||||
<p>TPM: {info.tpm_limit ?? "Unlimited"}</p>
|
||||
<p>RPM: {info.rpm_limit ?? "Unlimited"}</p>
|
||||
{info.max_parallel_requests && <p>Max Parallel Requests: {info.max_parallel_requests}</p>}
|
||||
{(() => {
|
||||
const modelTpm = (info.metadata?.model_tpm_limit ?? {}) as Record<string, number>;
|
||||
|
|
@ -1760,8 +1760,8 @@ const TeamInfoView: React.FC<TeamInfoProps> = ({
|
|||
</div>
|
||||
<div>
|
||||
<p className="font-medium">Rate Limits</p>
|
||||
<div>TPM: {info.tpm_limit || "Unlimited"}</div>
|
||||
<div>RPM: {info.rpm_limit || "Unlimited"}</div>
|
||||
<div>TPM: {info.tpm_limit ?? "Unlimited"}</div>
|
||||
<div>RPM: {info.rpm_limit ?? "Unlimited"}</div>
|
||||
{(() => {
|
||||
const modelTpm = (info.metadata?.model_tpm_limit ?? {}) as Record<string, number>;
|
||||
const modelRpm = (info.metadata?.model_rpm_limit ?? {}) as Record<string, number>;
|
||||
|
|
@ -1811,11 +1811,11 @@ const TeamInfoView: React.FC<TeamInfoProps> = ({
|
|||
<Info className="ml-1 inline size-3.5 align-text-bottom" />
|
||||
</SimpleTooltip>
|
||||
</p>
|
||||
<div>Max Budget: {info.team_member_budget_table?.max_budget || "No Limit"}</div>
|
||||
<div>Max Budget: {info.team_member_budget_table?.max_budget ?? "No Limit"}</div>
|
||||
<div>Budget Duration: {info.team_member_budget_table?.budget_duration || "No Limit"}</div>
|
||||
<div>Key Duration: {info.metadata?.team_member_key_duration || "No Limit"}</div>
|
||||
<div>TPM Limit: {info.team_member_budget_table?.tpm_limit || "No Limit"}</div>
|
||||
<div>RPM Limit: {info.team_member_budget_table?.rpm_limit || "No Limit"}</div>
|
||||
<div>TPM Limit: {info.team_member_budget_table?.tpm_limit ?? "No Limit"}</div>
|
||||
<div>RPM Limit: {info.team_member_budget_table?.rpm_limit ?? "No Limit"}</div>
|
||||
</div>
|
||||
<div>
|
||||
<p className="font-medium">Router Settings</p>
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
import { screen } from "@testing-library/react";
|
||||
import { screen, within } from "@testing-library/react";
|
||||
import userEvent from "@testing-library/user-event";
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { renderWithProviders } from "../../../tests/test-utils";
|
||||
|
|
@ -322,6 +322,43 @@ describe("TeamMembersComponent", () => {
|
|||
expect(mockSetSelectedEditMember).toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("keeps a member's stored 0 limits as 0 in the table and in the edit payload, never unlimited", async () => {
|
||||
const user = userEvent.setup();
|
||||
vi.mocked(isProxyAdminRole).mockReturnValue(true);
|
||||
const baseTeamData = createMockTeamData();
|
||||
const teamData = {
|
||||
...baseTeamData,
|
||||
team_memberships: baseTeamData.team_memberships.map((membership, index) =>
|
||||
index === 0
|
||||
? {
|
||||
...membership,
|
||||
litellm_budget_table: { ...membership.litellm_budget_table, max_budget: 0, tpm_limit: 0, rpm_limit: 0 },
|
||||
}
|
||||
: membership,
|
||||
),
|
||||
};
|
||||
|
||||
renderWithProviders(
|
||||
<TeamMembersComponent
|
||||
teamData={teamData}
|
||||
canEditTeam={true}
|
||||
handleMemberDelete={mockHandleMemberDelete}
|
||||
setSelectedEditMember={mockSetSelectedEditMember}
|
||||
setIsEditMemberModalVisible={mockSetIsEditMemberModalVisible}
|
||||
setIsAddMemberModalVisible={mockSetIsAddMemberModalVisible}
|
||||
/>,
|
||||
);
|
||||
|
||||
const memberRow = screen.getByRole("row", { name: /user1@test\.com/ });
|
||||
expect(within(memberRow).getByText("0 RPM / 0 TPM")).toBeInTheDocument();
|
||||
expect(within(memberRow).queryByText("No Limits")).not.toBeInTheDocument();
|
||||
|
||||
await user.click(within(memberRow).getByTestId("edit-member"));
|
||||
|
||||
const zeroLimitsMember = { user_id: "user1@test.com", max_budget_in_team: 0, tpm_limit: 0, rpm_limit: 0 };
|
||||
expect(mockSetSelectedEditMember).toHaveBeenCalledWith(expect.objectContaining(zeroLimitsMember));
|
||||
});
|
||||
|
||||
it("should call setIsAddMemberModalVisible when Add Member button is clicked", async () => {
|
||||
const user = userEvent.setup();
|
||||
|
||||
|
|
|
|||
|
|
@ -71,8 +71,8 @@ export default function TeamMemberTab({
|
|||
const rpmLimit = membership?.litellm_budget_table?.rpm_limit;
|
||||
const tpmLimit = membership?.litellm_budget_table?.tpm_limit;
|
||||
|
||||
const rpmText = rpmLimit ? `${formatNumber(rpmLimit)} RPM` : null;
|
||||
const tpmText = tpmLimit ? `${formatNumber(tpmLimit)} TPM` : null;
|
||||
const rpmText = rpmLimit != null ? `${formatNumber(rpmLimit)} RPM` : null;
|
||||
const tpmText = tpmLimit != null ? `${formatNumber(tpmLimit)} TPM` : null;
|
||||
|
||||
const limits = [rpmText, tpmText].filter(Boolean);
|
||||
return limits.length > 0 ? limits.join(" / ") : "No Limits";
|
||||
|
|
@ -191,9 +191,9 @@ export default function TeamMemberTab({
|
|||
const membership = teamData.team_memberships.find((tm) => tm.user_id === record.user_id);
|
||||
const enhancedMember = {
|
||||
...record,
|
||||
max_budget_in_team: membership?.litellm_budget_table?.max_budget || null,
|
||||
tpm_limit: membership?.litellm_budget_table?.tpm_limit || null,
|
||||
rpm_limit: membership?.litellm_budget_table?.rpm_limit || null,
|
||||
max_budget_in_team: membership?.litellm_budget_table?.max_budget ?? null,
|
||||
tpm_limit: membership?.litellm_budget_table?.tpm_limit ?? null,
|
||||
rpm_limit: membership?.litellm_budget_table?.rpm_limit ?? null,
|
||||
budget_duration: membership?.litellm_budget_table?.budget_duration || null,
|
||||
allowed_models: membership?.litellm_budget_table?.allowed_models || [],
|
||||
};
|
||||
|
|
|
|||
|
|
@ -82,7 +82,7 @@ describe("buildMemberFormValues", () => {
|
|||
});
|
||||
});
|
||||
|
||||
it("collapses falsy budgets and limits to null and a missing model list to an empty array", () => {
|
||||
it("keeps a stored budget or limit of 0 as 0 because only null means unlimited", () => {
|
||||
expect(
|
||||
buildMemberFormValues(
|
||||
"edit",
|
||||
|
|
@ -90,6 +90,19 @@ describe("buildMemberFormValues", () => {
|
|||
teamConfig,
|
||||
),
|
||||
).toStrictEqual({
|
||||
user_email: "a@b.com",
|
||||
user_id: "u1",
|
||||
role: "user",
|
||||
max_budget_in_team: 0,
|
||||
budget_duration: null,
|
||||
tpm_limit: 0,
|
||||
rpm_limit: 0,
|
||||
allowed_models: [],
|
||||
});
|
||||
});
|
||||
|
||||
it("collapses missing budgets and limits to null and a missing model list to an empty array", () => {
|
||||
const unlimitedMember = {
|
||||
user_email: "a@b.com",
|
||||
user_id: "u1",
|
||||
role: "user",
|
||||
|
|
@ -98,7 +111,10 @@ describe("buildMemberFormValues", () => {
|
|||
tpm_limit: null,
|
||||
rpm_limit: null,
|
||||
allowed_models: [],
|
||||
});
|
||||
};
|
||||
expect(
|
||||
buildMemberFormValues("edit", { user_email: "a@b.com", user_id: "u1", role: "user" }, teamConfig),
|
||||
).toStrictEqual(unlimitedMember);
|
||||
});
|
||||
|
||||
it("falls back to the configured default role when the member has none", () => {
|
||||
|
|
|
|||
|
|
@ -43,9 +43,9 @@ export const buildMemberFormValues = (
|
|||
const seeded: MemberFormValues = {
|
||||
...initialData,
|
||||
role: (initialData.role as string) || config.defaultRole,
|
||||
max_budget_in_team: initialData.max_budget_in_team || null,
|
||||
tpm_limit: initialData.tpm_limit || null,
|
||||
rpm_limit: initialData.rpm_limit || null,
|
||||
max_budget_in_team: initialData.max_budget_in_team ?? null,
|
||||
tpm_limit: initialData.tpm_limit ?? null,
|
||||
rpm_limit: initialData.rpm_limit ?? null,
|
||||
budget_duration: initialData.budget_duration || null,
|
||||
allowed_models: initialData.allowed_models || [],
|
||||
};
|
||||
|
|
|
|||
|
|
@ -100,6 +100,9 @@ export default function KeyInfoView({
|
|||
// Add local state to maintain key data and track regeneration
|
||||
const [currentKeyData, setCurrentKeyData] = useState<KeyResponse | undefined>(keyData);
|
||||
const [lastRegeneratedAt, setLastRegeneratedAt] = useState<Date | null>(null);
|
||||
const [keyDataUpdateHeldUntilModalClose, setKeyDataUpdateHeldUntilModalClose] = useState<Partial<KeyResponse> | null>(
|
||||
null,
|
||||
);
|
||||
const [isRecentlyRegenerated, setIsRecentlyRegenerated] = useState(false);
|
||||
const [policyGuardrails, setPolicyGuardrails] = useState<Record<string, string[]>>({});
|
||||
const [loadingPolicies, setLoadingPolicies] = useState(false);
|
||||
|
|
@ -352,6 +355,7 @@ export default function KeyInfoView({
|
|||
};
|
||||
|
||||
const handleRegenerateKeyUpdate = (updatedKeyData: Partial<KeyResponse>) => {
|
||||
const regeneratedAt = new Date();
|
||||
// Update local state immediately with ALL the new data
|
||||
setCurrentKeyData((prevData) => {
|
||||
if (!prevData) return undefined;
|
||||
|
|
@ -359,20 +363,26 @@ export default function KeyInfoView({
|
|||
...prevData,
|
||||
...updatedKeyData, // This should include the new token (key-id)
|
||||
// Update the created_at to show when it was regenerated
|
||||
created_at: new Date().toLocaleString(),
|
||||
created_at: regeneratedAt.toLocaleString(),
|
||||
};
|
||||
return newData;
|
||||
});
|
||||
|
||||
// Track regeneration timestamp
|
||||
setLastRegeneratedAt(new Date());
|
||||
setLastRegeneratedAt(regeneratedAt);
|
||||
setIsRecentlyRegenerated(true);
|
||||
|
||||
if (onKeyDataUpdate) {
|
||||
onKeyDataUpdate({
|
||||
...updatedKeyData,
|
||||
created_at: new Date().toLocaleString(),
|
||||
});
|
||||
setKeyDataUpdateHeldUntilModalClose({
|
||||
...updatedKeyData,
|
||||
created_at: regeneratedAt.toLocaleString(),
|
||||
});
|
||||
};
|
||||
|
||||
const handleRegenerateModalClose = () => {
|
||||
setIsRegenerateModalOpen(false);
|
||||
if (keyDataUpdateHeldUntilModalClose) {
|
||||
setKeyDataUpdateHeldUntilModalClose(null);
|
||||
onKeyDataUpdate?.(keyDataUpdateHeldUntilModalClose);
|
||||
}
|
||||
};
|
||||
|
||||
|
|
@ -506,7 +516,7 @@ export default function KeyInfoView({
|
|||
<RegenerateKeyModal
|
||||
selectedToken={currentKeyData}
|
||||
visible={isRegenerateModalOpen}
|
||||
onClose={() => setIsRegenerateModalOpen(false)}
|
||||
onClose={handleRegenerateModalClose}
|
||||
onKeyUpdate={handleRegenerateKeyUpdate}
|
||||
/>
|
||||
|
||||
|
|
|
|||
2
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
2
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -23142,7 +23142,7 @@ export interface components {
|
|||
* CallTypes
|
||||
* @enum {string}
|
||||
*/
|
||||
CallTypes: "embedding" | "aembedding" | "completion" | "acompletion" | "atext_completion" | "text_completion" | "image_generation" | "aimage_generation" | "image_edit" | "aimage_edit" | "moderation" | "amoderation" | "atranscription" | "transcription" | "aspeech" | "speech" | "rerank" | "arerank" | "search" | "asearch" | "_arealtime" | "_aresponses_websocket" | "create_batch" | "acreate_batch" | "aretrieve_batch" | "retrieve_batch" | "acancel_batch" | "cancel_batch" | "pass_through_endpoint" | "anthropic_messages" | "aanthropic_messages" | "get_assistants" | "aget_assistants" | "create_assistants" | "acreate_assistants" | "delete_assistant" | "adelete_assistant" | "acreate_thread" | "create_thread" | "aget_thread" | "get_thread" | "a_add_message" | "add_message" | "aget_messages" | "get_messages" | "arun_thread" | "run_thread" | "arun_thread_stream" | "run_thread_stream" | "afile_retrieve" | "file_retrieve" | "afile_delete" | "file_delete" | "afile_list" | "file_list" | "acreate_file" | "create_file" | "afile_content" | "file_content" | "create_fine_tuning_job" | "acreate_fine_tuning_job" | "create_video" | "acreate_video" | "avideo_retrieve" | "video_retrieve" | "avideo_content" | "video_content" | "video_remix" | "avideo_remix" | "video_list" | "avideo_list" | "video_retrieve_job" | "avideo_retrieve_job" | "video_delete" | "avideo_delete" | "video_create_character" | "avideo_create_character" | "video_get_character" | "avideo_get_character" | "video_edit" | "avideo_edit" | "video_extension" | "avideo_extension" | "vector_store_file_create" | "avector_store_file_create" | "vector_store_file_list" | "avector_store_file_list" | "vector_store_file_retrieve" | "avector_store_file_retrieve" | "vector_store_file_content" | "avector_store_file_content" | "vector_store_file_update" | "avector_store_file_update" | "vector_store_file_delete" | "avector_store_file_delete" | "vector_store_create" | "avector_store_create" | "vector_store_search" | "avector_store_search" | "ingest" | "aingest" | "query" | "aquery" | "create_container" | "acreate_container" | "list_containers" | "alist_containers" | "retrieve_container" | "aretrieve_container" | "delete_container" | "adelete_container" | "list_container_files" | "alist_container_files" | "upload_container_file" | "aupload_container_file" | "create_sandbox" | "acreate_sandbox" | "delete_sandbox" | "adelete_sandbox" | "run_code" | "arun_code" | "code_interpreter_tool" | "acode_interpreter_tool" | "acancel_fine_tuning_job" | "cancel_fine_tuning_job" | "alist_fine_tuning_jobs" | "list_fine_tuning_jobs" | "aretrieve_fine_tuning_job" | "retrieve_fine_tuning_job" | "responses" | "aresponses" | "alist_input_items" | "llm_passthrough_route" | "allm_passthrough_route" | "generate_content" | "agenerate_content" | "generate_content_stream" | "agenerate_content_stream" | "ocr" | "aocr" | "call_mcp_tool" | "list_mcp_tools" | "asend_message" | "send_message" | "acreate_skill";
|
||||
CallTypes: "embedding" | "aembedding" | "completion" | "acompletion" | "atext_completion" | "text_completion" | "image_generation" | "aimage_generation" | "image_edit" | "aimage_edit" | "moderation" | "amoderation" | "atranscription" | "transcription" | "aspeech" | "speech" | "rerank" | "arerank" | "search" | "asearch" | "_arealtime" | "_aresponses_websocket" | "create_batch" | "acreate_batch" | "aretrieve_batch" | "retrieve_batch" | "acancel_batch" | "cancel_batch" | "pass_through_endpoint" | "anthropic_messages" | "aanthropic_messages" | "get_assistants" | "aget_assistants" | "create_assistants" | "acreate_assistants" | "delete_assistant" | "adelete_assistant" | "acreate_thread" | "create_thread" | "aget_thread" | "get_thread" | "a_add_message" | "add_message" | "aget_messages" | "get_messages" | "arun_thread" | "run_thread" | "arun_thread_stream" | "run_thread_stream" | "afile_retrieve" | "file_retrieve" | "afile_delete" | "file_delete" | "afile_list" | "file_list" | "acreate_file" | "create_file" | "afile_content" | "file_content" | "create_fine_tuning_job" | "acreate_fine_tuning_job" | "create_video" | "acreate_video" | "avideo_retrieve" | "video_retrieve" | "avideo_content" | "video_content" | "video_remix" | "avideo_remix" | "video_list" | "avideo_list" | "video_retrieve_job" | "avideo_retrieve_job" | "video_delete" | "avideo_delete" | "video_create_character" | "avideo_create_character" | "video_get_character" | "avideo_get_character" | "video_edit" | "avideo_edit" | "video_extension" | "avideo_extension" | "vector_store_file_create" | "avector_store_file_create" | "vector_store_file_list" | "avector_store_file_list" | "vector_store_file_retrieve" | "avector_store_file_retrieve" | "vector_store_file_content" | "avector_store_file_content" | "vector_store_file_update" | "avector_store_file_update" | "vector_store_file_delete" | "avector_store_file_delete" | "vector_store_create" | "avector_store_create" | "vector_store_search" | "avector_store_search" | "ingest" | "aingest" | "query" | "aquery" | "create_interaction" | "acreate_interaction" | "create_container" | "acreate_container" | "list_containers" | "alist_containers" | "retrieve_container" | "aretrieve_container" | "delete_container" | "adelete_container" | "list_container_files" | "alist_container_files" | "upload_container_file" | "aupload_container_file" | "create_sandbox" | "acreate_sandbox" | "delete_sandbox" | "adelete_sandbox" | "run_code" | "arun_code" | "code_interpreter_tool" | "acode_interpreter_tool" | "acancel_fine_tuning_job" | "cancel_fine_tuning_job" | "alist_fine_tuning_jobs" | "list_fine_tuning_jobs" | "aretrieve_fine_tuning_job" | "retrieve_fine_tuning_job" | "responses" | "aresponses" | "alist_input_items" | "llm_passthrough_route" | "allm_passthrough_route" | "generate_content" | "agenerate_content" | "generate_content_stream" | "agenerate_content_stream" | "ocr" | "aocr" | "call_mcp_tool" | "list_mcp_tools" | "asend_message" | "send_message" | "acreate_skill";
|
||||
/** CallbackDelete */
|
||||
CallbackDelete: {
|
||||
/** Callback Name */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue