mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-24 00:52:24 +00:00
feat(cost): warn and count $0 cost on billable requests (#42345)
* feat(cost): warn and count $0 cost on billable requests
A request that carries usage but prices to $0 on a model whose pricing
entry has a non-zero rate now logs one warning naming the model, the
pricing entry, and the missing rate, and increments
litellm_zero_cost_requests_total{requested_model, model, model_id,
api_provider, reason}. Free models (every used rate is 0), requests
without usage, and unmapped models stay silent. The diagnostic rides on
the standard logging payload as zero_cost_diagnostic
* fix(cost): keep the zero-cost diagnostic importable on 3.10 and recursion-free
* fix(cost): warn once per request when a $0 result is priced again
* fix(cost): judge a free deployment by its own pricing and keep it silent on calculator errors
* fix(cost): warn once per request when a usage-less evaluation sits between two zero-cost findings
* fix(cost): judge zero-cost findings by the priced entry, skip cache hits, count failure rows
* test(cost): type the zero-cost diagnostic test helpers
* test(logging): flag a $0 terminal Responses stream event by its inner response
* chore: restore the lazy OpenAPI snapshot as CI's Python 3.12 generates it
---------
Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
1106b16745
commit
537e8ac068
10 changed files with 1268 additions and 61 deletions
|
|
@ -6267,6 +6267,63 @@
|
|||
],
|
||||
"title": "Spend update queue sizes (litellm_<queue>_size)",
|
||||
"type": "timeseries"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_PROMETHEUS}"
|
||||
},
|
||||
"description": "Requests that carried usage but were logged at $0 on a model whose pricing entry has a non-zero rate, by requested model and reason",
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"color": {
|
||||
"mode": "palette-classic"
|
||||
},
|
||||
"custom": {
|
||||
"drawStyle": "line",
|
||||
"fillOpacity": 10,
|
||||
"lineWidth": 1,
|
||||
"showPoints": "never",
|
||||
"spanNulls": false
|
||||
},
|
||||
"unit": "short"
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"gridPos": {
|
||||
"h": 8,
|
||||
"w": 12,
|
||||
"x": 0,
|
||||
"y": 430
|
||||
},
|
||||
"id": 110,
|
||||
"options": {
|
||||
"legend": {
|
||||
"calcs": [],
|
||||
"displayMode": "list",
|
||||
"placement": "bottom",
|
||||
"showLegend": true
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi",
|
||||
"sort": "desc"
|
||||
}
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_PROMETHEUS}"
|
||||
},
|
||||
"editorMode": "code",
|
||||
"expr": "sum(rate(litellm_zero_cost_requests_total[$__rate_interval])) by (requested_model, reason)",
|
||||
"legendFormat": "{{requested_model}} / {{reason}}",
|
||||
"range": true,
|
||||
"refId": "A"
|
||||
}
|
||||
],
|
||||
"title": "litellm_zero_cost_requests rate",
|
||||
"type": "timeseries"
|
||||
}
|
||||
],
|
||||
"preload": false,
|
||||
|
|
|
|||
|
|
@ -948,7 +948,7 @@ def _extract_service_tier(source: object) -> str | None:
|
|||
return None
|
||||
|
||||
|
||||
def _get_usage_object(
|
||||
def get_usage_object(
|
||||
completion_response: object,
|
||||
) -> Usage | None:
|
||||
usage_obj: Final = cast(
|
||||
|
|
@ -1336,7 +1336,7 @@ def completion_cost(
|
|||
cache_creation_input_tokens: int | None = None
|
||||
cache_read_input_tokens: int | None = None
|
||||
audio_transcription_file_duration: float = 0.0
|
||||
provider_usage_object: Final = _get_usage_object(completion_response=completion_response)
|
||||
provider_usage_object: Final = get_usage_object(completion_response=completion_response)
|
||||
cost_per_token_usage_object: Final[Usage | None] = (
|
||||
_without_provider_stated_cost(provider_usage_object) if custom_pricing else provider_usage_object
|
||||
)
|
||||
|
|
@ -2033,6 +2033,45 @@ def _cost_map_model_info(model: str, custom_llm_provider: str | None) -> ModelIn
|
|||
return None
|
||||
|
||||
|
||||
def _raw_cost_map_entry(key: str) -> Mapping[str, object] | None:
|
||||
raw_entry: Final = litellm.model_cost.get(key)
|
||||
return raw_entry if isinstance(raw_entry, Mapping) else None
|
||||
|
||||
|
||||
def pricing_entry_for_cost_calc(
|
||||
model: str | None,
|
||||
completion_response: object | None,
|
||||
custom_llm_provider: str | None,
|
||||
custom_pricing: bool | None,
|
||||
base_model: str | None,
|
||||
router_model_id: str | None,
|
||||
region_name: str | None,
|
||||
litellm_logging_obj: LitellmLoggingObject | None,
|
||||
) -> tuple[str, Mapping[str, object]] | None:
|
||||
deployment_entry: Final = _deployment_model_info(litellm_logging_obj, custom_pricing, router_model_id)
|
||||
deployment_key: Final = router_model_id or model
|
||||
if deployment_entry is not None and deployment_key is not None:
|
||||
registered_entry: Final = _raw_cost_map_entry(router_model_id) if router_model_id is not None else None
|
||||
return deployment_key, registered_entry or deployment_entry
|
||||
selected_model: Final = _select_model_name_for_cost_calc(
|
||||
model=model,
|
||||
completion_response=completion_response,
|
||||
base_model=base_model,
|
||||
custom_pricing=custom_pricing,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
router_model_id=router_model_id,
|
||||
region_name=region_name,
|
||||
)
|
||||
candidates: Final = (selected_model, _get_response_model(completion_response), model)
|
||||
resolved: Final = next(
|
||||
(info for info in (_cost_map_model_info(name, custom_llm_provider) for name in candidates if name) if info),
|
||||
None,
|
||||
)
|
||||
if resolved is None:
|
||||
return None
|
||||
return resolved["key"], _raw_cost_map_entry(resolved["key"]) or resolved
|
||||
|
||||
|
||||
def ocr_cost(
|
||||
model: str,
|
||||
custom_llm_provider: str | None,
|
||||
|
|
|
|||
|
|
@ -10,6 +10,7 @@ import sys
|
|||
from collections.abc import Awaitable, Callable, Mapping, Sequence
|
||||
from dataclasses import replace
|
||||
from datetime import datetime, timedelta
|
||||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal, Protocol, TypeAlias, TypeVar, cast
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
|
@ -66,6 +67,7 @@ from litellm.types.proxy.carried_budget_state import (
|
|||
from litellm.types.utils import (
|
||||
StandardLoggingGuardrailInformation,
|
||||
StandardLoggingPayload,
|
||||
StandardLoggingZeroCostDiagnostic,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
|
@ -713,6 +715,15 @@ class PrometheusLogger(CustomLogger):
|
|||
labelnames=self.get_labels_for_metric("litellm_requests_metric"),
|
||||
)
|
||||
|
||||
self.litellm_zero_cost_requests_total = self._counter_factory(
|
||||
name="litellm_zero_cost_requests_total",
|
||||
documentation=(
|
||||
"Requests that carried usage but were logged at $0 on a model whose pricing entry "
|
||||
"has a non-zero rate, by reason (missing_pricing_key, pricing_not_applied, cost_calculation_error)"
|
||||
),
|
||||
labelnames=self.get_labels_for_metric("litellm_zero_cost_requests_total"),
|
||||
)
|
||||
|
||||
# Cache metrics
|
||||
self.litellm_cache_hits_metric = self._counter_factory(
|
||||
name="litellm_cache_hits_metric",
|
||||
|
|
@ -1410,6 +1421,11 @@ class PrometheusLogger(CustomLogger):
|
|||
enum_values=enum_values,
|
||||
label_context=label_context,
|
||||
)
|
||||
self._increment_zero_cost_requests_metric(
|
||||
zero_cost_diagnostic=standard_logging_payload.get("zero_cost_diagnostic"),
|
||||
enum_values=enum_values,
|
||||
label_context=label_context,
|
||||
)
|
||||
|
||||
# input, output, total token metrics
|
||||
self._increment_token_metrics(
|
||||
|
|
@ -1983,6 +1999,30 @@ class PrometheusLogger(CustomLogger):
|
|||
amount=float(response_cost),
|
||||
)
|
||||
|
||||
def _increment_zero_cost_requests_metric(
|
||||
self,
|
||||
zero_cost_diagnostic: StandardLoggingZeroCostDiagnostic | None,
|
||||
enum_values: UserAPIKeyLabelValues,
|
||||
label_context: PrometheusLabelFactoryContext,
|
||||
) -> None:
|
||||
if zero_cost_diagnostic is None:
|
||||
return
|
||||
supported_labels: Final = self.get_labels_for_metric("litellm_zero_cost_requests_total")
|
||||
reason_label: Final = (
|
||||
MappingProxyType({ZERO_COST_REASON_LABEL: zero_cost_diagnostic["reason"]})
|
||||
if ZERO_COST_REASON_LABEL in supported_labels
|
||||
else MappingProxyType({})
|
||||
)
|
||||
labels: Final = MappingProxyType(
|
||||
{
|
||||
**prometheus_label_factory(
|
||||
supported_enum_labels=supported_labels, enum_values=enum_values, label_context=label_context
|
||||
),
|
||||
**reason_label,
|
||||
}
|
||||
)
|
||||
self.litellm_zero_cost_requests_total.labels(**labels).inc()
|
||||
|
||||
@staticmethod
|
||||
def _get_remaining_from_v3_rate_limit_headers(
|
||||
standard_logging_payload: StandardLoggingPayload | None,
|
||||
|
|
@ -2333,6 +2373,8 @@ class PrometheusLogger(CustomLogger):
|
|||
team_alias=user_api_team_alias,
|
||||
user=user_id,
|
||||
model_id=standard_logging_payload.get("model_id", ""),
|
||||
requested_model=standard_logging_payload.get("model_group"),
|
||||
api_provider=standard_logging_payload.get("custom_llm_provider"),
|
||||
custom_metadata_labels=get_custom_labels_from_metadata(
|
||||
metadata=_get_combined_custom_metadata_from_standard_logging_payload(
|
||||
standard_logging_payload=standard_logging_payload
|
||||
|
|
@ -2345,6 +2387,11 @@ class PrometheusLogger(CustomLogger):
|
|||
"litellm_llm_api_failed_requests_metric",
|
||||
enum_values,
|
||||
)
|
||||
self._increment_zero_cost_requests_metric(
|
||||
zero_cost_diagnostic=standard_logging_payload.get("zero_cost_diagnostic"),
|
||||
enum_values=enum_values,
|
||||
label_context=PrometheusLabelFactoryContext(enum_values),
|
||||
)
|
||||
self.set_llm_deployment_failure_metrics(kwargs)
|
||||
await self._set_org_budget_metrics_after_api_request(
|
||||
org_id=user_api_key_org_id,
|
||||
|
|
|
|||
|
|
@ -50,6 +50,8 @@ from litellm.cost_calculator import (
|
|||
RealtimeAPITokenUsageProcessor,
|
||||
ResponsesWebSocketTokenUsageProcessor,
|
||||
_select_model_name_for_cost_calc,
|
||||
get_usage_object,
|
||||
pricing_entry_for_cost_calc,
|
||||
)
|
||||
from litellm.exceptions import (
|
||||
BudgetExceededError,
|
||||
|
|
@ -89,6 +91,10 @@ from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
|
|||
from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import (
|
||||
InteractionsUsageObjectTransformation,
|
||||
)
|
||||
from litellm.litellm_core_utils.llm_cost_calc.zero_cost_diagnostic import (
|
||||
diagnose_zero_cost,
|
||||
zero_cost_warning,
|
||||
)
|
||||
from litellm.litellm_core_utils.logging_utils import (
|
||||
truncate_base64_in_messages,
|
||||
truncate_base64_in_messages_async,
|
||||
|
|
@ -157,6 +163,7 @@ from litellm.types.utils import (
|
|||
StandardLoggingPayloadStatusFields,
|
||||
StandardLoggingPromptManagementMetadata,
|
||||
StandardLoggingVectorStoreRequest,
|
||||
StandardLoggingZeroCostDiagnostic,
|
||||
TextCompletionResponse,
|
||||
TranscriptionResponse,
|
||||
Usage,
|
||||
|
|
@ -614,6 +621,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
self.truncated_messages_for_logging: str | list | dict | None = None # mutable-ok: logged messages shape
|
||||
## TIME TO FIRST TOKEN LOGGING ##
|
||||
self.completion_start_time: datetime.datetime | None = None
|
||||
self.zero_cost_warned: bool = False
|
||||
self._llm_caching_handler: LLMCachingHandler | None = None
|
||||
|
||||
# INITIAL LITELLM_PARAMS
|
||||
|
|
@ -1764,11 +1772,6 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
)
|
||||
|
||||
result_hidden_params: Final = getattr(priced_result, "_hidden_params", None) or MappingProxyType({})
|
||||
result_additional_headers: Final = (
|
||||
result_hidden_params.get("additional_headers")
|
||||
if isinstance(result_hidden_params, dict)
|
||||
else getattr(result_hidden_params, "additional_headers", None)
|
||||
)
|
||||
if isinstance(priced_result, (BaseModel, HttpxBinaryResponseContent)) and hasattr(
|
||||
priced_result, "_hidden_params"
|
||||
):
|
||||
|
|
@ -1776,6 +1779,12 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
if (
|
||||
"response_cost" in hidden_params and hidden_params["response_cost"] is not None
|
||||
): # use cost if already calculated
|
||||
self._record_zero_cost_diagnostic(
|
||||
priced_result,
|
||||
hidden_params["response_cost"],
|
||||
litellm_model_name=litellm_model_name,
|
||||
router_model_id=router_model_id or hidden_params.get("model_id"),
|
||||
)
|
||||
return hidden_params["response_cost"]
|
||||
elif router_model_id is None and "model_id" in hidden_params: # use model_id if not already set
|
||||
router_model_id = hidden_params["model_id"]
|
||||
|
|
@ -1787,18 +1796,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
router_model_id = self.get_router_model_id()
|
||||
|
||||
## RESPONSE COST ##
|
||||
spilled_over: Final = is_spilled_over_ptu_request(
|
||||
model_info=_deployment_model_info(self.litellm_params if hasattr(self, "litellm_params") else None),
|
||||
response_headers=self.model_call_details.get("response_headers"),
|
||||
additional_headers=result_additional_headers,
|
||||
)
|
||||
custom_pricing: Final = (
|
||||
False
|
||||
if spilled_over
|
||||
else use_custom_pricing_for_model(
|
||||
litellm_params=(self.litellm_params if hasattr(self, "litellm_params") else None)
|
||||
)
|
||||
)
|
||||
custom_pricing: Final = self._custom_pricing_for(priced_result)
|
||||
|
||||
prompt = self._prompt_for_cost_calculation()
|
||||
|
||||
|
|
@ -1850,9 +1848,18 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
|
||||
verbose_logger.debug("response_cost: %s", response_cost)
|
||||
additional_response_cost: Final[object] = self.model_call_details.get("additional_response_cost")
|
||||
if isinstance(additional_response_cost, (int, float)) and additional_response_cost > 0:
|
||||
return (response_cost or 0.0) + additional_response_cost
|
||||
return response_cost
|
||||
total_response_cost: Final = (
|
||||
(response_cost or 0.0) + additional_response_cost
|
||||
if isinstance(additional_response_cost, (int, float)) and additional_response_cost > 0
|
||||
else response_cost
|
||||
)
|
||||
self._record_zero_cost_diagnostic(
|
||||
priced_result,
|
||||
total_response_cost,
|
||||
litellm_model_name=litellm_model_name,
|
||||
router_model_id=router_model_id,
|
||||
)
|
||||
return total_response_cost
|
||||
except Exception as e: # error calculating cost
|
||||
debug_info = StandardLoggingModelCostFailureDebugInformation(
|
||||
error_str=str(e),
|
||||
|
|
@ -1866,9 +1873,108 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
)
|
||||
verbose_logger.debug("response_cost_failure_debug_information: %s", debug_info)
|
||||
self.model_call_details["response_cost_failure_debug_information"] = debug_info
|
||||
self._record_zero_cost_diagnostic(
|
||||
priced_result,
|
||||
None,
|
||||
calculation_failed=True,
|
||||
litellm_model_name=litellm_model_name,
|
||||
router_model_id=router_model_id,
|
||||
)
|
||||
|
||||
return None
|
||||
|
||||
def _record_zero_cost_diagnostic(
|
||||
self,
|
||||
result: object,
|
||||
response_cost: float | None,
|
||||
*,
|
||||
calculation_failed: bool = False,
|
||||
litellm_model_name: str | None = None,
|
||||
router_model_id: str | None = None,
|
||||
) -> None:
|
||||
if response_cost is None and not calculation_failed:
|
||||
return
|
||||
if self.model_call_details.get("cache_hit") is True:
|
||||
self.model_call_details["zero_cost_diagnostic"] = None
|
||||
return
|
||||
try:
|
||||
finding: Final = self._zero_cost_finding(
|
||||
result,
|
||||
response_cost,
|
||||
calculation_failed=calculation_failed,
|
||||
litellm_model_name=litellm_model_name,
|
||||
router_model_id=router_model_id,
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 # the pricing helpers raise plain Exception and a diagnostic must never break cost tracking
|
||||
verbose_logger.debug("zero_cost_diagnostic skipped: %s", e)
|
||||
return
|
||||
self.model_call_details["zero_cost_diagnostic"] = finding[0] if finding is not None else None
|
||||
if finding is None or self.zero_cost_warned:
|
||||
return
|
||||
self.zero_cost_warned = True
|
||||
verbose_logger.warning(finding[1])
|
||||
|
||||
def _zero_cost_finding(
|
||||
self,
|
||||
result: object,
|
||||
response_cost: float | None,
|
||||
*,
|
||||
calculation_failed: bool,
|
||||
litellm_model_name: str | None,
|
||||
router_model_id: str | None,
|
||||
) -> tuple[StandardLoggingZeroCostDiagnostic, str] | None:
|
||||
metadata: Final = StandardLoggingPayloadSetup.merge_litellm_metadata(self.litellm_params)
|
||||
if response_cost or is_unbilled_non_inference_call(self.call_type, metadata, result):
|
||||
return None
|
||||
usage: Final = get_usage_object(completion_response=result)
|
||||
if usage is None:
|
||||
return None
|
||||
model: Final = litellm_model_name or self.model
|
||||
custom_llm_provider: Final = self.model_call_details.get("custom_llm_provider")
|
||||
pricing: Final = pricing_entry_for_cost_calc(
|
||||
model=model,
|
||||
completion_response=result,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
custom_pricing=self._custom_pricing_for(result),
|
||||
base_model=_get_base_model_from_metadata(model_call_details=self.model_call_details),
|
||||
router_model_id=router_model_id or self.get_router_model_id(),
|
||||
region_name=_resolve_mantle_region_for_cost(
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
litellm_params=self.model_call_details.get("litellm_params"),
|
||||
),
|
||||
litellm_logging_obj=self,
|
||||
)
|
||||
if pricing is None:
|
||||
return None
|
||||
diagnostic: Final = diagnose_zero_cost(
|
||||
usage=usage, pricing_model=pricing[0], pricing_entry=pricing[1], calculation_failed=calculation_failed
|
||||
)
|
||||
if diagnostic is None:
|
||||
return None
|
||||
model_group: Final = metadata.get("model_group")
|
||||
return diagnostic, zero_cost_warning(
|
||||
diagnostic,
|
||||
model_group=model_group if isinstance(model_group, str) else None,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
usage=usage,
|
||||
)
|
||||
|
||||
def _custom_pricing_for(self, result: object) -> bool:
|
||||
litellm_params: Final = getattr(self, "litellm_params", None)
|
||||
result_hidden_params: Final = getattr(result, "_hidden_params", None) or MappingProxyType({})
|
||||
additional_headers: Final = (
|
||||
result_hidden_params.get("additional_headers")
|
||||
if isinstance(result_hidden_params, dict)
|
||||
else getattr(result_hidden_params, "additional_headers", None)
|
||||
)
|
||||
spilled_over: Final = is_spilled_over_ptu_request(
|
||||
model_info=_deployment_model_info(litellm_params),
|
||||
response_headers=self.model_call_details.get("response_headers"),
|
||||
additional_headers=additional_headers,
|
||||
)
|
||||
return False if spilled_over else use_custom_pricing_for_model(litellm_params=litellm_params)
|
||||
|
||||
def _prompt_for_cost_calculation(self) -> str:
|
||||
"""
|
||||
The raw input string is only priced directly for text-to-speech, which bills per character.
|
||||
|
|
@ -2213,6 +2319,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
self.model_call_details["response_cost"] = 0.0
|
||||
elif "response_cost" in hidden_params:
|
||||
self.model_call_details["response_cost"] = hidden_params["response_cost"]
|
||||
self._record_zero_cost_diagnostic(logging_result, hidden_params["response_cost"])
|
||||
elif (existing_cost := self.model_call_details.get("response_cost")) is not None and existing_cost != 0:
|
||||
# Preserve response_cost if already calculated (e.g., by pass-through
|
||||
# handlers like Gemini/Vertex which call completion_cost directly).
|
||||
|
|
@ -6507,6 +6614,7 @@ def get_standard_logging_object_payload(
|
|||
error_str=error_str,
|
||||
error_information=error_information,
|
||||
response_cost_failure_debug_info=kwargs.get("response_cost_failure_debug_information"),
|
||||
zero_cost_diagnostic=kwargs.get("zero_cost_diagnostic"),
|
||||
guardrail_information=metadata.get("standard_logging_guardrail_information", None),
|
||||
standard_built_in_tools_params=standard_built_in_tools_params,
|
||||
)
|
||||
|
|
@ -6685,6 +6793,7 @@ def create_dummy_standard_logging_payload() -> StandardLoggingPayload:
|
|||
response_cost=response_cost,
|
||||
autorouter_savings=None,
|
||||
response_cost_failure_debug_info=None,
|
||||
zero_cost_diagnostic=None,
|
||||
status="success",
|
||||
total_tokens=int(DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT + DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT),
|
||||
prompt_tokens=int(DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT),
|
||||
|
|
|
|||
146
litellm/litellm_core_utils/llm_cost_calc/zero_cost_diagnostic.py
Normal file
146
litellm/litellm_core_utils/llm_cost_calc/zero_cost_diagnostic.py
Normal file
|
|
@ -0,0 +1,146 @@
|
|||
from collections.abc import Mapping
|
||||
from functools import reduce
|
||||
from typing import Final
|
||||
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
from typing_extensions import assert_never
|
||||
|
||||
from litellm.types.utils import StandardLoggingZeroCostDiagnostic, Usage
|
||||
|
||||
ZERO_COST_COUNTER_NAME: Final = "litellm_zero_cost_requests_total"
|
||||
|
||||
_TEXT_INPUT_RATE: Final = "input_cost_per_token"
|
||||
_AUDIO_INPUT_RATE: Final = "input_cost_per_audio_token"
|
||||
_TEXT_OUTPUT_RATE: Final = "output_cost_per_token"
|
||||
_AUDIO_OUTPUT_RATE: Final = "output_cost_per_audio_token"
|
||||
_RATE_KEY_MARKERS: Final = ("cost", "pricing")
|
||||
_NESTED_PRICING: Final = TypeAdapter(Mapping[str, object] | tuple[object, ...])
|
||||
_MAX_PRICING_DEPTH: Final = 4
|
||||
|
||||
|
||||
def _audio_tokens(details: object) -> int:
|
||||
audio_tokens: Final = getattr(details, "audio_tokens", None)
|
||||
return audio_tokens if isinstance(audio_tokens, int) and audio_tokens > 0 else 0
|
||||
|
||||
|
||||
def _tokens(value: object) -> int:
|
||||
return value if isinstance(value, int) and value > 0 else 0
|
||||
|
||||
|
||||
def used_pricing_keys(usage: Usage) -> tuple[str, ...]:
|
||||
prompt_audio: Final = _audio_tokens(usage.prompt_tokens_details)
|
||||
completion_audio: Final = _audio_tokens(usage.completion_tokens_details)
|
||||
prompt_text: Final = _tokens(usage.prompt_tokens) - prompt_audio
|
||||
completion_text: Final = _tokens(usage.completion_tokens) - completion_audio
|
||||
components: Final = (
|
||||
(_TEXT_INPUT_RATE, prompt_text),
|
||||
(_AUDIO_INPUT_RATE, prompt_audio),
|
||||
(_TEXT_OUTPUT_RATE, completion_text),
|
||||
(_AUDIO_OUTPUT_RATE, completion_audio),
|
||||
)
|
||||
return tuple(key for key, count in components if count > 0)
|
||||
|
||||
|
||||
def _nested_pricing(value: object) -> Mapping[str, object] | tuple[object, ...] | None:
|
||||
try:
|
||||
return _NESTED_PRICING.validate_python(value)
|
||||
except ValidationError:
|
||||
return None
|
||||
|
||||
|
||||
def _is_rate_key(key: str) -> bool:
|
||||
return any(marker in key for marker in _RATE_KEY_MARKERS)
|
||||
|
||||
|
||||
def _rate_values(value: object) -> tuple[object, ...]:
|
||||
nested: Final = _nested_pricing(value)
|
||||
if isinstance(nested, Mapping):
|
||||
return tuple(child for key, child in nested.items() if _is_rate_key(key))
|
||||
if nested is None:
|
||||
return (value,)
|
||||
return nested
|
||||
|
||||
|
||||
def _expand_rate_values(values: tuple[object, ...], _depth: int) -> tuple[object, ...]:
|
||||
return tuple(nested for value in values for nested in _rate_values(value))
|
||||
|
||||
|
||||
def _is_positive_number(value: object) -> bool:
|
||||
return not isinstance(value, bool) and isinstance(value, (int, float)) and value > 0
|
||||
|
||||
|
||||
def _declares_a_rate(pricing_entry: Mapping[str, object]) -> bool:
|
||||
leaves: Final = reduce(_expand_rate_values, range(_MAX_PRICING_DEPTH), (pricing_entry,))
|
||||
return any(_is_positive_number(leaf) for leaf in leaves)
|
||||
|
||||
|
||||
def _is_explicit_zero(value: object) -> bool:
|
||||
return not isinstance(value, bool) and isinstance(value, (int, float)) and value == 0
|
||||
|
||||
|
||||
def diagnose_zero_cost(
|
||||
usage: Usage,
|
||||
pricing_model: str,
|
||||
pricing_entry: Mapping[str, object],
|
||||
calculation_failed: bool,
|
||||
) -> StandardLoggingZeroCostDiagnostic | None:
|
||||
used_keys: Final = used_pricing_keys(usage)
|
||||
if not used_keys:
|
||||
return None
|
||||
missing_keys: Final = tuple(key for key in used_keys if pricing_entry.get(key) is None)
|
||||
if not missing_keys and all(_is_explicit_zero(pricing_entry[key]) for key in used_keys):
|
||||
return None
|
||||
if not _declares_a_rate(pricing_entry):
|
||||
return None
|
||||
if calculation_failed:
|
||||
return StandardLoggingZeroCostDiagnostic(
|
||||
reason="cost_calculation_error", pricing_model=pricing_model, missing_pricing_keys=()
|
||||
)
|
||||
if missing_keys:
|
||||
return StandardLoggingZeroCostDiagnostic(
|
||||
reason="missing_pricing_key", pricing_model=pricing_model, missing_pricing_keys=missing_keys
|
||||
)
|
||||
return StandardLoggingZeroCostDiagnostic(
|
||||
reason="pricing_not_applied", pricing_model=pricing_model, missing_pricing_keys=()
|
||||
)
|
||||
|
||||
|
||||
def _cause(diagnostic: StandardLoggingZeroCostDiagnostic) -> str:
|
||||
reason: Final = diagnostic["reason"]
|
||||
match reason:
|
||||
case "missing_pricing_key":
|
||||
return (
|
||||
f"pricing entry '{diagnostic['pricing_model']}' has no {', '.join(diagnostic['missing_pricing_keys'])}. "
|
||||
"Set the missing rate in the deployment's model_info or in the model cost map, "
|
||||
"or set every rate to 0 to mark the model free"
|
||||
)
|
||||
case "pricing_not_applied":
|
||||
return (
|
||||
f"pricing entry '{diagnostic['pricing_model']}' declares non-zero rates for this usage, "
|
||||
"but the cost calculator returned $0"
|
||||
)
|
||||
case "cost_calculation_error":
|
||||
return (
|
||||
f"cost calculation raised for pricing entry '{diagnostic['pricing_model']}', "
|
||||
"see response_cost_failure_debug_information"
|
||||
)
|
||||
case _:
|
||||
return assert_never(reason)
|
||||
|
||||
|
||||
def zero_cost_warning(
|
||||
diagnostic: StandardLoggingZeroCostDiagnostic,
|
||||
*,
|
||||
model_group: str | None,
|
||||
model: str,
|
||||
custom_llm_provider: str | None,
|
||||
usage: Usage,
|
||||
) -> str:
|
||||
request: Final = (
|
||||
f"model_group={model_group or model} model={model} provider={custom_llm_provider or 'unknown'} "
|
||||
f"prompt_tokens={_tokens(usage.prompt_tokens)} completion_tokens={_tokens(usage.completion_tokens)}"
|
||||
)
|
||||
return (
|
||||
f"Billable request priced at $0 and logged as such ({request}): {_cause(diagnostic)}. "
|
||||
f'Counted in {ZERO_COST_COUNTER_NAME}{{reason="{diagnostic["reason"]}"}}'
|
||||
)
|
||||
|
|
@ -131,6 +131,7 @@ EXCEPTION_STATUS: Final = "exception_status"
|
|||
EXCEPTION_CLASS: Final = "exception_class"
|
||||
RATE_LIMIT_CATEGORY: Final = "rate_limit_category"
|
||||
RATE_LIMIT_TYPE: Final = "rate_limit_type"
|
||||
ZERO_COST_REASON_LABEL: Final = "reason"
|
||||
STATUS_CODE: Final = "status_code"
|
||||
EXCEPTION_LABELS: Final = [EXCEPTION_STATUS, EXCEPTION_CLASS]
|
||||
LATENCY_BUCKETS: Final = (
|
||||
|
|
@ -279,6 +280,7 @@ DEFINED_PROMETHEUS_METRICS = Literal[
|
|||
"litellm_guardrail_latency_seconds",
|
||||
"litellm_guardrail_errors_total",
|
||||
"litellm_guardrail_requests_total",
|
||||
"litellm_zero_cost_requests_total",
|
||||
# Cache metrics
|
||||
"litellm_cache_hits_metric",
|
||||
"litellm_cache_misses_metric",
|
||||
|
|
@ -590,6 +592,14 @@ class PrometheusMetricLabels:
|
|||
UserAPIKeyLabelNames.SERVICE_TIER.value,
|
||||
]
|
||||
|
||||
litellm_zero_cost_requests_total = (
|
||||
UserAPIKeyLabelNames.REQUESTED_MODEL.value,
|
||||
UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value,
|
||||
UserAPIKeyLabelNames.MODEL_ID.value,
|
||||
UserAPIKeyLabelNames.API_PROVIDER.value,
|
||||
ZERO_COST_REASON_LABEL,
|
||||
)
|
||||
|
||||
litellm_input_tokens_metric = [
|
||||
UserAPIKeyLabelNames.END_USER.value,
|
||||
UserAPIKeyLabelNames.API_KEY_HASH.value,
|
||||
|
|
|
|||
|
|
@ -3206,6 +3206,15 @@ class StandardLoggingModelCostFailureDebugInformation(TypedDict, total=False):
|
|||
custom_pricing: bool | None
|
||||
|
||||
|
||||
ZeroCostReason = Literal["missing_pricing_key", "pricing_not_applied", "cost_calculation_error"]
|
||||
|
||||
|
||||
class StandardLoggingZeroCostDiagnostic(TypedDict):
|
||||
reason: ReadOnly[ZeroCostReason]
|
||||
pricing_model: ReadOnly[str]
|
||||
missing_pricing_keys: ReadOnly[tuple[str, ...]]
|
||||
|
||||
|
||||
class StandardLoggingPayloadErrorInformation(TypedDict, total=False):
|
||||
error_code: str | None
|
||||
error_class: str | None
|
||||
|
|
@ -3524,6 +3533,7 @@ class StandardLoggingPayload(ClassifierAudit):
|
|||
autorouter_savings_estimate: ReadOnly[Mapping[str, JsonValue] | None]
|
||||
autorouter_baseline_observation: ReadOnly[str | None]
|
||||
response_cost_failure_debug_info: StandardLoggingModelCostFailureDebugInformation | None
|
||||
zero_cost_diagnostic: NotRequired[ReadOnly[StandardLoggingZeroCostDiagnostic | None]]
|
||||
status: StandardLoggingPayloadStatus
|
||||
status_fields: StandardLoggingPayloadStatusFields
|
||||
custom_llm_provider: str | None
|
||||
|
|
|
|||
|
|
@ -0,0 +1,179 @@
|
|||
import datetime
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from prometheus_client import REGISTRY
|
||||
from prometheus_client.samples import Sample
|
||||
|
||||
import litellm
|
||||
from litellm.integrations.prometheus import PrometheusLogger
|
||||
from litellm.types.utils import StandardLoggingZeroCostDiagnostic
|
||||
|
||||
METRIC: Final = "litellm_zero_cost_requests_total"
|
||||
MISSING_KEY_DIAGNOSTIC: Final[StandardLoggingZeroCostDiagnostic] = {
|
||||
"reason": "missing_pricing_key",
|
||||
"pricing_model": "dep-1",
|
||||
"missing_pricing_keys": ("input_cost_per_token", "output_cost_per_token"),
|
||||
}
|
||||
|
||||
|
||||
def _clear_prometheus_registry() -> None:
|
||||
for collector in list(REGISTRY._collector_to_names.keys()):
|
||||
try:
|
||||
REGISTRY.unregister(collector)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _samples(metric_name: str) -> list[Sample]:
|
||||
return [sample for metric in REGISTRY.collect() for sample in metric.samples if sample.name == metric_name]
|
||||
|
||||
|
||||
def _payload(zero_cost_diagnostic: StandardLoggingZeroCostDiagnostic | None) -> dict[str, object]:
|
||||
return {
|
||||
"id": "t",
|
||||
"call_type": "completion",
|
||||
"response_cost": 0.0,
|
||||
"status": "success",
|
||||
"total_tokens": 30,
|
||||
"prompt_tokens": 20,
|
||||
"completion_tokens": 10,
|
||||
"startTime": 1.0,
|
||||
"endTime": 2.0,
|
||||
"completionStartTime": 1.5,
|
||||
"model": "openai/gpt-5.4-nano",
|
||||
"model_id": "dep-1",
|
||||
"model_group": "per-second-priced-chat",
|
||||
"api_base": "https://api.openai.com",
|
||||
"custom_llm_provider": "openai",
|
||||
"request_tags": [],
|
||||
"end_user": None,
|
||||
"cache_hit": False,
|
||||
"stream": False,
|
||||
"response": {"id": "chatcmpl-1"},
|
||||
"model_parameters": {},
|
||||
"zero_cost_diagnostic": zero_cost_diagnostic,
|
||||
"metadata": {
|
||||
"user_api_key_hash": "h",
|
||||
"user_api_key_alias": "a",
|
||||
"user_api_key_team_id": "t",
|
||||
"user_api_key_team_alias": "ta",
|
||||
"user_api_key_user_id": "u",
|
||||
"user_api_key_user_email": "e@x.com",
|
||||
"user_api_key_org_id": None,
|
||||
"user_api_key_org_alias": None,
|
||||
"requester_metadata": None,
|
||||
"user_api_key_end_user_id": None,
|
||||
"usage_object": None,
|
||||
},
|
||||
"hidden_params": {"litellm_overhead_time_ms": None, "additional_headers": None},
|
||||
}
|
||||
|
||||
|
||||
async def _log_success(
|
||||
logger: PrometheusLogger, zero_cost_diagnostic: StandardLoggingZeroCostDiagnostic | None
|
||||
) -> None:
|
||||
now: Final = datetime.datetime.now()
|
||||
kwargs: Final = {
|
||||
"model": "openai/gpt-5.4-nano",
|
||||
"litellm_params": {"metadata": {}},
|
||||
"standard_logging_object": _payload(zero_cost_diagnostic),
|
||||
"stream": False,
|
||||
"start_time": now - datetime.timedelta(seconds=3),
|
||||
"api_call_start_time": now - datetime.timedelta(seconds=2),
|
||||
"completion_start_time": now - datetime.timedelta(seconds=1),
|
||||
"end_time": now,
|
||||
}
|
||||
await logger.async_log_success_event(kwargs, None, now, now)
|
||||
|
||||
|
||||
async def _log_failure(
|
||||
logger: PrometheusLogger, zero_cost_diagnostic: StandardLoggingZeroCostDiagnostic | None
|
||||
) -> None:
|
||||
now: Final = datetime.datetime.now()
|
||||
kwargs: Final = {
|
||||
"model": "openai/gpt-5.4-nano",
|
||||
"litellm_params": {"metadata": {}},
|
||||
"standard_logging_object": {**_payload(zero_cost_diagnostic), "status": "failure"},
|
||||
"exception": Exception("stream cut off after the usage chunk"),
|
||||
"stream": True,
|
||||
"start_time": now - datetime.timedelta(seconds=3),
|
||||
"end_time": now,
|
||||
}
|
||||
await logger.async_log_failure_event(kwargs, None, now, now)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_failure_event_counts_a_zero_cost_request_by_model_and_reason() -> None:
|
||||
_clear_prometheus_registry()
|
||||
try:
|
||||
logger: Final = PrometheusLogger()
|
||||
await _log_failure(logger, None)
|
||||
assert _samples(METRIC) == []
|
||||
|
||||
await _log_failure(logger, MISSING_KEY_DIAGNOSTIC)
|
||||
|
||||
samples: Final = _samples(METRIC)
|
||||
assert len(samples) == 1
|
||||
assert samples[0].labels == {
|
||||
"requested_model": "per-second-priced-chat",
|
||||
"model": "openai/gpt-5.4-nano",
|
||||
"model_id": "dep-1",
|
||||
"api_provider": "openai",
|
||||
"reason": "missing_pricing_key",
|
||||
}
|
||||
assert samples[0].value == 1.0
|
||||
finally:
|
||||
_clear_prometheus_registry()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_success_event_counts_a_zero_cost_request_by_model_and_reason() -> None:
|
||||
_clear_prometheus_registry()
|
||||
try:
|
||||
logger: Final = PrometheusLogger()
|
||||
await _log_success(logger, MISSING_KEY_DIAGNOSTIC)
|
||||
await _log_success(logger, MISSING_KEY_DIAGNOSTIC)
|
||||
|
||||
samples: Final = _samples(METRIC)
|
||||
assert len(samples) == 1
|
||||
assert samples[0].labels == {
|
||||
"requested_model": "per-second-priced-chat",
|
||||
"model": "openai/gpt-5.4-nano",
|
||||
"model_id": "dep-1",
|
||||
"api_provider": "openai",
|
||||
"reason": "missing_pricing_key",
|
||||
}
|
||||
assert samples[0].value == 2.0
|
||||
finally:
|
||||
_clear_prometheus_registry()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_request_without_a_diagnostic_leaves_the_counter_untouched() -> None:
|
||||
_clear_prometheus_registry()
|
||||
try:
|
||||
await _log_success(PrometheusLogger(), None)
|
||||
|
||||
assert _samples(METRIC) == []
|
||||
finally:
|
||||
_clear_prometheus_registry()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_label_filter_that_drops_reason_still_counts_the_request() -> None:
|
||||
_clear_prometheus_registry()
|
||||
previous_config: Final = litellm.prometheus_metrics_config
|
||||
litellm.prometheus_metrics_config = [
|
||||
{"group": "zero_cost", "metrics": [METRIC], "include_labels": ["requested_model"]}
|
||||
]
|
||||
try:
|
||||
await _log_success(PrometheusLogger(), MISSING_KEY_DIAGNOSTIC)
|
||||
|
||||
samples: Final = _samples(METRIC)
|
||||
assert len(samples) == 1
|
||||
assert samples[0].labels == {"requested_model": "per-second-priced-chat"}
|
||||
assert samples[0].value == 1.0
|
||||
finally:
|
||||
litellm.prometheus_metrics_config = previous_config
|
||||
_clear_prometheus_registry()
|
||||
|
|
@ -0,0 +1,157 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.litellm_core_utils.llm_cost_calc.zero_cost_diagnostic import (
|
||||
ZERO_COST_COUNTER_NAME,
|
||||
diagnose_zero_cost,
|
||||
used_pricing_keys,
|
||||
zero_cost_warning,
|
||||
)
|
||||
from litellm.types.utils import CompletionTokensDetailsWrapper, PromptTokensDetailsWrapper, Usage
|
||||
|
||||
PER_SECOND_ENTRY: Final = {"input_cost_per_second": 0.00042, "output_cost_per_second": 0.00042}
|
||||
FREE_ENTRY: Final = {"input_cost_per_token": 0, "output_cost_per_token": 0, "cache_read_input_token_cost": 2e-08}
|
||||
PRICED_ENTRY: Final = {"input_cost_per_token": 1e-06, "output_cost_per_token": 2e-06}
|
||||
TEXT_USAGE: Final = Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
|
||||
|
||||
def test_missing_pricing_key_names_every_rate_the_usage_needs() -> None:
|
||||
diagnostic = diagnose_zero_cost(
|
||||
usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry=PER_SECOND_ENTRY, calculation_failed=False
|
||||
)
|
||||
|
||||
assert diagnostic == {
|
||||
"reason": "missing_pricing_key",
|
||||
"pricing_model": "dep-1",
|
||||
"missing_pricing_keys": ("input_cost_per_token", "output_cost_per_token"),
|
||||
}
|
||||
|
||||
|
||||
def test_only_the_absent_rate_is_reported() -> None:
|
||||
diagnostic = diagnose_zero_cost(
|
||||
usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry={"input_cost_per_token": 1e-06}, calculation_failed=False
|
||||
)
|
||||
|
||||
assert diagnostic is not None
|
||||
assert diagnostic["missing_pricing_keys"] == ("output_cost_per_token",)
|
||||
|
||||
|
||||
def test_free_model_stays_silent() -> None:
|
||||
assert (
|
||||
diagnose_zero_cost(usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry=FREE_ENTRY, calculation_failed=False)
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("calculation_failed", [False, True])
|
||||
def test_request_without_usage_stays_silent(calculation_failed: bool) -> None:
|
||||
usage = Usage(prompt_tokens=0, completion_tokens=0, total_tokens=0)
|
||||
|
||||
assert (
|
||||
diagnose_zero_cost(
|
||||
usage=usage, pricing_model="dep-1", pricing_entry=PER_SECOND_ENTRY, calculation_failed=calculation_failed
|
||||
)
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"entry",
|
||||
[
|
||||
{"litellm_provider": "openai", "mode": "chat", "supports_prompt_caching": True},
|
||||
{"tiered_pricing": [{"range": [0, 128000], "input_cost_per_token": 0, "output_cost_per_token": 0}]},
|
||||
{"tiered_pricing": "not a tier table", "litellm_provider": "openai"},
|
||||
],
|
||||
)
|
||||
def test_entry_that_declares_no_rate_stays_silent(entry: Mapping[str, object]) -> None:
|
||||
assert (
|
||||
diagnose_zero_cost(usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry=entry, calculation_failed=False)
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
def test_tiered_rate_counts_as_a_declared_rate() -> None:
|
||||
entry = {"tiered_pricing": [{"range": [0, 128000], "input_cost_per_token": 1e-06, "output_cost_per_token": 2e-06}]}
|
||||
|
||||
diagnostic = diagnose_zero_cost(
|
||||
usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry=entry, calculation_failed=False
|
||||
)
|
||||
|
||||
assert diagnostic is not None
|
||||
assert diagnostic["reason"] == "missing_pricing_key"
|
||||
|
||||
|
||||
def test_priced_entry_that_still_prices_to_zero_is_pricing_not_applied() -> None:
|
||||
diagnostic = diagnose_zero_cost(
|
||||
usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry=PRICED_ENTRY, calculation_failed=False
|
||||
)
|
||||
|
||||
assert diagnostic == {"reason": "pricing_not_applied", "pricing_model": "dep-1", "missing_pricing_keys": ()}
|
||||
|
||||
|
||||
def test_calculator_failure_on_a_priced_entry_is_cost_calculation_error() -> None:
|
||||
diagnostic = diagnose_zero_cost(
|
||||
usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry=PRICED_ENTRY, calculation_failed=True
|
||||
)
|
||||
|
||||
assert diagnostic == {"reason": "cost_calculation_error", "pricing_model": "dep-1", "missing_pricing_keys": ()}
|
||||
|
||||
|
||||
def test_calculator_failure_on_a_free_entry_stays_silent() -> None:
|
||||
assert (
|
||||
diagnose_zero_cost(usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry=FREE_ENTRY, calculation_failed=True)
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
def test_calculator_failure_on_an_entry_that_declares_no_rate_stays_silent() -> None:
|
||||
entry: Final = {"litellm_provider": "openai", "mode": "chat", "supports_prompt_caching": True}
|
||||
assert (
|
||||
diagnose_zero_cost(usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry=entry, calculation_failed=True)
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
def test_audio_tokens_need_the_audio_rates() -> None:
|
||||
usage = Usage(
|
||||
prompt_tokens=10,
|
||||
completion_tokens=20,
|
||||
total_tokens=30,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(audio_tokens=10, text_tokens=0),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(audio_tokens=5, text_tokens=15),
|
||||
)
|
||||
|
||||
assert used_pricing_keys(usage) == (
|
||||
"input_cost_per_audio_token",
|
||||
"output_cost_per_token",
|
||||
"output_cost_per_audio_token",
|
||||
)
|
||||
diagnostic = diagnose_zero_cost(
|
||||
usage=usage, pricing_model="gemini-audio", pricing_entry=PRICED_ENTRY, calculation_failed=False
|
||||
)
|
||||
assert diagnostic is not None
|
||||
assert diagnostic["missing_pricing_keys"] == ("input_cost_per_audio_token", "output_cost_per_audio_token")
|
||||
|
||||
|
||||
def test_warning_names_the_request_the_entry_the_missing_keys_and_the_counter() -> None:
|
||||
diagnostic = diagnose_zero_cost(
|
||||
usage=TEXT_USAGE, pricing_model="dep-1", pricing_entry=PER_SECOND_ENTRY, calculation_failed=False
|
||||
)
|
||||
assert diagnostic is not None
|
||||
|
||||
message = zero_cost_warning(
|
||||
diagnostic,
|
||||
model_group="per-second-priced-chat",
|
||||
model="openai/gpt-5.4-nano",
|
||||
custom_llm_provider="openai",
|
||||
usage=TEXT_USAGE,
|
||||
)
|
||||
|
||||
assert "model_group=per-second-priced-chat" in message
|
||||
assert "model=openai/gpt-5.4-nano" in message
|
||||
assert "provider=openai" in message
|
||||
assert "prompt_tokens=10 completion_tokens=20" in message
|
||||
assert "pricing entry 'dep-1' has no input_cost_per_token, output_cost_per_token" in message
|
||||
assert f'{ZERO_COST_COUNTER_NAME}{{reason="missing_pricing_key"}}' in message
|
||||
|
|
@ -1,9 +1,10 @@
|
|||
import asyncio
|
||||
import contextlib
|
||||
import datetime
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from collections.abc import Callable
|
||||
from collections.abc import Callable, Iterator, Mapping
|
||||
from typing import Final, Literal
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
|
|
@ -302,6 +303,398 @@ def test_response_cost_calculator_uses_router_model_id_from_litellm_metadata():
|
|||
litellm.model_cost.pop(custom_model_id, None)
|
||||
|
||||
|
||||
class TestZeroCostDiagnostic:
|
||||
DEPLOYMENT_ID: Final = "lit7898-per-second-priced-deployment"
|
||||
MODEL_GROUP: Final = "per-second-priced-chat"
|
||||
PER_SECOND_PRICING: Final = {"input_cost_per_second": 0.00042, "output_cost_per_second": 0.00042}
|
||||
FREE_PRICING: Final = {"input_cost_per_token": 0, "output_cost_per_token": 0}
|
||||
|
||||
@pytest.fixture(params=["per_second", "free"])
|
||||
def deployment_pricing(self, request: pytest.FixtureRequest) -> Iterator[Mapping[str, float]]:
|
||||
pricing: Final = self.PER_SECOND_PRICING if request.param == "per_second" else self.FREE_PRICING
|
||||
litellm.register_model(model_cost={self.DEPLOYMENT_ID: pricing}, persist_across_reloads=False)
|
||||
try:
|
||||
yield pricing
|
||||
finally:
|
||||
litellm.model_cost.pop(self.DEPLOYMENT_ID, None)
|
||||
|
||||
def _logging_obj(
|
||||
self,
|
||||
pricing: Mapping[str, object],
|
||||
stream: bool = False,
|
||||
model: str = "openai/gpt-5.4-nano",
|
||||
call_type: str = "completion",
|
||||
deployment_id: str | None = DEPLOYMENT_ID,
|
||||
custom_llm_provider: str = "openai",
|
||||
) -> LitellmLogging:
|
||||
logging_obj: Final = LitellmLogging(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
stream=stream,
|
||||
call_type=call_type,
|
||||
start_time=time.time(),
|
||||
litellm_call_id="lit7898",
|
||||
function_id="fn",
|
||||
)
|
||||
self._route_to_deployment(
|
||||
logging_obj, pricing, model=model, deployment_id=deployment_id, custom_llm_provider=custom_llm_provider
|
||||
)
|
||||
return logging_obj
|
||||
|
||||
def _route_to_deployment(
|
||||
self,
|
||||
logging_obj: LitellmLogging,
|
||||
pricing: Mapping[str, object],
|
||||
model: str = "openai/gpt-5.4-nano",
|
||||
deployment_id: str | None = DEPLOYMENT_ID,
|
||||
custom_llm_provider: str = "openai",
|
||||
) -> None:
|
||||
model_info: Final = pricing if deployment_id is None else {"id": deployment_id, **pricing}
|
||||
logging_obj.update_environment_variables(
|
||||
model=model,
|
||||
user="",
|
||||
optional_params={},
|
||||
litellm_params={"metadata": {"model_group": self.MODEL_GROUP, "model_info": model_info}},
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _response(
|
||||
usage: litellm.Usage | None = None, model: str = "gpt-5.4-nano", **hidden_params: object
|
||||
) -> ModelResponse:
|
||||
response: Final = ModelResponse(
|
||||
model=model,
|
||||
choices=[litellm.Choices(message=litellm.Message(role="assistant", content="hello"))],
|
||||
usage=usage,
|
||||
)
|
||||
response._hidden_params = {"custom_llm_provider": "openai", **hidden_params}
|
||||
return response
|
||||
|
||||
@staticmethod
|
||||
def _zero_cost_warnings(caplog: pytest.LogCaptureFixture) -> list[str]:
|
||||
return [
|
||||
record.getMessage()
|
||||
for record in caplog.records
|
||||
if record.name == "LiteLLM" and record.levelno == logging.WARNING and "priced at $0" in record.getMessage()
|
||||
]
|
||||
|
||||
def _assert_flagged(self, logging_obj: LitellmLogging, caplog: pytest.LogCaptureFixture) -> None:
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] == {
|
||||
"reason": "missing_pricing_key",
|
||||
"pricing_model": self.DEPLOYMENT_ID,
|
||||
"missing_pricing_keys": ("input_cost_per_token", "output_cost_per_token"),
|
||||
}
|
||||
warnings: Final = self._zero_cost_warnings(caplog)
|
||||
assert len(warnings) == 1
|
||||
assert f"model_group={self.MODEL_GROUP}" in warnings[0]
|
||||
assert f"pricing entry '{self.DEPLOYMENT_ID}' has no input_cost_per_token, output_cost_per_token" in warnings[0]
|
||||
|
||||
def test_zero_cost_with_a_missing_rate_warns_once_and_is_recorded(
|
||||
self, deployment_pricing: Mapping[str, float], caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
logging_obj: Final = self._logging_obj(deployment_pricing)
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
first_cost: Final = logging_obj._response_cost_calculator(result=self._response(usage))
|
||||
second_cost: Final = logging_obj._response_cost_calculator(result=self._response(usage))
|
||||
|
||||
assert first_cost == 0.0
|
||||
assert second_cost == 0.0
|
||||
if deployment_pricing is self.FREE_PRICING:
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
return
|
||||
self._assert_flagged(logging_obj, caplog)
|
||||
|
||||
def test_usage_less_stream_chunk_does_not_hide_the_final_response_diagnostic(
|
||||
self, deployment_pricing: Mapping[str, float], caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
usage: Final = litellm.Usage(prompt_tokens=8, completion_tokens=2, total_tokens=10)
|
||||
logging_obj: Final = self._logging_obj(deployment_pricing, stream=True)
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
logging_obj._response_cost_calculator(result=self._response(usage=None))
|
||||
logging_obj._response_cost_calculator(result=self._response(usage))
|
||||
|
||||
if deployment_pricing is self.FREE_PRICING:
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
return
|
||||
self._assert_flagged(logging_obj, caplog)
|
||||
|
||||
def test_terminal_responses_stream_event_is_judged_by_its_inner_response(
|
||||
self, deployment_pricing: Mapping[str, float], caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
logging_obj: Final = self._logging_obj(deployment_pricing, stream=True, call_type="aresponses")
|
||||
event: Final = ResponseCompletedEvent(
|
||||
type="response.completed",
|
||||
response=ResponsesAPIResponse(
|
||||
id="resp-lit7898",
|
||||
created_at=1,
|
||||
object="response",
|
||||
status="completed",
|
||||
model="gpt-5.4-nano",
|
||||
output=[],
|
||||
usage=ResponseAPIUsage(input_tokens=10, output_tokens=20, total_tokens=30),
|
||||
),
|
||||
)
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
cost: Final = logging_obj._response_cost_calculator(result=event)
|
||||
|
||||
assert cost == 0.0
|
||||
if deployment_pricing is self.FREE_PRICING:
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
return
|
||||
self._assert_flagged(logging_obj, caplog)
|
||||
|
||||
def test_precomputed_zero_hidden_cost_is_flagged_and_lands_in_the_payload(
|
||||
self, deployment_pricing: Mapping[str, float], caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
logging_obj: Final = self._logging_obj(deployment_pricing)
|
||||
response: Final = self._response(usage, response_cost=0.0, model_id=self.DEPLOYMENT_ID)
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
logging_obj._process_hidden_params_and_response_cost(
|
||||
response, start_time=datetime.datetime.now(), end_time=datetime.datetime.now()
|
||||
)
|
||||
|
||||
payload: Final = logging_obj.model_call_details["standard_logging_object"]
|
||||
assert payload["response_cost"] == 0.0
|
||||
if deployment_pricing is self.FREE_PRICING:
|
||||
assert payload["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
return
|
||||
self._assert_flagged(logging_obj, caplog)
|
||||
assert payload["zero_cost_diagnostic"] == logging_obj.model_call_details["zero_cost_diagnostic"]
|
||||
|
||||
def test_uncomputed_hidden_cost_is_not_a_zero_cost(
|
||||
self, deployment_pricing: Mapping[str, float], caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
logging_obj: Final = self._logging_obj(deployment_pricing)
|
||||
response: Final = self._response(usage, response_cost=None, model_id=self.DEPLOYMENT_ID)
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
logging_obj._process_hidden_params_and_response_cost(
|
||||
response, start_time=datetime.datetime.now(), end_time=datetime.datetime.now()
|
||||
)
|
||||
|
||||
assert logging_obj.model_call_details["standard_logging_object"]["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
|
||||
def test_unbilled_read_route_with_usage_stays_silent(
|
||||
self, deployment_pricing: Mapping[str, float], caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
logging_obj: Final = self._logging_obj(deployment_pricing, call_type="aget_responses")
|
||||
response: Final = self._response(usage, response_cost=0.0, model_id=self.DEPLOYMENT_ID)
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
logging_obj._process_hidden_params_and_response_cost(
|
||||
response, start_time=datetime.datetime.now(), end_time=datetime.datetime.now()
|
||||
)
|
||||
|
||||
assert logging_obj.model_call_details["standard_logging_object"]["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
|
||||
def test_unmapped_model_that_fails_cost_calculation_stays_silent(self, caplog: pytest.LogCaptureFixture) -> None:
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
logging_obj: Final = self._logging_obj(
|
||||
{}, model="openai/lit7898-unmapped-model", deployment_id="lit7898-unmapped-deployment"
|
||||
)
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
cost: Final = logging_obj._response_cost_calculator(
|
||||
result=self._response(usage, model="lit7898-unmapped-model")
|
||||
)
|
||||
|
||||
assert cost is None
|
||||
assert logging_obj.model_call_details["response_cost_failure_debug_information"] is not None
|
||||
assert logging_obj.model_call_details.get("zero_cost_diagnostic") is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
|
||||
def test_malformed_usage_never_raises_out_of_the_cost_calculator(
|
||||
self, deployment_pricing: Mapping[str, float], caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
logging_obj: Final = self._logging_obj(deployment_pricing)
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
cost: Final = logging_obj._response_cost_calculator(
|
||||
result={"model": "gpt-5.4-nano", "usage": {"prompt_tokens": "n/a", "completion_tokens": 3}}
|
||||
)
|
||||
|
||||
assert cost is None
|
||||
assert logging_obj.model_call_details.get("zero_cost_diagnostic") is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
|
||||
def test_usage_less_evaluation_between_two_zero_cost_findings_does_not_warn_twice(
|
||||
self, deployment_pricing: Mapping[str, float], caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
usage: Final = litellm.Usage(prompt_tokens=8, completion_tokens=2, total_tokens=10)
|
||||
logging_obj: Final = self._logging_obj(deployment_pricing, stream=True, call_type="anthropic_messages")
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
logging_obj._response_cost_calculator(result=self._response(usage=None))
|
||||
logging_obj._response_cost_calculator(result=self._response(usage))
|
||||
logging_obj._response_cost_calculator(result=self._response(usage=None))
|
||||
logging_obj._response_cost_calculator(result=self._response(usage))
|
||||
|
||||
if deployment_pricing is self.FREE_PRICING:
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
return
|
||||
self._assert_flagged(logging_obj, caplog)
|
||||
|
||||
def test_retry_that_prices_clears_the_diagnostic_and_a_later_zero_cost_is_recorded_silently(
|
||||
self, caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
priced_id: Final = "lit7898-priced-deployment"
|
||||
priced_pricing: Final = {"input_cost_per_token": 1e-06, "output_cost_per_token": 2e-06}
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
litellm.register_model(
|
||||
model_cost={self.DEPLOYMENT_ID: self.PER_SECOND_PRICING, priced_id: priced_pricing},
|
||||
persist_across_reloads=False,
|
||||
)
|
||||
try:
|
||||
logging_obj: Final = self._logging_obj(self.PER_SECOND_PRICING)
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
assert logging_obj._response_cost_calculator(result=self._response(usage)) == 0.0
|
||||
self._assert_flagged(logging_obj, caplog)
|
||||
|
||||
self._route_to_deployment(logging_obj, priced_pricing, deployment_id=priced_id)
|
||||
assert logging_obj._response_cost_calculator(result=self._response(usage)) == pytest.approx(5e-05)
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
|
||||
|
||||
self._route_to_deployment(logging_obj, self.PER_SECOND_PRICING)
|
||||
assert logging_obj._response_cost_calculator(result=self._response(usage)) == 0.0
|
||||
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"]["reason"] == "missing_pricing_key"
|
||||
assert len(self._zero_cost_warnings(caplog)) == 1
|
||||
finally:
|
||||
litellm.model_cost.pop(self.DEPLOYMENT_ID, None)
|
||||
litellm.model_cost.pop(priced_id, None)
|
||||
|
||||
def test_one_request_evaluated_against_two_cost_map_entries_warns_once(
|
||||
self, caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
dated_model: Final = "lit7898-nano-2026-03-17"
|
||||
requested_model: Final = "lit7898-nano"
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
cost_map_entry: Final = {"litellm_provider": "openai", "mode": "chat", **self.PER_SECOND_PRICING}
|
||||
litellm.register_model(
|
||||
model_cost={dated_model: cost_map_entry, requested_model: cost_map_entry}, persist_across_reloads=False
|
||||
)
|
||||
try:
|
||||
logging_obj: Final = self._logging_obj(
|
||||
{}, model=f"openai/{requested_model}", deployment_id="lit7898-cost-map-deployment"
|
||||
)
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
assert logging_obj._response_cost_calculator(result=self._response(usage, model=dated_model)) == 0.0
|
||||
assert logging_obj._response_cost_calculator(result=self._response(usage, model=requested_model)) == 0.0
|
||||
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"]["pricing_model"] == requested_model
|
||||
warnings: Final = self._zero_cost_warnings(caplog)
|
||||
assert len(warnings) == 1
|
||||
assert f"pricing entry '{dated_model}' has no input_cost_per_token, output_cost_per_token" in warnings[0]
|
||||
finally:
|
||||
litellm.model_cost.pop(dated_model, None)
|
||||
litellm.model_cost.pop(requested_model, None)
|
||||
|
||||
def test_free_deployment_without_a_router_id_is_judged_by_its_own_pricing(
|
||||
self, caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
global_model: Final = "lit7898-priced-global"
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
litellm.register_model(
|
||||
model_cost={
|
||||
global_model: {
|
||||
"litellm_provider": "openai",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1e-06,
|
||||
"output_cost_per_token": 2e-06,
|
||||
}
|
||||
},
|
||||
persist_across_reloads=False,
|
||||
)
|
||||
try:
|
||||
logging_obj: Final = self._logging_obj(self.FREE_PRICING, model=global_model, deployment_id=None)
|
||||
response: Final = self._response(usage, model=global_model, response_cost=0.0)
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
logging_obj._process_hidden_params_and_response_cost(
|
||||
response, start_time=datetime.datetime.now(), end_time=datetime.datetime.now()
|
||||
)
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
finally:
|
||||
litellm.model_cost.pop(global_model, None)
|
||||
|
||||
def test_cache_hit_priced_for_saved_cost_stays_silent(
|
||||
self, deployment_pricing: Mapping[str, float], caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
logging_obj: Final = self._logging_obj(deployment_pricing)
|
||||
logging_obj.model_call_details["cache_hit"] = True
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
assert logging_obj._response_cost_calculator(result=self._response(usage), cache_hit=False) == 0.0
|
||||
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
|
||||
@pytest.mark.parametrize("spilled_over", [True, False])
|
||||
def test_ptu_deployment_is_judged_by_the_entry_the_calculator_priced_with(
|
||||
self, spilled_over: bool, monkeypatch: pytest.MonkeyPatch, caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
router_model_id: Final = "lit7898-ptu-router-model-id"
|
||||
served_model: Final = "azure/lit7898-ptu-served-model"
|
||||
ptu_model_info: Final = {
|
||||
"team_id": "team-1",
|
||||
"ptu_count": 100,
|
||||
"cost_per_ptu_per_hour": 1.0,
|
||||
"ptu_effective_from": "2026-01-01",
|
||||
**self.FREE_PRICING,
|
||||
}
|
||||
usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
litellm.register_model(
|
||||
model_cost={
|
||||
router_model_id: {**self.FREE_PRICING, "litellm_provider": "azure", "mode": "chat"},
|
||||
served_model: {**self.PER_SECOND_PRICING, "litellm_provider": "azure", "mode": "chat"},
|
||||
},
|
||||
persist_across_reloads=False,
|
||||
)
|
||||
monkeypatch.setenv("LITELLM_ENABLE_PTU_COST_ATTRIBUTION", "True")
|
||||
try:
|
||||
logging_obj: Final = self._logging_obj(
|
||||
ptu_model_info, model=served_model, deployment_id=router_model_id, custom_llm_provider="azure"
|
||||
)
|
||||
spillover_headers: Final = {"llm_provider-x-ms-is-spilled-over": "true"} if spilled_over else {}
|
||||
response: Final = self._response(
|
||||
usage, model=served_model, custom_llm_provider="azure", additional_headers=spillover_headers
|
||||
)
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
assert logging_obj._response_cost_calculator(result=response) == 0.0
|
||||
|
||||
warnings: Final = self._zero_cost_warnings(caplog)
|
||||
if not spilled_over:
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
|
||||
assert warnings == []
|
||||
return
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] == {
|
||||
"reason": "missing_pricing_key",
|
||||
"pricing_model": served_model,
|
||||
"missing_pricing_keys": ("input_cost_per_token", "output_cost_per_token"),
|
||||
}
|
||||
assert len(warnings) == 1
|
||||
assert f"pricing entry '{served_model}' has no input_cost_per_token, output_cost_per_token" in warnings[0]
|
||||
finally:
|
||||
litellm.model_cost.pop(router_model_id, None)
|
||||
litellm.model_cost.pop(served_model, None)
|
||||
|
||||
|
||||
class TestGetRouterModelId:
|
||||
"""Tests for the get_router_model_id helper method."""
|
||||
|
||||
|
|
@ -407,7 +800,6 @@ class TestGetRouterDeploymentModelInfo:
|
|||
logging_obj.litellm_params = {"api_base": ""}
|
||||
assert logging_obj.get_router_deployment_model_info() is None
|
||||
|
||||
|
||||
def test_a_published_batch_rate_never_displaces_a_declared_standard_rate(self) -> None:
|
||||
"""Ownership is per token direction, not per field.
|
||||
|
||||
|
|
@ -1111,7 +1503,9 @@ async def test_arealtime_marks_litellm_params_async(monkeypatch):
|
|||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_aresponses_websocket_hands_back_the_provider_failure_without_a_success_log(monkeypatch: pytest.MonkeyPatch):
|
||||
async def test_aresponses_websocket_hands_back_the_provider_failure_without_a_success_log(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
):
|
||||
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
|
||||
from litellm.responses.main import base_llm_http_handler
|
||||
|
||||
|
|
@ -7066,22 +7460,41 @@ async def test_classifier_audit_matches_provider_transport(provider: str) -> Non
|
|||
|
||||
return httpx.Response(200, json=mock_responses_api_response(content).model_dump())
|
||||
if provider == "anthropic":
|
||||
return httpx.Response(200, json={
|
||||
"id": "msg-audit", "type": "message", "role": "assistant", "model": "claude-haiku-4-5",
|
||||
"content": [{"type": "text", "text": content}], "stop_reason": "end_turn",
|
||||
"usage": {"input_tokens": 10, "output_tokens": 5},
|
||||
})
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"id": "msg-audit",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "claude-haiku-4-5",
|
||||
"content": [{"type": "text", "text": content}],
|
||||
"stop_reason": "end_turn",
|
||||
"usage": {"input_tokens": 10, "output_tokens": 5},
|
||||
},
|
||||
)
|
||||
if provider == "bedrock":
|
||||
return httpx.Response(200, json={
|
||||
"output": {"message": {"role": "assistant", "content": [{"text": content}]}},
|
||||
"stopReason": "end_turn", "usage": {"inputTokens": 10, "outputTokens": 5, "totalTokens": 15},
|
||||
"metrics": {"latencyMs": 1},
|
||||
})
|
||||
return httpx.Response(200, json={
|
||||
"id": "chatcmpl-audit", "object": "chat.completion", "created": 0, "model": "gpt-5.6",
|
||||
"choices": [{"index": 0, "message": {"role": "assistant", "content": content}, "finish_reason": "stop"}],
|
||||
"usage": {"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15},
|
||||
})
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"output": {"message": {"role": "assistant", "content": [{"text": content}]}},
|
||||
"stopReason": "end_turn",
|
||||
"usage": {"inputTokens": 10, "outputTokens": 5, "totalTokens": 15},
|
||||
"metrics": {"latencyMs": 1},
|
||||
},
|
||||
)
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"id": "chatcmpl-audit",
|
||||
"object": "chat.completion",
|
||||
"created": 0,
|
||||
"model": "gpt-5.6",
|
||||
"choices": [
|
||||
{"index": 0, "message": {"role": "assistant", "content": content}, "finish_reason": "stop"}
|
||||
],
|
||||
"usage": {"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15},
|
||||
},
|
||||
)
|
||||
|
||||
async def capture(kwargs, response_obj, start_time, end_time):
|
||||
logs.put_nowait(kwargs["standard_logging_object"])
|
||||
|
|
@ -7092,11 +7505,15 @@ async def test_classifier_audit_matches_provider_transport(provider: str) -> Non
|
|||
handler.client = http_client
|
||||
client: Final = (
|
||||
AsyncAzureOpenAI(
|
||||
api_key="transport-only", azure_endpoint="https://azure.invalid",
|
||||
api_version="2025-04-01-preview", http_client=http_client,
|
||||
api_key="transport-only",
|
||||
azure_endpoint="https://azure.invalid",
|
||||
api_version="2025-04-01-preview",
|
||||
http_client=http_client,
|
||||
)
|
||||
if provider == "azure" else AsyncOpenAI(api_key="transport-only", http_client=http_client)
|
||||
if provider == "openai" else handler
|
||||
if provider == "azure"
|
||||
else AsyncOpenAI(api_key="transport-only", http_client=http_client)
|
||||
if provider == "openai"
|
||||
else handler
|
||||
)
|
||||
model: Final = {
|
||||
"openai": "openai/gpt-5.6",
|
||||
|
|
@ -7109,23 +7526,44 @@ async def test_classifier_audit_matches_provider_transport(provider: str) -> Non
|
|||
async def run(marker: str) -> None:
|
||||
if provider == "responses":
|
||||
await litellm.aresponses(
|
||||
model=model, api_key="transport-only", client=client, max_output_tokens=128,
|
||||
instructions="classifier-rubric", input=marker,
|
||||
model=model,
|
||||
api_key="transport-only",
|
||||
client=client,
|
||||
max_output_tokens=128,
|
||||
instructions="classifier-rubric",
|
||||
input=marker,
|
||||
metadata={"internal_call_origin": "autorouter_classifier"},
|
||||
proxy_server_request={"body": {}, "originating_request_masked": {"input": f"source-only-{marker}"}},
|
||||
success_callback=[capture], num_retries=0,
|
||||
success_callback=[capture],
|
||||
num_retries=0,
|
||||
)
|
||||
return
|
||||
await litellm.acompletion(
|
||||
model=model, api_key="transport-only", client=client, max_tokens=128,
|
||||
aws_access_key_id="transport-only", aws_secret_access_key="transport-only", aws_region_name="us-east-1",
|
||||
model=model,
|
||||
api_key="transport-only",
|
||||
client=client,
|
||||
max_tokens=128,
|
||||
aws_access_key_id="transport-only",
|
||||
aws_secret_access_key="transport-only",
|
||||
aws_region_name="us-east-1",
|
||||
messages=[{"role": "system", "content": "classifier-rubric"}, {"role": "user", "content": marker}],
|
||||
metadata={"internal_call_origin": "autorouter_classifier"},
|
||||
proxy_server_request={"body": {}, "originating_request_masked": {"input": f"source-only-{marker}"}},
|
||||
success_callback=[capture], num_retries=0,
|
||||
**({"api_base": "https://azure.invalid", "api_version": "2025-04-01-preview"} if provider == "azure" else {}),
|
||||
**({"extra_body": {"audit_context": "provider-extra"}, "extra_headers": {"X-Audit": "header-only-secret"}}
|
||||
if provider in ("openai", "azure") else {}),
|
||||
success_callback=[capture],
|
||||
num_retries=0,
|
||||
**(
|
||||
{"api_base": "https://azure.invalid", "api_version": "2025-04-01-preview"}
|
||||
if provider == "azure"
|
||||
else {}
|
||||
),
|
||||
**(
|
||||
{
|
||||
"extra_body": {"audit_context": "provider-extra"},
|
||||
"extra_headers": {"X-Audit": "header-only-secret"},
|
||||
}
|
||||
if provider in ("openai", "azure")
|
||||
else {}
|
||||
),
|
||||
)
|
||||
|
||||
await asyncio.gather(run("request-one"), run("request-two"))
|
||||
|
|
@ -7149,14 +7587,17 @@ async def test_classifier_audit_matches_provider_transport(provider: str) -> Non
|
|||
@pytest.mark.parametrize("redaction", ["none", "global", "request", "header"])
|
||||
@pytest.mark.parametrize("status", ["success", "failure"])
|
||||
@pytest.mark.parametrize("call_type", ["completion", "acompletion", "responses", "aresponses"])
|
||||
def test_classifier_audit_obeys_message_logging_before_payload_emission(logging_obj, monkeypatch, redaction, status, call_type):
|
||||
def test_classifier_audit_obeys_message_logging_before_payload_emission(
|
||||
logging_obj, monkeypatch, redaction, status, call_type
|
||||
):
|
||||
from litellm.litellm_core_utils.litellm_logging import get_standard_logging_object_payload
|
||||
|
||||
monkeypatch.setattr(litellm, "turn_off_message_logging", redaction == "global")
|
||||
params: Final = {
|
||||
"metadata": {"internal_call_origin": "autorouter_classifier", **(
|
||||
{"headers": {"x-litellm-enable-message-redaction": "true"}} if redaction == "header" else {}
|
||||
)},
|
||||
"metadata": {
|
||||
"internal_call_origin": "autorouter_classifier",
|
||||
**({"headers": {"x-litellm-enable-message-redaction": "true"}} if redaction == "header" else {}),
|
||||
},
|
||||
"proxy_server_request": {"body": {}, "originating_request_masked": {"input": "source-only"}},
|
||||
}
|
||||
logging_obj.call_type = call_type
|
||||
|
|
@ -7169,8 +7610,12 @@ def test_classifier_audit_obeys_message_logging_before_payload_emission(logging_
|
|||
)
|
||||
now: Final = datetime.datetime.now()
|
||||
payload: Final = get_standard_logging_object_payload(
|
||||
kwargs={**logging_obj.model_call_details, "call_type": call_type}, init_response_obj={},
|
||||
start_time=now, end_time=now, logging_obj=logging_obj, status=status,
|
||||
kwargs={**logging_obj.model_call_details, "call_type": call_type},
|
||||
init_response_obj={},
|
||||
start_time=now,
|
||||
end_time=now,
|
||||
logging_obj=logging_obj,
|
||||
status=status,
|
||||
)
|
||||
assert payload is not None
|
||||
if redaction == "none":
|
||||
|
|
@ -7421,7 +7866,13 @@ def _completed_responses_event(usage: ResponseAPIUsage) -> ResponseCompletedEven
|
|||
return ResponseCompletedEvent(
|
||||
type="response.completed",
|
||||
response=ResponsesAPIResponse(
|
||||
id="resp-1", created_at=1, object="response", status="completed", model="codex-mini-latest", output=[], usage=usage
|
||||
id="resp-1",
|
||||
created_at=1,
|
||||
object="response",
|
||||
status="completed",
|
||||
model="codex-mini-latest",
|
||||
output=[],
|
||||
usage=usage,
|
||||
),
|
||||
)
|
||||
|
||||
|
|
@ -7441,7 +7892,9 @@ def test_get_assembled_streaming_response_bills_a_provider_reported_usage_cost()
|
|||
now = datetime.datetime.now()
|
||||
|
||||
assembled = logging_obj._get_assembled_streaming_response(
|
||||
result=_completed_responses_event(ResponseAPIUsage(input_tokens=12, output_tokens=2, total_tokens=14, cost=0.0042)),
|
||||
result=_completed_responses_event(
|
||||
ResponseAPIUsage(input_tokens=12, output_tokens=2, total_tokens=14, cost=0.0042)
|
||||
),
|
||||
start_time=now,
|
||||
end_time=now,
|
||||
is_async=True,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue