From 8603d6e259fdcba31970b7b52e920282a57a194a Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 22:15:35 -0700 Subject: [PATCH] fix(cost): honor per-second custom pricing on chat completions for every provider (#42403) * fix(cost): honor per-second custom pricing on chat completions for every provider * test(cost): assert a per-second priced deployment bills instead of staying at $0 The zero-cost diagnostic tests from #42345 used a per-second-only entry as their misconfigured fixture, which this branch now bills. Switch that fixture to a per-query-only entry, which is still selected as the deployment's own pricing and still prices chat usage at $0, and add a per-second test asserting the call duration is billed with no diagnostic Also let a caller's explicit total_time outrank the logging window in completion_cost, so the SDK precedence stays stamped response, caller, logging * test(response_metadata): move the per-second pricing regression into the mapped tests/unit file * fix(cost_calculator): keep media-mode per-second rates off the wall-clock path A video, transcription, speech, or realtime entry's per-second rates price media seconds, which their dedicated cost paths bill from the media itself. The generic per-second branch now skips those modes, so a video status poll on a per-second video model bills nothing instead of the seconds the poll took to answer. --------- Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> --- litellm/cost_calculator.py | 119 +++++++++----- litellm/litellm_core_utils/litellm_logging.py | 18 ++- .../llm_response_utils/response_metadata.py | 2 +- .../test_litellm_logging.py | 51 ++++-- tests/test_litellm/test_cost_calculator.py | 151 +++++++++++++++++- .../test_response_metadata.py | 58 ++++++- 6 files changed, 336 insertions(+), 63 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 37743a9ce33..7b02a5ede00 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -115,6 +115,7 @@ from litellm.types.utils import ( LlmProviders, LlmProvidersSet, ModelInfo, + ModelInfoBase, PromptTokensDetailsWrapper, ServiceTier, StandardBuiltInToolsParams, @@ -322,6 +323,48 @@ class OCRPricing(TypedDict, total=False): annotation_cost_per_page: ReadOnly[float | None] +_WALL_CLOCK_PRICED_MODES: Final = frozenset({"chat", "completion", "embedding", "responses"}) + + +def _has_token_or_tiered_pricing(model_info: ModelInfoBase) -> bool: + return ( + (model_info.get("input_cost_per_token") or 0.0) > 0 + or (model_info.get("output_cost_per_token") or 0.0) > 0 + or model_info.get("tiered_pricing") is not None + ) + + +def _bills_wall_clock_seconds(model_info: ModelInfoBase) -> bool: + mode: Final = model_info.get("mode") + return mode is None or mode in _WALL_CLOCK_PRICED_MODES + + +def _per_second_pricing_cost( + model: str, + custom_llm_provider: str | None, + response_time_ms: float | None, +) -> tuple[float, float] | None: + try: + model_info: Final = _cached_get_model_info_helper(model=model, custom_llm_provider=custom_llm_provider) + except Exception: # noqa: BLE001 # the lookup raises plain Exception for an unmapped model + return None + if _has_token_or_tiered_pricing(model_info) or not _bills_wall_clock_seconds(model_info): + return None + input_cost_per_second: Final = model_info.get("input_cost_per_second") + output_cost_per_second: Final = model_info.get("output_cost_per_second") + if input_cost_per_second is None and output_cost_per_second is None: + return None + seconds: Final = (response_time_ms or 0.0) / 1000 + verbose_logger.debug( + "For model=%s - input_cost_per_second: %s; output_cost_per_second: %s; response time: %s", + model, + input_cost_per_second, + output_cost_per_second, + response_time_ms, + ) + return (input_cost_per_second or 0.0) * seconds, (output_cost_per_second or 0.0) * seconds + + def cost_per_token( model: str = "", prompt_tokens: int = 0, @@ -448,9 +491,6 @@ def cost_per_token( if response_cost is not None: return response_cost[0], response_cost[1] - # given - prompt_tokens_cost_usd_dollar: float = 0 - completion_tokens_cost_usd_dollar: float = 0 model_cost_ref: Final = litellm.model_cost # Only callers that explicitly pass `custom_llm_provider` get the # dedup/prefix-join treatment. When provider is omitted, preserve legacy @@ -611,6 +651,14 @@ def cost_per_token( number_of_queries=number_of_queries or 1, optional_params=(getattr(response, "_hidden_params", None) if response else None), ) + elif ( + per_second_cost := _per_second_pricing_cost( + model=model, + custom_llm_provider=custom_llm_provider, + response_time_ms=response_time_ms, + ) + ) is not None: + return per_second_cost elif custom_llm_provider == "vertex_ai": cost_router: Final = google_cost_router( model=model_without_prefix, @@ -685,12 +733,7 @@ def cost_per_token( ) else: model_info: Final = _cached_get_model_info_helper(model=model, custom_llm_provider=custom_llm_provider) - - if ( - (model_info.get("input_cost_per_token") or 0.0) > 0 - or (model_info.get("output_cost_per_token") or 0.0) > 0 - or model_info.get("tiered_pricing") is not None - ): + if _has_token_or_tiered_pricing(model_info): return generic_cost_per_token( model=model, usage=usage_block, @@ -698,36 +741,8 @@ def cost_per_token( service_tier=service_tier, data_residency=data_residency, ) - - input_cost_per_second: Final = model_info.get("input_cost_per_second") - if input_cost_per_second is not None and response_time_ms is not None: - verbose_logger.debug( - "For model=%s - input_cost_per_second: %s; response time: %s", - model, - input_cost_per_second, - response_time_ms, - ) - ## COST PER SECOND ## - prompt_tokens_cost_usd_dollar = input_cost_per_second * response_time_ms / 1000 - - output_cost_per_second: Final = model_info.get("output_cost_per_second") - if output_cost_per_second is not None and response_time_ms is not None: - verbose_logger.debug( - "For model=%s - output_cost_per_second: %s; response time: %s", - model, - output_cost_per_second, - response_time_ms, - ) - ## COST PER SECOND ## - completion_tokens_cost_usd_dollar = output_cost_per_second * response_time_ms / 1000 - - verbose_logger.debug( - "Returned custom cost for model=%s - prompt_tokens_cost_usd_dollar: %s, completion_tokens_cost_usd_dollar: %s", - model, - prompt_tokens_cost_usd_dollar, - completion_tokens_cost_usd_dollar, - ) - return prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar + verbose_logger.debug("No per-token, tiered, or per-second pricing for model=%s; cost is 0", model) + return 0.0, 0.0 def get_replicate_completion_pricing(completion_response: dict, total_time=0.0): @@ -1222,6 +1237,21 @@ def _split_responses_ws_logging_object_by_service_tier( ) +def _response_time_ms_for_cost( + completion_response: object, + litellm_logging_obj: LitellmLoggingObject | None, + total_time: float | None, +) -> float: + stamped: Final = getattr(completion_response, "_response_ms", None) + if isinstance(stamped, (int, float)): + return float(stamped) + if total_time: + return total_time + if litellm_logging_obj is not None: + return litellm_logging_obj.get_response_ms() + return 0.0 + + def completion_cost( completion_response: object | None = None, model: str | None = None, @@ -1443,8 +1473,6 @@ def completion_cost( prompt_tokens_details = _usage.get("prompt_tokens_details") or {} cache_read_input_tokens = prompt_tokens_details.get("cached_tokens", 0) - total_time = getattr(completion_response, "_response_ms", 0) - hidden_params = getattr(completion_response, "_hidden_params", None) if hidden_params is not None: custom_llm_provider = hidden_params.get("custom_llm_provider", custom_llm_provider or None) @@ -1676,6 +1704,11 @@ def completion_cost( ) return MCPCostCalculator.calculate_mcp_tool_call_cost(litellm_logging_obj=litellm_logging_obj) + response_time_ms = _response_time_ms_for_cost( + completion_response=completion_response, + litellm_logging_obj=litellm_logging_obj, + total_time=total_time, + ) # Calculate cost based on prompt_tokens, completion_tokens if ( "togethercomputer" in model or "together_ai" in model or custom_llm_provider == "together_ai" @@ -1686,7 +1719,7 @@ def completion_cost( # see https://replicate.com/pricing elif (model in litellm.replicate_models or "replicate" in model) and model not in litellm.model_cost: # for unmapped replicate model, default to replicate's time tracking logic - return get_replicate_completion_pricing(completion_response, total_time) + return get_replicate_completion_pricing(completion_response, response_time_ms) if model is None: raise ValueError( @@ -1718,7 +1751,7 @@ def completion_cost( prompt_tokens=prompt_tokens or 0, completion_tokens=completion_tokens or 0, custom_llm_provider=custom_llm_provider, - response_time_ms=total_time, + response_time_ms=response_time_ms, region_name=None if explicit_pricing else region_name, custom_cost_per_second=custom_cost_per_second, custom_cost_per_token=custom_cost_per_token, diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 0f14b46c8ba..d06780dda53 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -496,6 +496,14 @@ def mask_api_base_credentials(api_base: str) -> str: return api_base[:key_end] + "*" * 5 + api_base[-4:] +def _timestamp_seconds(moment: object) -> float | None: + if isinstance(moment, datetime.datetime): + return moment.timestamp() + if isinstance(moment, (int, float)): + return float(moment) + return None + + class Logging(LiteLLMLoggingBaseClass): global \ supabaseClient, \ @@ -1634,10 +1642,12 @@ class Logging(LiteLLMLoggingBaseClass): return response.mcp_tool_call_response def get_response_ms(self) -> float: - return ( - self.model_call_details.get("end_time", datetime.datetime.now()) - - self.model_call_details.get("start_time", datetime.datetime.now()) - ).total_seconds() * 1000 + now: Final = datetime.datetime.now() + start_seconds: Final = _timestamp_seconds(self.model_call_details.get("start_time", now)) + end_seconds: Final = _timestamp_seconds(self.model_call_details.get("end_time", now)) + if start_seconds is None or end_seconds is None: + return 0.0 + return (end_seconds - start_seconds) * 1000 def set_cost_breakdown( self, diff --git a/litellm/litellm_core_utils/llm_response_utils/response_metadata.py b/litellm/litellm_core_utils/llm_response_utils/response_metadata.py index 4a3c8de78c5..503814cc143 100644 --- a/litellm/litellm_core_utils/llm_response_utils/response_metadata.py +++ b/litellm/litellm_core_utils/llm_response_utils/response_metadata.py @@ -251,6 +251,6 @@ def update_response_metadata( return metadata: Final = ResponseMetadata(result) - metadata.set_hidden_params(logging_obj, model, kwargs) metadata.set_timing_metrics(start_time, end_time, logging_obj, include_overhead) + metadata.set_hidden_params(logging_obj, model, kwargs) metadata.apply() diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index 1bfc55d1486..021e012f29d 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -305,14 +305,15 @@ def test_response_cost_calculator_uses_router_model_id_from_litellm_metadata(): class TestZeroCostDiagnostic: - DEPLOYMENT_ID: Final = "lit7898-per-second-priced-deployment" - MODEL_GROUP: Final = "per-second-priced-chat" + DEPLOYMENT_ID: Final = "lit7898-query-only-priced-deployment" + MODEL_GROUP: Final = "query-only-priced-chat" + QUERY_ONLY_PRICING: Final = {"input_cost_per_query": 0.00042} PER_SECOND_PRICING: Final = {"input_cost_per_second": 0.00042, "output_cost_per_second": 0.00042} FREE_PRICING: Final = {"input_cost_per_token": 0, "output_cost_per_token": 0} - @pytest.fixture(params=["per_second", "free"]) + @pytest.fixture(params=["query_only", "free"]) def deployment_pricing(self, request: pytest.FixtureRequest) -> Iterator[Mapping[str, float]]: - pricing: Final = self.PER_SECOND_PRICING if request.param == "per_second" else self.FREE_PRICING + pricing: Final = self.QUERY_ONLY_PRICING if request.param == "query_only" else self.FREE_PRICING litellm.register_model(model_cost={self.DEPLOYMENT_ID: pricing}, persist_across_reloads=False) try: yield pricing @@ -557,11 +558,11 @@ class TestZeroCostDiagnostic: priced_pricing: Final = {"input_cost_per_token": 1e-06, "output_cost_per_token": 2e-06} usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30) litellm.register_model( - model_cost={self.DEPLOYMENT_ID: self.PER_SECOND_PRICING, priced_id: priced_pricing}, + model_cost={self.DEPLOYMENT_ID: self.QUERY_ONLY_PRICING, priced_id: priced_pricing}, persist_across_reloads=False, ) try: - logging_obj: Final = self._logging_obj(self.PER_SECOND_PRICING) + logging_obj: Final = self._logging_obj(self.QUERY_ONLY_PRICING) with caplog.at_level(logging.WARNING, logger="LiteLLM"): assert logging_obj._response_cost_calculator(result=self._response(usage)) == 0.0 self._assert_flagged(logging_obj, caplog) @@ -570,7 +571,7 @@ class TestZeroCostDiagnostic: assert logging_obj._response_cost_calculator(result=self._response(usage)) == pytest.approx(5e-05) assert logging_obj.model_call_details["zero_cost_diagnostic"] is None - self._route_to_deployment(logging_obj, self.PER_SECOND_PRICING) + self._route_to_deployment(logging_obj, self.QUERY_ONLY_PRICING) assert logging_obj._response_cost_calculator(result=self._response(usage)) == 0.0 assert logging_obj.model_call_details["zero_cost_diagnostic"]["reason"] == "missing_pricing_key" @@ -585,7 +586,7 @@ class TestZeroCostDiagnostic: dated_model: Final = "lit7898-nano-2026-03-17" requested_model: Final = "lit7898-nano" usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30) - cost_map_entry: Final = {"litellm_provider": "openai", "mode": "chat", **self.PER_SECOND_PRICING} + cost_map_entry: Final = {"litellm_provider": "openai", "mode": "chat", **self.QUERY_ONLY_PRICING} litellm.register_model( model_cost={dated_model: cost_map_entry, requested_model: cost_map_entry}, persist_across_reloads=False ) @@ -646,6 +647,24 @@ class TestZeroCostDiagnostic: assert logging_obj.model_call_details["zero_cost_diagnostic"] is None assert self._zero_cost_warnings(caplog) == [] + def test_per_second_priced_deployment_bills_the_call_duration_and_stays_silent( + self, caplog: pytest.LogCaptureFixture + ) -> None: + per_second_id: Final = "lit8315-per-second-priced-deployment" + usage: Final = litellm.Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30) + litellm.register_model(model_cost={per_second_id: self.PER_SECOND_PRICING}, persist_across_reloads=False) + try: + logging_obj: Final = self._logging_obj(self.PER_SECOND_PRICING, deployment_id=per_second_id) + response: Final = self._response(usage) + response._response_ms = 1000.0 + with caplog.at_level(logging.WARNING, logger="LiteLLM"): + assert logging_obj._response_cost_calculator(result=response) == pytest.approx(0.00084) + + assert logging_obj.model_call_details["zero_cost_diagnostic"] is None + assert self._zero_cost_warnings(caplog) == [] + finally: + litellm.model_cost.pop(per_second_id, None) + @pytest.mark.parametrize("spilled_over", [True, False]) def test_ptu_deployment_is_judged_by_the_entry_the_calculator_priced_with( self, spilled_over: bool, monkeypatch: pytest.MonkeyPatch, caplog: pytest.LogCaptureFixture @@ -663,7 +682,7 @@ class TestZeroCostDiagnostic: litellm.register_model( model_cost={ router_model_id: {**self.FREE_PRICING, "litellm_provider": "azure", "mode": "chat"}, - served_model: {**self.PER_SECOND_PRICING, "litellm_provider": "azure", "mode": "chat"}, + served_model: {**self.QUERY_ONLY_PRICING, "litellm_provider": "azure", "mode": "chat"}, }, persist_across_reloads=False, ) @@ -3504,6 +3523,20 @@ def _make_logging_obj(stream: bool) -> LitellmLogging: ) +def test_get_response_ms_measures_a_float_start_time_against_a_datetime_end_time(): + """The files paths construct the logging object with ``time.time()`` while the success + handler stamps a datetime end, and the per-second cost path reads this window.""" + logging_obj = _make_logging_obj(stream=False) + logging_obj.update_environment_variables( + model="openai/codex-mini-latest", user="", optional_params={}, litellm_params={} + ) + start_seconds = logging_obj.model_call_details["start_time"] + assert isinstance(start_seconds, float) + logging_obj.model_call_details["end_time"] = datetime.datetime.fromtimestamp(start_seconds + 1.5) + + assert logging_obj.get_response_ms() == pytest.approx(1500) + + def test_get_assembled_streaming_response_returns_none_for_non_streaming(): """Non-streaming requests should return None so the streaming block is skipped.""" import datetime diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 84c60e029cb..bdeedb5f38c 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -1,3 +1,4 @@ +import datetime import time from typing import Final @@ -29,6 +30,7 @@ from litellm.types.utils import ( PromptTokensDetailsWrapper, Usage, ) +from litellm.types.videos.main import VideoObject @pytest.fixture @@ -3038,21 +3040,23 @@ def test_completion_cost_logs_cache_and_reasoning_breakdown_for_custom_pricing() assert total == pytest.approx(100 * 1e-6 + 800 * 1e-7 + 100 * 1.25e-6 + 500 * 2e-6) -def test_cost_per_token_per_second_pricing(monkeypatch): +@pytest.mark.parametrize("custom_llm_provider", ["together_ai", "openai", "anthropic", "bedrock", "azure"]) +def test_cost_per_token_per_second_pricing(monkeypatch, custom_llm_provider: str): """ Models priced by duration (input/output_cost_per_second) with no per-token rates - must be billed as cost_per_second * response_time_ms / 1000 in cost_per_token. + must be billed as cost_per_second * response_time_ms / 1000 in cost_per_token, + whether or not the provider has its own cost calculator. """ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) - model = "test-per-second-pricing-model" + model = f"test-per-second-pricing-{custom_llm_provider}" litellm.register_model( model_cost={ model: { "input_cost_per_second": 0.02, "output_cost_per_second": 0.04, - "litellm_provider": "together_ai", + "litellm_provider": custom_llm_provider, "mode": "chat", } } @@ -3060,7 +3064,7 @@ def test_cost_per_token_per_second_pricing(monkeypatch): prompt_cost, completion_cost_value = cost_per_token( model=model, - custom_llm_provider="together_ai", + custom_llm_provider=custom_llm_provider, prompt_tokens=10, completion_tokens=20, response_time_ms=1500.0, @@ -3070,6 +3074,143 @@ def test_cost_per_token_per_second_pricing(monkeypatch): assert completion_cost_value == pytest.approx(0.04 * 1.5) +def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(monkeypatch): + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + model = "test-token-and-per-second-pricing-model" + litellm.register_model( + model_cost={ + model: { + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "input_cost_per_second": 0.02, + "output_cost_per_second": 0.04, + "litellm_provider": "openai", + "mode": "chat", + } + } + ) + + prompt_cost, completion_cost_value = cost_per_token( + model=model, + custom_llm_provider="openai", + prompt_tokens=10, + completion_tokens=20, + response_time_ms=1500.0, + ) + + assert prompt_cost == pytest.approx(10 * 1e-6) + assert completion_cost_value == pytest.approx(20 * 2e-6) + + +def _logging_obj_with_call_window(duration_ms: float) -> Logging: + start_time: Final = datetime.datetime(2026, 9, 21, 12, 0, 0) + logging_obj: Final = Logging( + model="gpt-5.4-nano", + messages=[{"role": "user", "content": "Hello"}], + stream=False, + call_type="completion", + start_time=start_time, + litellm_call_id="per-second-call-window", + function_id="f", + ) + logging_obj.model_call_details["start_time"] = start_time + logging_obj.model_call_details["end_time"] = start_time + datetime.timedelta(milliseconds=duration_ms) + return logging_obj + + +@pytest.mark.parametrize( + ("stamped_response_ms", "total_time", "logged_duration_ms", "expected_seconds"), + [(None, 0.0, 1500.0, 1.5), (3000.0, 0.0, 1500.0, 3.0), (None, 2500.0, 1500.0, 2.5), (3000.0, 2500.0, 1500.0, 3.0)], +) +def test_completion_cost_per_second_deployment_bills_the_call_duration( + monkeypatch, + stamped_response_ms: float | None, + total_time: float, + logged_duration_ms: float, + expected_seconds: float, +): + """ + A deployment priced only per second bills the stamped ``_response_ms`` when there is one, + then the caller's explicit ``total_time``, and the logging object's start/end window otherwise + (a streamed response is never stamped). + """ + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + deployment_id = "per-second-openai-deployment" + litellm.register_model( + model_cost={ + deployment_id: { + "input_cost_per_second": 0.02, + "output_cost_per_second": 0.04, + "litellm_provider": "openai", + "mode": "chat", + } + } + ) + response = ModelResponse( + model="gpt-5.4-nano", + usage=Usage(prompt_tokens=11, completion_tokens=7, total_tokens=18), + ) + response._response_ms = stamped_response_ms + + cost = completion_cost( + completion_response=response, + model="openai/gpt-5.4-nano", + custom_llm_provider="openai", + custom_pricing=True, + router_model_id=deployment_id, + total_time=total_time, + litellm_logging_obj=_logging_obj_with_call_window(logged_duration_ms), + ) + + assert cost == pytest.approx((0.02 + 0.04) * expected_seconds) + + +@pytest.mark.parametrize("mode", ["audio_transcription", "audio_speech", "video_generation", "realtime"]) +def test_cost_per_token_leaves_media_second_rates_to_their_dedicated_paths(monkeypatch, mode: str): + """ + A media-mode entry's per-second rates price audio or video seconds, which the dedicated + transcription, speech, video, and realtime paths bill from the media itself, so a call that + reaches the generic path with only a wall-clock duration must not bill them. + """ + model = f"test-media-per-second-{mode}" + monkeypatch.setitem( + litellm.model_cost, + model, + {"input_cost_per_second": 0.02, "output_cost_per_second": 0.4, "litellm_provider": "openai", "mode": mode}, + ) + + assert cost_per_token(model=model, custom_llm_provider="openai", response_time_ms=2000.0) == (0.0, 0.0) + + +def test_completion_cost_video_status_poll_bills_nothing_on_a_per_second_video_model(monkeypatch): + """ + Polling a video job returns a ``VideoObject`` with no stamped duration, so the cost path falls + back to the logging object's call window; on a video model priced per output second that + window must not be billed, or every status poll would charge for the seconds it took to answer. + """ + model = "test-veo-per-second-poll" + monkeypatch.setitem( + litellm.model_cost, + model, + {"output_cost_per_second": 0.4, "litellm_provider": "vertex_ai", "mode": "video_generation"}, + ) + video = VideoObject(id="video_1", object="video", status="completed", model=model, progress=100) + + cost = completion_cost( + completion_response=video, + model=model, + custom_llm_provider="vertex_ai", + call_type=CallTypes.video_retrieve.value, + litellm_logging_obj=_logging_obj_with_call_window(2000.0), + ) + + assert cost == 0.0 + + def _batch_cache_usage() -> Usage: return Usage( prompt_tokens=11000, diff --git a/tests/unit/litellm_core_utils/llm_response_utils/test_response_metadata.py b/tests/unit/litellm_core_utils/llm_response_utils/test_response_metadata.py index 50409b2ea2c..37201e8155b 100644 --- a/tests/unit/litellm_core_utils/llm_response_utils/test_response_metadata.py +++ b/tests/unit/litellm_core_utils/llm_response_utils/test_response_metadata.py @@ -7,10 +7,12 @@ through _hidden_params to the x-litellm-callback-duration-ms response header. import asyncio import datetime +from typing import Final from unittest.mock import MagicMock import pytest +import litellm import litellm.litellm_core_utils.llm_response_utils.response_metadata as response_metadata_mod import litellm.proxy.common_request_processing as common_request_processing_mod from litellm.litellm_core_utils.litellm_logging import Logging @@ -22,7 +24,7 @@ from litellm.litellm_core_utils.llm_response_utils.response_metadata import ( ) from litellm.proxy._types import UserAPIKeyAuth from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing -from litellm.types.utils import ModelResponse +from litellm.types.utils import ModelResponse, Usage class TestCallbackDurationMs: @@ -583,3 +585,57 @@ class TestLoggingInitCallbackDuration: # Should still be set (deep copy of None is essentially a no-op) assert hasattr(obj, "callback_duration_ms") assert obj.callback_duration_ms >= 0 + + +def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_duration(monkeypatch): + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + deployment_id: Final = "per-second-deployment-response-metadata" + litellm.register_model( + model_cost={ + deployment_id: { + "input_cost_per_second": 0.02, + "output_cost_per_second": 0.04, + "litellm_provider": "openai", + "mode": "chat", + } + } + ) + start_time: Final = datetime.datetime(2026, 9, 21, 12, 0, 0) + logging_obj: Final = Logging( + model="gpt-5.4-nano", + messages=[{"role": "user", "content": "Hello"}], + stream=False, + call_type="completion", + start_time=start_time, + litellm_call_id="per-second-response-metadata", + function_id="f", + ) + logging_obj.update_environment_variables( + model="gpt-5.4-nano", + litellm_params={ + "input_cost_per_second": 0.02, + "output_cost_per_second": 0.04, + "metadata": {"model_info": {"id": deployment_id}}, + }, + optional_params={}, + custom_llm_provider="openai", + ) + logging_obj.model_call_details["end_time"] = start_time + datetime.timedelta(seconds=10) + result: Final = ModelResponse( + model="gpt-5.4-nano", + usage=Usage(prompt_tokens=11, completion_tokens=7, total_tokens=18), + ) + + update_response_metadata( + result=result, + logging_obj=logging_obj, + model="gpt-5.4-nano", + kwargs={"model_info": {"id": deployment_id}}, + start_time=start_time, + end_time=start_time + datetime.timedelta(seconds=2), + ) + + assert result._response_ms == pytest.approx(2000) + assert result._hidden_params["response_cost"] == pytest.approx((0.02 + 0.04) * 2)