diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 145de2c1bdc..eb66967b878 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -1857,6 +1857,7 @@ def response_cost_calculator( base_model: Optional[str] = None, custom_pricing: Optional[bool] = None, custom_cost_per_token: Optional[CostPerToken] = None, + custom_cost_per_second: Optional[float] = None, prompt: str = "", standard_built_in_tools_params: Optional[StandardBuiltInToolsParams] = None, litellm_model_name: Optional[str] = None, @@ -1895,6 +1896,7 @@ def response_cost_calculator( optional_params=optional_params, custom_pricing=custom_pricing, custom_cost_per_token=custom_cost_per_token, + custom_cost_per_second=custom_cost_per_second, base_model=base_model, prompt=prompt, standard_built_in_tools_params=standard_built_in_tools_params, diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index ab992939551..2d23f29b118 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -1525,8 +1525,12 @@ class Logging(LiteLLMLoggingBaseClass): # litellm.model_cost (model-cost-map reloads replace the map wholesale, # after which cost calc would fall back to the model's public price). custom_cost_per_token: Optional[CostPerToken] = None + custom_cost_per_second: Optional[float] = None if custom_pricing is True: - custom_cost_per_token = get_custom_cost_per_token_from_litellm_params( + ( + custom_cost_per_token, + custom_cost_per_second, + ) = get_custom_cost_from_litellm_params( litellm_params=( self.litellm_params if hasattr(self, "litellm_params") else None ) @@ -1555,6 +1559,7 @@ class Logging(LiteLLMLoggingBaseClass): "optional_params": self.optional_params, "custom_pricing": custom_pricing, "custom_cost_per_token": custom_cost_per_token, + "custom_cost_per_second": custom_cost_per_second, "prompt": prompt, "standard_built_in_tools_params": self.standard_built_in_tools_params, "router_model_id": router_model_id, @@ -4788,17 +4793,21 @@ def use_custom_pricing_for_model(litellm_params: Optional[dict]) -> bool: return False -def get_custom_cost_per_token_from_litellm_params( +def get_custom_cost_from_litellm_params( litellm_params: Optional[dict], -) -> Optional[CostPerToken]: +) -> Tuple[Optional[CostPerToken], Optional[float]]: """ - Extract explicit per-token rates from litellm_params or its + Extract explicit (per-token, per-second) rates from litellm_params or its metadata/litellm_metadata model_info, mirroring the lookup order of `use_custom_pricing_for_model`. A literal 0 is a valid rate (e.g. zero-cost - BYOK deployments). Returns None unless both base rates are set. + BYOK deployments). Per-token pricing requires both base rates. Per-second + pricing requires input_cost_per_second (the same rule pricing registration + uses in main.py); an output_cost_per_second is folded into litellm's + single custom_cost_per_second rate so the billed total matches a + registered per-second cost-map entry. """ if litellm_params is None: - return None + return None, None sources = [litellm_params] for metadata_key in ("metadata", "litellm_metadata"): @@ -4810,21 +4819,25 @@ def get_custom_cost_per_token_from_litellm_params( for source in sources: input_cost = source.get("input_cost_per_token") output_cost = source.get("output_cost_per_token") - if input_cost is None or output_cost is None: - continue - custom_cost: CostPerToken = { - "input_cost_per_token": input_cost, - "output_cost_per_token": output_cost, - } - for cache_key in ( - "cache_read_input_token_cost", - "cache_creation_input_token_cost", - ): - if source.get(cache_key) is not None: - custom_cost[cache_key] = source[cache_key] # type: ignore[literal-required] - return custom_cost + if input_cost is not None and output_cost is not None: + custom_cost: CostPerToken = { + "input_cost_per_token": input_cost, + "output_cost_per_token": output_cost, + } + for cache_key in ( + "cache_read_input_token_cost", + "cache_creation_input_token_cost", + ): + if source.get(cache_key) is not None: + custom_cost[cache_key] = source[cache_key] # type: ignore[literal-required] + return custom_cost, None + input_cost_per_second = source.get("input_cost_per_second") + if input_cost_per_second is not None: + return None, input_cost_per_second + ( + source.get("output_cost_per_second") or 0.0 + ) - return None + return None, None def is_valid_sha256_hash(value: str) -> bool: diff --git a/litellm/main.py b/litellm/main.py index 759db30f306..39c6f3b4c24 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1092,8 +1092,10 @@ def _model_has_known_pricing(model: str, custom_llm_provider: str) -> bool: shared entry and re-prices every other request for that model in the process (e.g. a zero-cost BYOK deployment zeroing the real model's billing). Known models get their per-request rates via - custom_cost_per_token at cost-calculation time instead; registration is - only needed so unknown models can be priced at all. + custom_cost_per_token / custom_cost_per_second at cost-calculation time + instead (see get_custom_cost_from_litellm_params, which covers the same + pricing dimensions checked here); registration is only needed so unknown + models can be priced at all. """ for key in (model, f"{custom_llm_provider}/{model}"): entry = litellm.model_cost.get(key) diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index e7f90416571..143e7bba3bf 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -2639,3 +2639,27 @@ def test_response_cost_calculator_custom_pricing_survives_model_cost_reload( result=_response_with_usage("claude-opus-4-6") ) assert cost_after_reload == pytest.approx(expected) + + +def test_response_cost_calculator_uses_deployment_per_second_rates(): + """A deployment priced per second must be billed at its own rates, not the + concrete model's public per-token price (same custom-pricing path as the + per-token tests above, for the other pricing dimension).""" + logging_obj = _logging_obj_with_custom_pricing( + litellm_params={ + "metadata": { + "model_info": { + "id": "per-second-deployment-id-not-in-model-cost", + "input_cost_per_second": 0.001, + "output_cost_per_second": 0.002, + } + } + }, + model="anthropic/claude-opus-4-6", + ) + response = _response_with_usage("claude-opus-4-6") + response._response_ms = 5_000 + + cost = logging_obj._response_cost_calculator(result=response) + + assert cost == pytest.approx((0.001 + 0.002) * 5) diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index 09580530b0f..6c06486e12a 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -1988,19 +1988,24 @@ def test_completion_custom_pricing_does_not_overwrite_canonical_model_cost( assert response._hidden_params["response_cost"] == 0.0 -def test_completion_custom_pricing_still_registers_unknown_model(): +def test_completion_custom_pricing_still_registers_unknown_model(monkeypatch): model = "openai/unknown-custom-priced-model-xyz" + monkeypatch.setattr(litellm, "model_cost", dict(litellm.model_cost)) + monkeypatch.setattr( + litellm, + "open_ai_chat_completion_models", + set(litellm.open_ai_chat_completion_models), + ) litellm.model_cost.pop(model, None) - try: - litellm.completion( - model=model, - messages=[{"role": "user", "content": "hi"}], - mock_response="ok", - api_key="sk-test", - input_cost_per_token=1e-07, - output_cost_per_token=2e-07, - ) - assert litellm.model_cost[model]["input_cost_per_token"] == 1e-07 - assert litellm.model_cost[model]["output_cost_per_token"] == 2e-07 - finally: - litellm.model_cost.pop(model, None) + + litellm.completion( + model=model, + messages=[{"role": "user", "content": "hi"}], + mock_response="ok", + api_key="sk-test", + input_cost_per_token=1e-07, + output_cost_per_token=2e-07, + ) + + assert litellm.model_cost[model]["input_cost_per_token"] == 1e-07 + assert litellm.model_cost[model]["output_cost_per_token"] == 2e-07