mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
feat(cost-calc): infer data_residency from response hidden_params
completion_cost(completion_response=resp) is a user-facing entry point that reruns the cost calculation from the response object alone — it does not have access to the logging object's litellm_params. Without this, users calling completion_cost() after a completion would see the base cost, while the internal cost calculation (via _response_cost_calculator in litellm_logging.py) would apply the uplift and store it on _hidden_params["response_cost"]. Symmetry: also read api_base from hidden_params and infer data_residency there. https://claude.ai/code/session_012ebH44s7ohYxjoix5CXzTW
This commit is contained in:
parent
7697518fae
commit
6c313e9b5e
2 changed files with 52 additions and 0 deletions
|
|
@ -15,6 +15,7 @@ from litellm.constants import (
|
|||
DEFAULT_MAX_LRU_CACHE_SIZE,
|
||||
DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND,
|
||||
)
|
||||
from litellm.litellm_core_utils.data_residency import infer_openai_data_residency
|
||||
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
|
||||
StandardBuiltInToolCostTracking,
|
||||
)
|
||||
|
|
@ -1302,6 +1303,15 @@ def completion_cost( # noqa: PLR0915
|
|||
)
|
||||
region_name = hidden_params.get("region_name", region_name)
|
||||
|
||||
# For OpenAI, infer data_residency from api_base if it
|
||||
# wasn't passed in explicitly. Keeps external
|
||||
# completion_cost(response) in sync with the internal
|
||||
# cost calculation (which reads it from litellm_params).
|
||||
if data_residency is None and custom_llm_provider == "openai":
|
||||
data_residency = infer_openai_data_residency(
|
||||
hidden_params.get("api_base")
|
||||
)
|
||||
|
||||
# For Gemini/Vertex AI responses, trafficType is stored in
|
||||
# provider_specific_fields. Map it to the service_tier used
|
||||
# by the cost key lookup (_priority / _flex suffixes) so that
|
||||
|
|
|
|||
|
|
@ -1530,3 +1530,45 @@ def test_data_residency_composes_with_service_tier(_local_model_cost_map):
|
|||
|
||||
assert priority_base_total > 0
|
||||
assert priority_eu_total == pytest.approx(priority_base_total * 1.10, rel=1e-9)
|
||||
|
||||
|
||||
def test_completion_cost_infers_data_residency_from_hidden_params(
|
||||
_local_model_cost_map,
|
||||
):
|
||||
"""External completion_cost(response) infers data_residency from
|
||||
the api_base recorded on hidden_params, so it matches the internal cost
|
||||
calculation from the completion flow."""
|
||||
from litellm import ModelResponse, completion_cost
|
||||
from litellm.types.utils import Choices, Message
|
||||
from litellm.types.utils import Usage as U
|
||||
|
||||
def _resp(api_base):
|
||||
r = ModelResponse(
|
||||
id="test",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(content="ok", role="assistant"),
|
||||
)
|
||||
],
|
||||
model="gpt-5",
|
||||
object="chat.completion",
|
||||
created=0,
|
||||
usage=U(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
|
||||
)
|
||||
r._hidden_params = {
|
||||
"custom_llm_provider": "openai",
|
||||
"api_base": api_base,
|
||||
}
|
||||
return r
|
||||
|
||||
global_cost = completion_cost(
|
||||
completion_response=_resp("https://api.openai.com/v1")
|
||||
)
|
||||
eu_cost = completion_cost(completion_response=_resp("https://eu.api.openai.com/v1"))
|
||||
us_cost = completion_cost(completion_response=_resp("https://us.api.openai.com/v1"))
|
||||
|
||||
assert global_cost > 0
|
||||
assert eu_cost == pytest.approx(global_cost * 1.10, rel=1e-9)
|
||||
assert us_cost == pytest.approx(global_cost * 1.10, rel=1e-9)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue