mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(logging): stop billing and logging response reads as LLM calls
Retrieving, deleting or cancelling a stored response, and vector store management calls, run through the same logging lifecycle as inference. A retrieved response replays the usage of the call that created it, so every read priced it again and wrote a second spend log row for the same tokens. Non-inference calls now cost 0, report no usage, log no placeholder chat message, and get a litellm.responses_management operation name instead of reading as chat. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
0a25756e78
commit
6285192141
8 changed files with 200 additions and 2 deletions
|
|
@ -1736,6 +1736,33 @@ BROWSER_SECURITY_HEADERS: Final[frozenset[str]] = frozenset(
|
|||
|
||||
UNSAFE_PROXY_RESPONSE_HEADERS: Final[frozenset[str]] = HTTP_FRAMING_HEADERS | BROWSER_SECURITY_HEADERS
|
||||
|
||||
# Read/management routes that run through the same request processing and logging
|
||||
# lifecycle as inference, but never send a prompt to a model. Their responses can still
|
||||
# carry a usage object (a retrieved response replays the usage of the call that created
|
||||
# it), so pricing them bills the same tokens twice.
|
||||
NON_INFERENCE_CALL_TYPES: Final[frozenset[str]] = frozenset(
|
||||
{
|
||||
"get_responses",
|
||||
"aget_responses",
|
||||
"delete_responses",
|
||||
"adelete_responses",
|
||||
"cancel_responses",
|
||||
"acancel_responses",
|
||||
"list_input_items",
|
||||
"alist_input_items",
|
||||
"vector_store_create",
|
||||
"avector_store_create",
|
||||
"vector_store_retrieve",
|
||||
"avector_store_retrieve",
|
||||
"vector_store_list",
|
||||
"avector_store_list",
|
||||
"vector_store_update",
|
||||
"avector_store_update",
|
||||
"vector_store_delete",
|
||||
"avector_store_delete",
|
||||
}
|
||||
)
|
||||
|
||||
# PTU reservation rollup writes rows to LiteLLM_DailyTeamSpend with this
|
||||
# sentinel api_key so PTU flat cost stays distinguishable from real per-request
|
||||
# spend under the table's composite unique constraint.
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ class GenAIOperation(str, Enum):
|
|||
EXECUTE_TOOL = "execute_tool" # MCP tool-call spans
|
||||
LITELLM_VECTOR_STORE_MANAGEMENT = "litellm.vector_store_management"
|
||||
LITELLM_VECTOR_STORE_FILE_MANAGEMENT = "litellm.vector_store_file_management"
|
||||
LITELLM_RESPONSES_MANAGEMENT = "litellm.responses_management" # fetch/delete/cancel a stored response
|
||||
|
||||
|
||||
class GenAIProvider(str, Enum):
|
||||
|
|
@ -348,6 +349,14 @@ _OPERATION_BY_CALL_TYPE: Final[dict[str, GenAIOperation]] = {
|
|||
"aembedding": GenAIOperation.EMBEDDINGS,
|
||||
"responses": GenAIOperation.CHAT,
|
||||
"aresponses": GenAIOperation.CHAT,
|
||||
"get_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
|
||||
"aget_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
|
||||
"delete_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
|
||||
"adelete_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
|
||||
"cancel_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
|
||||
"acancel_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
|
||||
"list_input_items": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
|
||||
"alist_input_items": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
|
||||
"call_mcp_tool": GenAIOperation.EXECUTE_TOOL,
|
||||
"vector_store_search": GenAIOperation.RETRIEVAL,
|
||||
"avector_store_search": GenAIOperation.RETRIEVAL,
|
||||
|
|
|
|||
|
|
@ -41,6 +41,7 @@ from litellm.caching.caching_handler import LLMCachingHandler
|
|||
from litellm.constants import (
|
||||
DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT,
|
||||
DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT,
|
||||
NON_INFERENCE_CALL_TYPES,
|
||||
SENTRY_DENYLIST,
|
||||
SENTRY_PII_DENYLIST,
|
||||
)
|
||||
|
|
@ -1435,6 +1436,9 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
if cache_hit is True:
|
||||
return 0.0
|
||||
|
||||
if self.call_type in NON_INFERENCE_CALL_TYPES:
|
||||
return 0.0
|
||||
|
||||
transformed_result: Final = self._generate_content_result_as_model_response(result)
|
||||
if transformed_result is not None:
|
||||
result = transformed_result
|
||||
|
|
@ -5487,7 +5491,7 @@ def get_standard_logging_object_payload(
|
|||
cache_hit: Final = kwargs.get("cache_hit", False)
|
||||
# Extract usage as a plain dict, avoiding Pydantic round-trip
|
||||
raw_usage_dict: Final = StandardLoggingPayloadSetup.get_usage_as_dict(
|
||||
response_obj=response_obj,
|
||||
response_obj=None if call_type in NON_INFERENCE_CALL_TYPES else response_obj,
|
||||
combined_usage_object=cast(Usage | None, kwargs.get("combined_usage_object")),
|
||||
)
|
||||
usage_dict: Final = (
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ from litellm._logging import verbose_proxy_logger
|
|||
from litellm.constants import (
|
||||
LITELLM_TRUNCATED_PAYLOAD_FIELD,
|
||||
LITELLM_TRUNCATION_DB_SAFEGUARD_NOTE,
|
||||
NON_INFERENCE_CALL_TYPES,
|
||||
REDACTED_BY_LITELM_STRING,
|
||||
)
|
||||
from litellm.constants import (
|
||||
|
|
@ -258,7 +259,7 @@ def get_logging_payload(kwargs, response_obj, start_time, end_time) -> SpendLogs
|
|||
usage: dict = {}
|
||||
if call_type in ["ocr", "aocr"]:
|
||||
usage = _extract_usage_for_ocr_call(response_obj, response_obj_dict)
|
||||
else:
|
||||
elif call_type not in NON_INFERENCE_CALL_TYPES:
|
||||
# Use response_obj_dict instead of response_obj to avoid calling .get() on Pydantic models
|
||||
_usage: Final = response_obj_dict.get("usage", None) or {}
|
||||
if isinstance(_usage, litellm.Usage):
|
||||
|
|
|
|||
|
|
@ -72,6 +72,7 @@ from litellm.constants import (
|
|||
MAX_RETRY_DELAY,
|
||||
MAX_TOKEN_TRIMMING_ATTEMPTS,
|
||||
MINIMUM_PROMPT_CACHE_TOKEN_COUNT_OVERRIDE,
|
||||
NON_INFERENCE_CALL_TYPES,
|
||||
OPENAI_EMBEDDING_PARAMS,
|
||||
TOOL_CHOICE_OBJECT_TOKEN_COUNT,
|
||||
)
|
||||
|
|
@ -1070,6 +1071,8 @@ def function_setup(
|
|||
except Exception as e:
|
||||
verbose_logger.debug("Error extracting messages from Google contents: %s", e)
|
||||
messages = "default-message-value"
|
||||
elif call_type in NON_INFERENCE_CALL_TYPES:
|
||||
messages = ()
|
||||
else:
|
||||
messages = "default-message-value"
|
||||
stream = False
|
||||
|
|
|
|||
|
|
@ -264,6 +264,27 @@ def test_vector_store_file_management_is_not_chat(call_type):
|
|||
assert resolve_operation(call_type).value == "litellm.vector_store_file_management"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"call_type",
|
||||
[
|
||||
f"{prefix}{operation}"
|
||||
for operation in ("get_responses", "delete_responses", "cancel_responses", "list_input_items")
|
||||
for prefix in ("", "a")
|
||||
],
|
||||
)
|
||||
def test_responses_management_is_not_chat(call_type):
|
||||
"""Fetching, deleting or cancelling a stored response runs no inference, so it must not
|
||||
read as a chat completion: the retrieved object replays the original call's tokens and
|
||||
would inflate the chat series on every read. Regression test for LIT-5602."""
|
||||
assert resolve_operation(call_type) is GenAIOperation.LITELLM_RESPONSES_MANAGEMENT
|
||||
assert resolve_operation(call_type).value == "litellm.responses_management"
|
||||
|
||||
|
||||
def test_creating_a_response_is_still_chat():
|
||||
"""Guards the test above: ``/v1/responses`` itself is a chat completion."""
|
||||
assert resolve_operation("aresponses") is GenAIOperation.CHAT
|
||||
|
||||
|
||||
def test_vendor_operation_values_are_namespaced():
|
||||
"""A vendor value must stay under the ``litellm.`` prefix: an unprefixed invented
|
||||
name could collide with a value the convention adds later, silently changing what
|
||||
|
|
|
|||
|
|
@ -4539,3 +4539,93 @@ async def test_restore_correlation_context_works_across_asyncio_task_boundary():
|
|||
finally:
|
||||
trace_id_var.set("")
|
||||
session_id_var.set("")
|
||||
|
||||
|
||||
class TestNonInferenceCallTypesAreNotBilled:
|
||||
"""A retrieved response replays the usage of the call that created it, so pricing a read
|
||||
of it double bills the same tokens. Regression tests for LIT-5602."""
|
||||
|
||||
RETRIEVED_RESPONSE_USAGE = {"input_tokens": 4000, "output_tokens": 2000, "total_tokens": 6000}
|
||||
|
||||
def _logging_obj(self, call_type: str):
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
||||
obj = LiteLLMLoggingObj(
|
||||
model="gpt-4o",
|
||||
messages=[],
|
||||
stream=False,
|
||||
call_type=call_type,
|
||||
start_time=time.time(),
|
||||
litellm_call_id=f"lit5602-{call_type}",
|
||||
function_id="fn-lit5602",
|
||||
)
|
||||
obj.update_environment_variables(
|
||||
model="gpt-4o",
|
||||
user="",
|
||||
optional_params={},
|
||||
litellm_params={"api_base": "", "custom_llm_provider": "openai"},
|
||||
)
|
||||
return obj
|
||||
|
||||
def _retrieved_response(self):
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
return ResponsesAPIResponse(
|
||||
id="resp_lit5602",
|
||||
created_at=1234567890,
|
||||
model="gpt-4o",
|
||||
output=[],
|
||||
usage=self.RETRIEVED_RESPONSE_USAGE,
|
||||
)
|
||||
|
||||
def test_creating_a_response_is_still_priced(self):
|
||||
"""Guards the tests below: the same response object must cost money on the create path."""
|
||||
cost = self._logging_obj("aresponses")._response_cost_calculator(result=self._retrieved_response())
|
||||
assert cost is not None and cost > 0
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"call_type",
|
||||
["aget_responses", "adelete_responses", "acancel_responses", "alist_input_items", "avector_store_delete"],
|
||||
)
|
||||
def test_read_and_management_calls_cost_nothing(self, call_type):
|
||||
cost = self._logging_obj(call_type)._response_cost_calculator(result=self._retrieved_response())
|
||||
assert cost == 0.0
|
||||
|
||||
def test_retrieved_usage_is_not_re_reported_in_standard_logging_payload(self):
|
||||
from litellm.litellm_core_utils.litellm_logging import (
|
||||
get_standard_logging_object_payload,
|
||||
)
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
logging_obj = self._logging_obj("aget_responses")
|
||||
now = datetime.now()
|
||||
payload = get_standard_logging_object_payload(
|
||||
kwargs={
|
||||
"litellm_call_id": "lit5602-payload",
|
||||
"model": "gpt-4o",
|
||||
"call_type": "aget_responses",
|
||||
"litellm_params": {},
|
||||
},
|
||||
init_response_obj=self._retrieved_response(),
|
||||
start_time=now,
|
||||
end_time=now,
|
||||
logging_obj=logging_obj,
|
||||
status="success",
|
||||
)
|
||||
|
||||
assert payload is not None
|
||||
assert payload["prompt_tokens"] == 0
|
||||
assert payload["completion_tokens"] == 0
|
||||
assert payload["total_tokens"] == 0
|
||||
assert payload["response_cost"] == 0.0
|
||||
|
||||
def test_read_calls_do_not_log_a_placeholder_chat_message(self):
|
||||
logging_obj, _ = litellm.utils.function_setup(
|
||||
original_function="aget_responses",
|
||||
rules_obj=litellm.utils.Rules(),
|
||||
start_time=time.time(),
|
||||
**{"litellm_call_id": "lit5602-setup", "response_id": "resp_lit5602"},
|
||||
)
|
||||
|
||||
assert logging_obj.model_call_details["messages"] == ()
|
||||
|
|
|
|||
|
|
@ -2959,3 +2959,46 @@ def test_user_traffic_carries_no_internal_call_origin():
|
|||
)
|
||||
metadata = json.loads(payload["metadata"])
|
||||
assert metadata["internal_call_origin"] is None
|
||||
|
||||
|
||||
def _spend_log_for_call_type(call_type: str) -> dict:
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
return cast(
|
||||
dict,
|
||||
get_logging_payload(
|
||||
kwargs={
|
||||
"model": "gpt-4o",
|
||||
"call_type": call_type,
|
||||
"response_cost": 0.0,
|
||||
"litellm_params": {"metadata": {"user_api_key": "test-key"}},
|
||||
},
|
||||
response_obj=ResponsesAPIResponse(
|
||||
id="resp_lit5602",
|
||||
created_at=1234567890,
|
||||
model="gpt-4o",
|
||||
output=[],
|
||||
usage={"input_tokens": 4000, "output_tokens": 2000, "total_tokens": 6000},
|
||||
),
|
||||
start_time=datetime.datetime.now(timezone.utc),
|
||||
end_time=datetime.datetime.now(timezone.utc),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def test_spend_log_for_response_retrieval_does_not_replay_the_created_responses_tokens():
|
||||
"""A retrieved response carries the usage of the call that created it, so counting it again
|
||||
bills the same tokens twice. Regression test for LIT-5602."""
|
||||
payload = _spend_log_for_call_type("aget_responses")
|
||||
|
||||
assert payload["prompt_tokens"] == 0
|
||||
assert payload["completion_tokens"] == 0
|
||||
assert payload["total_tokens"] == 0
|
||||
assert payload["spend"] == 0.0
|
||||
|
||||
|
||||
def test_spend_log_for_response_creation_still_counts_tokens():
|
||||
"""Guards the test above: the same response object must still be counted on the create path."""
|
||||
payload = _spend_log_for_call_type("aresponses")
|
||||
|
||||
assert payload["total_tokens"] == 6000
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue