mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge pull request #9509 from BerriAI/litellm_dev_03_24_2025_p1
Log 'api_base' on spend logs
This commit is contained in:
commit
a2ed9e4b80
5 changed files with 339 additions and 14 deletions
|
|
@ -518,6 +518,16 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
}
|
||||
return data
|
||||
|
||||
def _get_masked_api_base(self, api_base: str) -> str:
|
||||
if "key=" in api_base:
|
||||
# Find the position of "key=" in the string
|
||||
key_index = api_base.find("key=") + 4
|
||||
# Mask the last 5 characters after "key="
|
||||
masked_api_base = api_base[:key_index] + "*" * 5 + api_base[-4:]
|
||||
else:
|
||||
masked_api_base = api_base
|
||||
return str(masked_api_base)
|
||||
|
||||
def _pre_call(self, input, api_key, model=None, additional_args={}):
|
||||
"""
|
||||
Common helper function across the sync + async pre-call function
|
||||
|
|
@ -531,6 +541,9 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
model
|
||||
): # if model name was changes pre-call, overwrite the initial model call name with the new one
|
||||
self.model_call_details["model"] = model
|
||||
self.model_call_details["litellm_params"]["api_base"] = (
|
||||
self._get_masked_api_base(additional_args.get("api_base", ""))
|
||||
)
|
||||
|
||||
def pre_call(self, input, api_key, model=None, additional_args={}): # noqa: PLR0915
|
||||
|
||||
|
|
@ -714,15 +727,6 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
headers = {}
|
||||
data = additional_args.get("complete_input_dict", {})
|
||||
api_base = str(additional_args.get("api_base", ""))
|
||||
if "key=" in api_base:
|
||||
# Find the position of "key=" in the string
|
||||
key_index = api_base.find("key=") + 4
|
||||
# Mask the last 5 characters after "key="
|
||||
masked_api_base = api_base[:key_index] + "*" * 5 + api_base[-4:]
|
||||
else:
|
||||
masked_api_base = api_base
|
||||
self.model_call_details["litellm_params"]["api_base"] = masked_api_base
|
||||
|
||||
curl_command = self._get_request_curl_command(
|
||||
api_base=api_base,
|
||||
headers=headers,
|
||||
|
|
@ -737,11 +741,12 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
def _get_request_curl_command(
|
||||
self, api_base: str, headers: Optional[dict], additional_args: dict, data: dict
|
||||
) -> str:
|
||||
masked_api_base = self._get_masked_api_base(api_base)
|
||||
if headers is None:
|
||||
headers = {}
|
||||
curl_command = "\n\nPOST Request Sent from LiteLLM:\n"
|
||||
curl_command += "curl -X POST \\\n"
|
||||
curl_command += f"{api_base} \\\n"
|
||||
curl_command += f"{masked_api_base} \\\n"
|
||||
masked_headers = self._get_masked_headers(headers)
|
||||
formatted_headers = " ".join(
|
||||
[f"-H '{k}: {v}'" for k, v in masked_headers.items()]
|
||||
|
|
|
|||
|
|
@ -138,13 +138,22 @@ class ModelParamHelper:
|
|||
TranscriptionCreateParamsNonStreaming,
|
||||
TranscriptionCreateParamsStreaming,
|
||||
)
|
||||
non_streaming_kwargs = set(getattr(TranscriptionCreateParamsNonStreaming, "__annotations__", {}).keys())
|
||||
streaming_kwargs = set(getattr(TranscriptionCreateParamsStreaming, "__annotations__", {}).keys())
|
||||
|
||||
non_streaming_kwargs = set(
|
||||
getattr(
|
||||
TranscriptionCreateParamsNonStreaming, "__annotations__", {}
|
||||
).keys()
|
||||
)
|
||||
streaming_kwargs = set(
|
||||
getattr(
|
||||
TranscriptionCreateParamsStreaming, "__annotations__", {}
|
||||
).keys()
|
||||
)
|
||||
|
||||
all_transcription_kwargs = non_streaming_kwargs.union(streaming_kwargs)
|
||||
return all_transcription_kwargs
|
||||
except Exception as e:
|
||||
verbose_logger.warning("Error getting transcription kwargs %s", str(e))
|
||||
verbose_logger.debug("Error getting transcription kwargs %s", str(e))
|
||||
return set()
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -5,7 +5,10 @@ model_list:
|
|||
api_key: os.environ/AZURE_API_KEY
|
||||
api_base: http://0.0.0.0:8090
|
||||
rpm: 3
|
||||
|
||||
- model_name: "gpt-4o-mini-openai"
|
||||
litellm_params:
|
||||
model: gpt-4o-mini
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
litellm_settings:
|
||||
num_retries: 0
|
||||
|
||||
|
|
|
|||
34
tests/litellm/litellm_core_utils/test_litellm_logging.py
Normal file
34
tests/litellm/litellm_core_utils/test_litellm_logging.py
Normal file
|
|
@ -0,0 +1,34 @@
|
|||
import json
|
||||
import os
|
||||
import sys
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath("../../..")
|
||||
) # Adds the parent directory to the system path
|
||||
|
||||
import time
|
||||
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LitellmLogging
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def logging_obj():
|
||||
return LitellmLogging(
|
||||
model="bedrock/claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=[{"role": "user", "content": "Hey"}],
|
||||
stream=True,
|
||||
call_type="completion",
|
||||
start_time=time.time(),
|
||||
litellm_call_id="12345",
|
||||
function_id="1245",
|
||||
)
|
||||
|
||||
|
||||
def test_get_masked_api_base(logging_obj):
|
||||
api_base = "https://api.openai.com/v1"
|
||||
masked_api_base = logging_obj._get_masked_api_base(api_base)
|
||||
assert masked_api_base == "https://api.openai.com/v1"
|
||||
assert type(masked_api_base) == str
|
||||
|
|
@ -1,3 +1,4 @@
|
|||
import asyncio
|
||||
import datetime
|
||||
import json
|
||||
import os
|
||||
|
|
@ -11,7 +12,13 @@ sys.path.insert(
|
|||
0, os.path.abspath("../../../..")
|
||||
) # Adds the parent directory to the system path
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import litellm
|
||||
from litellm.proxy._types import SpendLogsPayload
|
||||
from litellm.proxy.hooks.proxy_track_cost_callback import _ProxyDBLogger
|
||||
from litellm.proxy.proxy_server import app, prisma_client
|
||||
from litellm.router import Router
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
|
|
@ -400,3 +407,270 @@ async def test_ui_view_spend_logs_unauthorized(client):
|
|||
headers={"Authorization": "Bearer invalid-token"},
|
||||
)
|
||||
assert response.status_code == 401 or response.status_code == 403
|
||||
|
||||
|
||||
class TestSpendLogsPayload:
|
||||
@pytest.mark.asyncio
|
||||
async def test_spend_logs_payload_e2e(self):
|
||||
litellm.callbacks = [_ProxyDBLogger(message_logging=False)]
|
||||
# litellm._turn_on_debug()
|
||||
|
||||
with patch.object(
|
||||
litellm.proxy.proxy_server, "_set_spend_logs_payload"
|
||||
) as mock_client, patch.object(litellm.proxy.proxy_server, "prisma_client"):
|
||||
response = await litellm.acompletion(
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
mock_response="Hello, world!",
|
||||
metadata={"user_api_key_end_user_id": "test_user_1"},
|
||||
)
|
||||
|
||||
assert response.choices[0].message.content == "Hello, world!"
|
||||
|
||||
await asyncio.sleep(1)
|
||||
|
||||
mock_client.assert_called_once()
|
||||
|
||||
kwargs = mock_client.call_args.kwargs
|
||||
payload: SpendLogsPayload = kwargs["payload"]
|
||||
expected_payload = SpendLogsPayload(
|
||||
**{
|
||||
"request_id": "chatcmpl-34df56d5-4807-45c1-bb99-61e52586b802",
|
||||
"call_type": "acompletion",
|
||||
"api_key": "",
|
||||
"cache_hit": "None",
|
||||
"startTime": datetime.datetime(
|
||||
2025, 3, 24, 22, 2, 42, 975883, tzinfo=datetime.timezone.utc
|
||||
),
|
||||
"endTime": datetime.datetime(
|
||||
2025, 3, 24, 22, 2, 42, 989132, tzinfo=datetime.timezone.utc
|
||||
),
|
||||
"completionStartTime": datetime.datetime(
|
||||
2025, 3, 24, 22, 2, 42, 989132, tzinfo=datetime.timezone.utc
|
||||
),
|
||||
"model": "gpt-4o",
|
||||
"user": "",
|
||||
"team_id": "",
|
||||
"metadata": '{"applied_guardrails": [], "batch_models": null, "additional_usage_values": {"completion_tokens_details": null, "prompt_tokens_details": null}}',
|
||||
"cache_key": "Cache OFF",
|
||||
"spend": 0.00022500000000000002,
|
||||
"total_tokens": 30,
|
||||
"prompt_tokens": 10,
|
||||
"completion_tokens": 20,
|
||||
"request_tags": "[]",
|
||||
"end_user": "test_user_1",
|
||||
"api_base": "",
|
||||
"model_group": "",
|
||||
"model_id": "",
|
||||
"requester_ip_address": None,
|
||||
"custom_llm_provider": "openai",
|
||||
"messages": "{}",
|
||||
"response": "{}",
|
||||
}
|
||||
)
|
||||
|
||||
for key, value in expected_payload.items():
|
||||
if key in [
|
||||
"request_id",
|
||||
"startTime",
|
||||
"endTime",
|
||||
"completionStartTime",
|
||||
"endTime",
|
||||
]:
|
||||
assert payload[key] is not None
|
||||
else:
|
||||
assert (
|
||||
payload[key] == value
|
||||
), f"Expected {key} to be {value}, but got {payload[key]}"
|
||||
|
||||
def mock_anthropic_response(*args, **kwargs):
|
||||
mock_response = MagicMock()
|
||||
mock_response.status_code = 200
|
||||
mock_response.headers = {"Content-Type": "application/json"}
|
||||
mock_response.json.return_value = {
|
||||
"content": [{"text": "Hi! My name is Claude.", "type": "text"}],
|
||||
"id": "msg_013Zva2CMHLNnXjNJJKqJ2EF",
|
||||
"model": "claude-3-7-sonnet-20250219",
|
||||
"role": "assistant",
|
||||
"stop_reason": "end_turn",
|
||||
"stop_sequence": None,
|
||||
"type": "message",
|
||||
"usage": {"input_tokens": 2095, "output_tokens": 503},
|
||||
}
|
||||
return mock_response
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_spend_logs_payload_success_log_with_api_base(self):
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
|
||||
|
||||
litellm.callbacks = [_ProxyDBLogger(message_logging=False)]
|
||||
# litellm._turn_on_debug()
|
||||
|
||||
client = AsyncHTTPHandler()
|
||||
|
||||
with patch.object(
|
||||
litellm.proxy.proxy_server, "_set_spend_logs_payload"
|
||||
) as mock_client, patch.object(
|
||||
litellm.proxy.proxy_server, "prisma_client"
|
||||
), patch.object(
|
||||
client, "post", side_effect=self.mock_anthropic_response
|
||||
):
|
||||
response = await litellm.acompletion(
|
||||
model="claude-3-7-sonnet-20250219",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
metadata={"user_api_key_end_user_id": "test_user_1"},
|
||||
client=client,
|
||||
)
|
||||
|
||||
assert response.choices[0].message.content == "Hi! My name is Claude."
|
||||
|
||||
await asyncio.sleep(1)
|
||||
|
||||
mock_client.assert_called_once()
|
||||
|
||||
kwargs = mock_client.call_args.kwargs
|
||||
payload: SpendLogsPayload = kwargs["payload"]
|
||||
expected_payload = SpendLogsPayload(
|
||||
**{
|
||||
"request_id": "chatcmpl-34df56d5-4807-45c1-bb99-61e52586b802",
|
||||
"call_type": "acompletion",
|
||||
"api_key": "",
|
||||
"cache_hit": "None",
|
||||
"startTime": datetime.datetime(
|
||||
2025, 3, 24, 22, 2, 42, 975883, tzinfo=datetime.timezone.utc
|
||||
),
|
||||
"endTime": datetime.datetime(
|
||||
2025, 3, 24, 22, 2, 42, 989132, tzinfo=datetime.timezone.utc
|
||||
),
|
||||
"completionStartTime": datetime.datetime(
|
||||
2025, 3, 24, 22, 2, 42, 989132, tzinfo=datetime.timezone.utc
|
||||
),
|
||||
"model": "claude-3-7-sonnet-20250219",
|
||||
"user": "",
|
||||
"team_id": "",
|
||||
"metadata": '{"applied_guardrails": [], "batch_models": null, "additional_usage_values": {"completion_tokens_details": null, "prompt_tokens_details": {"audio_tokens": null, "cached_tokens": 0, "text_tokens": null, "image_tokens": null}, "cache_creation_input_tokens": 0, "cache_read_input_tokens": 0}}',
|
||||
"cache_key": "Cache OFF",
|
||||
"spend": 0.01383,
|
||||
"total_tokens": 2598,
|
||||
"prompt_tokens": 2095,
|
||||
"completion_tokens": 503,
|
||||
"request_tags": "[]",
|
||||
"end_user": "test_user_1",
|
||||
"api_base": "https://api.anthropic.com/v1/messages",
|
||||
"model_group": "",
|
||||
"model_id": "",
|
||||
"requester_ip_address": None,
|
||||
"custom_llm_provider": "anthropic",
|
||||
"messages": "{}",
|
||||
"response": "{}",
|
||||
}
|
||||
)
|
||||
|
||||
for key, value in expected_payload.items():
|
||||
if key in [
|
||||
"request_id",
|
||||
"startTime",
|
||||
"endTime",
|
||||
"completionStartTime",
|
||||
"endTime",
|
||||
]:
|
||||
assert payload[key] is not None
|
||||
else:
|
||||
assert (
|
||||
payload[key] == value
|
||||
), f"Expected {key} to be {value}, but got {payload[key]}"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_spend_logs_payload_success_log_with_router(self):
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
|
||||
|
||||
litellm.callbacks = [_ProxyDBLogger(message_logging=False)]
|
||||
# litellm._turn_on_debug()
|
||||
|
||||
client = AsyncHTTPHandler()
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "my-anthropic-model-group",
|
||||
"litellm_params": {
|
||||
"model": "claude-3-7-sonnet-20250219",
|
||||
},
|
||||
"model_info": {
|
||||
"id": "my-unique-model-id",
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
with patch.object(
|
||||
litellm.proxy.proxy_server, "_set_spend_logs_payload"
|
||||
) as mock_client, patch.object(
|
||||
litellm.proxy.proxy_server, "prisma_client"
|
||||
), patch.object(
|
||||
client, "post", side_effect=self.mock_anthropic_response
|
||||
):
|
||||
response = await router.acompletion(
|
||||
model="my-anthropic-model-group",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
metadata={"user_api_key_end_user_id": "test_user_1"},
|
||||
client=client,
|
||||
)
|
||||
|
||||
assert response.choices[0].message.content == "Hi! My name is Claude."
|
||||
|
||||
await asyncio.sleep(1)
|
||||
|
||||
mock_client.assert_called_once()
|
||||
|
||||
kwargs = mock_client.call_args.kwargs
|
||||
payload: SpendLogsPayload = kwargs["payload"]
|
||||
expected_payload = SpendLogsPayload(
|
||||
**{
|
||||
"request_id": "chatcmpl-34df56d5-4807-45c1-bb99-61e52586b802",
|
||||
"call_type": "acompletion",
|
||||
"api_key": "",
|
||||
"cache_hit": "None",
|
||||
"startTime": datetime.datetime(
|
||||
2025, 3, 24, 22, 2, 42, 975883, tzinfo=datetime.timezone.utc
|
||||
),
|
||||
"endTime": datetime.datetime(
|
||||
2025, 3, 24, 22, 2, 42, 989132, tzinfo=datetime.timezone.utc
|
||||
),
|
||||
"completionStartTime": datetime.datetime(
|
||||
2025, 3, 24, 22, 2, 42, 989132, tzinfo=datetime.timezone.utc
|
||||
),
|
||||
"model": "claude-3-7-sonnet-20250219",
|
||||
"user": "",
|
||||
"team_id": "",
|
||||
"metadata": '{"applied_guardrails": [], "batch_models": null, "additional_usage_values": {"completion_tokens_details": null, "prompt_tokens_details": {"audio_tokens": null, "cached_tokens": 0, "text_tokens": null, "image_tokens": null}, "cache_creation_input_tokens": 0, "cache_read_input_tokens": 0}}',
|
||||
"cache_key": "Cache OFF",
|
||||
"spend": 0.01383,
|
||||
"total_tokens": 2598,
|
||||
"prompt_tokens": 2095,
|
||||
"completion_tokens": 503,
|
||||
"request_tags": "[]",
|
||||
"end_user": "test_user_1",
|
||||
"api_base": "https://api.anthropic.com/v1/messages",
|
||||
"model_group": "my-anthropic-model-group",
|
||||
"model_id": "my-unique-model-id",
|
||||
"requester_ip_address": None,
|
||||
"custom_llm_provider": "anthropic",
|
||||
"messages": "{}",
|
||||
"response": "{}",
|
||||
}
|
||||
)
|
||||
|
||||
for key, value in expected_payload.items():
|
||||
if key in [
|
||||
"request_id",
|
||||
"startTime",
|
||||
"endTime",
|
||||
"completionStartTime",
|
||||
"endTime",
|
||||
]:
|
||||
assert payload[key] is not None
|
||||
else:
|
||||
assert (
|
||||
payload[key] == value
|
||||
), f"Expected {key} to be {value}, but got {payload[key]}"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue