From 2f773d9cb6388c6e1dcd7a742101ecd17506181b Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Thu, 25 Jul 2024 22:11:32 -0700 Subject: [PATCH] fix(litellm_cost_calc/google.py): support meta llama vertex ai cost tracking --- litellm/litellm_core_utils/llm_cost_calc/google.py | 2 +- litellm/proxy/_new_secret_config.yaml | 11 ++--------- litellm/tests/test_amazing_vertex_completion.py | 9 ++++++++- litellm/tests/test_completion_cost.py | 11 +++++++++++ litellm/utils.py | 3 +++ 5 files changed, 25 insertions(+), 11 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/google.py b/litellm/litellm_core_utils/llm_cost_calc/google.py index 76da0da51e4..26eeb7b7a74 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/google.py +++ b/litellm/litellm_core_utils/llm_cost_calc/google.py @@ -44,7 +44,7 @@ def cost_router( Returns - str, the specific google cost calc function it should route to. """ - if custom_llm_provider == "vertex_ai" and "claude" in model: + if custom_llm_provider == "vertex_ai" and ("claude" in model or "llama" in model): return "cost_per_token" elif custom_llm_provider == "gemini": return "cost_per_token" diff --git a/litellm/proxy/_new_secret_config.yaml b/litellm/proxy/_new_secret_config.yaml index 173624c252f..f4a89cc3ab7 100644 --- a/litellm/proxy/_new_secret_config.yaml +++ b/litellm/proxy/_new_secret_config.yaml @@ -1,11 +1,4 @@ model_list: - - model_name: "test-model" + - model_name: "gpt-3.5-turbo" litellm_params: - model: "openai/text-embedding-ada-002" - - model_name: "my-custom-model" - litellm_params: - model: "my-custom-llm/my-model" - -litellm_settings: - custom_provider_map: - - {"provider": "my-custom-llm", "custom_handler": custom_handler.my_custom_llm} + model: "openai/gpt-3.5-turbo" diff --git a/litellm/tests/test_amazing_vertex_completion.py b/litellm/tests/test_amazing_vertex_completion.py index b9762afcbfd..aa0ea471ad9 100644 --- a/litellm/tests/test_amazing_vertex_completion.py +++ b/litellm/tests/test_amazing_vertex_completion.py @@ -901,7 +901,12 @@ from litellm.tests.test_completion import response_format_tests @pytest.mark.parametrize( "model", ["vertex_ai/meta/llama3-405b-instruct-maas"] ) # "vertex_ai", -@pytest.mark.parametrize("sync_mode", [True, False]) # "vertex_ai", +@pytest.mark.parametrize( + "sync_mode", + [ + True, + ], +) # False @pytest.mark.asyncio async def test_llama_3_httpx(model, sync_mode): try: @@ -932,6 +937,8 @@ async def test_llama_3_httpx(model, sync_mode): response_format_tests(response=response) print(f"response: {response}") + + assert False except litellm.RateLimitError as e: pass except Exception as e: diff --git a/litellm/tests/test_completion_cost.py b/litellm/tests/test_completion_cost.py index 289e200d904..41448bd562b 100644 --- a/litellm/tests/test_completion_cost.py +++ b/litellm/tests/test_completion_cost.py @@ -907,6 +907,17 @@ def test_vertex_ai_gemini_predict_cost(): assert predictive_cost > 0 +def test_vertex_ai_llama_predict_cost(): + model = "meta/llama3-405b-instruct-maas" + messages = [{"role": "user", "content": "Hey, hows it going???"}] + custom_llm_provider = "vertex_ai" + predictive_cost = completion_cost( + model=model, messages=messages, custom_llm_provider=custom_llm_provider + ) + + assert predictive_cost == 0 + + @pytest.mark.parametrize("model", ["openai/tts-1", "azure/tts-1"]) def test_completion_cost_tts(model): os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" diff --git a/litellm/utils.py b/litellm/utils.py index eecc704b719..7c22953bc2a 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -4919,6 +4919,9 @@ def get_model_info(model: str, custom_llm_provider: Optional[str] = None) -> Mod azure_llms = litellm.azure_llms if model in azure_llms: model = azure_llms[model] + if custom_llm_provider is not None and custom_llm_provider == "vertex_ai": + if "meta/" + model in litellm.vertex_llama3_models: + model = "meta/" + model ########################## if custom_llm_provider is None: # Get custom_llm_provider