From d4288b134ba32c2bc8a1085bdfa5f55b6825797b Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 11 May 2024 14:24:48 -0700 Subject: [PATCH 1/9] fix - use csv list for batch completions --- litellm/proxy/proxy_server.py | 5 +++-- tests/test_openai_endpoints.py | 5 +---- 2 files changed, 4 insertions(+), 6 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 0ae498baa9f..210912fcfc7 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -3673,8 +3673,9 @@ async def chat_completion( # skip router if user passed their key if "api_key" in data: tasks.append(litellm.acompletion(**data)) - elif isinstance(data["model"], list) and llm_router is not None: - _models = data.pop("model") + elif "," in data["model"] and llm_router is not None: + _models_csv_string = data.pop("model") + _models = _models_csv_string.split(",") tasks.append(llm_router.abatch_completion(models=_models, **data)) elif "user_config" in data: # initialize a new router instance. make request using this Router diff --git a/tests/test_openai_endpoints.py b/tests/test_openai_endpoints.py index 7bc97ca5930..43dcae3cd7b 100644 --- a/tests/test_openai_endpoints.py +++ b/tests/test_openai_endpoints.py @@ -424,10 +424,7 @@ async def test_batch_chat_completions(): response = await chat_completion( session=session, key="sk-1234", - model=[ - "gpt-3.5-turbo", - "fake-openai-endpoint", - ], + model="gpt-3.5-turbo,fake-openai-endpoint", ) print(f"response: {response}") From bf2194d7fc5af49acc24f0e505bd14322f92fe3e Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 11 May 2024 14:27:20 -0700 Subject: [PATCH 2/9] feat - support model as csv on proxy --- docs/my-website/docs/proxy/user_keys.md | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/docs/my-website/docs/proxy/user_keys.md b/docs/my-website/docs/proxy/user_keys.md index 7aba832eb8a..d86feb29c37 100644 --- a/docs/my-website/docs/proxy/user_keys.md +++ b/docs/my-website/docs/proxy/user_keys.md @@ -365,12 +365,14 @@ curl --location 'http://0.0.0.0:4000/moderations' \ ## Advanced -### (BETA) Batch Completions - pass `model` as List +### (BETA) Batch Completions - pass multiple models Use this when you want to send 1 request to N Models #### Expected Request Format +Pass model as a string of comma separated value of models. Example `"model"="llama3,gpt-3.5-turbo"` + This same request will be sent to the following model groups on the [litellm proxy config.yaml](https://docs.litellm.ai/docs/proxy/configs) - `model_name="llama3"` - `model_name="gpt-3.5-turbo"` @@ -380,7 +382,7 @@ curl --location 'http://localhost:4000/chat/completions' \ --header 'Authorization: Bearer sk-1234' \ --header 'Content-Type: application/json' \ --data '{ - "model": ["llama3", "gpt-3.5-turbo"], + "model": "llama3,gpt-3.5-turbo", "max_tokens": 10, "user": "litellm2", "messages": [ From 25febe41c470b32b1dddd57ea64b199bdd18f9a6 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 11 May 2024 14:37:32 -0700 Subject: [PATCH 3/9] docs - using batch completions with python --- docs/my-website/docs/proxy/user_keys.md | 96 +++++++++++++++++++++++++ 1 file changed, 96 insertions(+) diff --git a/docs/my-website/docs/proxy/user_keys.md b/docs/my-website/docs/proxy/user_keys.md index d86feb29c37..cda3a46af9f 100644 --- a/docs/my-website/docs/proxy/user_keys.md +++ b/docs/my-website/docs/proxy/user_keys.md @@ -377,6 +377,95 @@ This same request will be sent to the following model groups on the [litellm pro - `model_name="llama3"` - `model_name="gpt-3.5-turbo"` + + + + + +```python +import openai + +client = openai.OpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000") + +response = client.chat.completions.create( + model="gpt-3.5-turbo,llama3", + messages=[ + {"role": "user", "content": "this is a test request, write a short poem"} + ], +) + +print(response) +``` + + + +#### Expected Response Format + +Get a list of responses when `model` is passed as a list + +```python +[ + ChatCompletion( + id='chatcmpl-9NoYhS2G0fswot0b6QpoQgmRQMaIf', + choices=[ + Choice( + finish_reason='stop', + index=0, + logprobs=None, + message=ChatCompletionMessage( + content='In the depths of my soul, a spark ignites\nA light that shines so pure and bright\nIt dances and leaps, refusing to die\nA flame of hope that reaches the sky\n\nIt warms my heart and fills me with bliss\nA reminder that in darkness, there is light to kiss\nSo I hold onto this fire, this guiding light\nAnd let it lead me through the darkest night.', + role='assistant', + function_call=None, + tool_calls=None + ) + ) + ], + created=1715462919, + model='gpt-3.5-turbo-0125', + object='chat.completion', + system_fingerprint=None, + usage=CompletionUsage( + completion_tokens=83, + prompt_tokens=17, + total_tokens=100 + ) + ), + ChatCompletion( + id='chatcmpl-4ac3e982-da4e-486d-bddb-ed1d5cb9c03c', + choices=[ + Choice( + finish_reason='stop', + index=0, + logprobs=None, + message=ChatCompletionMessage( + content="A test request, and I'm delighted!\nHere's a short poem, just for you:\n\nMoonbeams dance upon the sea,\nA path of light, for you to see.\nThe stars up high, a twinkling show,\nA night of wonder, for all to know.\n\nThe world is quiet, save the night,\nA peaceful hush, a gentle light.\nThe world is full, of beauty rare,\nA treasure trove, beyond compare.\n\nI hope you enjoyed this little test,\nA poem born, of whimsy and jest.\nLet me know, if there's anything else!", + role='assistant', + function_call=None, + tool_calls=None + ) + ) + ], + created=1715462919, + model='groq/llama3-8b-8192', + object='chat.completion', + system_fingerprint='fp_a2c8d063cb', + usage=CompletionUsage( + completion_tokens=120, + prompt_tokens=20, + total_tokens=140 + ) + ) +] +``` + + + + + + + + + ```shell curl --location 'http://localhost:4000/chat/completions' \ --header 'Authorization: Bearer sk-1234' \ @@ -395,6 +484,8 @@ curl --location 'http://localhost:4000/chat/completions' \ ``` + + #### Expected Response Format Get a list of responses when `model` is passed as a list @@ -449,6 +540,11 @@ Get a list of responses when `model` is passed as a list ``` + + + + + ### Pass User LLM API Keys, Fallbacks From c3293474dda4bda2bfb96d277d3df26e9a79fdff Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Mon, 13 May 2024 08:46:44 -0700 Subject: [PATCH 4/9] fix(proxy_server.py): return 'allowed-model-region' in headers --- litellm/__init__.py | 126 ++++++++++++------------ litellm/proxy/_super_secret_config.yaml | 2 +- litellm/proxy/proxy_server.py | 16 +++ 3 files changed, 80 insertions(+), 64 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 6c7b26617e1..08b3a70ef12 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -406,69 +406,69 @@ replicate_models: List = [ ] clarifai_models: List = [ - 'clarifai/meta.Llama-3.Llama-3-8B-Instruct', - 'clarifai/gcp.generate.gemma-1_1-7b-it', - 'clarifai/mistralai.completion.mixtral-8x22B', - 'clarifai/cohere.generate.command-r-plus', - 'clarifai/databricks.drbx.dbrx-instruct', - 'clarifai/mistralai.completion.mistral-large', - 'clarifai/mistralai.completion.mistral-medium', - 'clarifai/mistralai.completion.mistral-small', - 'clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1', - 'clarifai/gcp.generate.gemma-2b-it', - 'clarifai/gcp.generate.gemma-7b-it', - 'clarifai/deci.decilm.deciLM-7B-instruct', - 'clarifai/mistralai.completion.mistral-7B-Instruct', - 'clarifai/gcp.generate.gemini-pro', - 'clarifai/anthropic.completion.claude-v1', - 'clarifai/anthropic.completion.claude-instant-1_2', - 'clarifai/anthropic.completion.claude-instant', - 'clarifai/anthropic.completion.claude-v2', - 'clarifai/anthropic.completion.claude-2_1', - 'clarifai/meta.Llama-2.codeLlama-70b-Python', - 'clarifai/meta.Llama-2.codeLlama-70b-Instruct', - 'clarifai/openai.completion.gpt-3_5-turbo-instruct', - 'clarifai/meta.Llama-2.llama2-7b-chat', - 'clarifai/meta.Llama-2.llama2-13b-chat', - 'clarifai/meta.Llama-2.llama2-70b-chat', - 'clarifai/openai.chat-completion.gpt-4-turbo', - 'clarifai/microsoft.text-generation.phi-2', - 'clarifai/meta.Llama-2.llama2-7b-chat-vllm', - 'clarifai/upstage.solar.solar-10_7b-instruct', - 'clarifai/openchat.openchat.openchat-3_5-1210', - 'clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B', - 'clarifai/gcp.generate.text-bison', - 'clarifai/meta.Llama-2.llamaGuard-7b', - 'clarifai/fblgit.una-cybertron.una-cybertron-7b-v2', - 'clarifai/openai.chat-completion.GPT-4', - 'clarifai/openai.chat-completion.GPT-3_5-turbo', - 'clarifai/ai21.complete.Jurassic2-Grande', - 'clarifai/ai21.complete.Jurassic2-Grande-Instruct', - 'clarifai/ai21.complete.Jurassic2-Jumbo-Instruct', - 'clarifai/ai21.complete.Jurassic2-Jumbo', - 'clarifai/ai21.complete.Jurassic2-Large', - 'clarifai/cohere.generate.cohere-generate-command', - 'clarifai/wizardlm.generate.wizardCoder-Python-34B', - 'clarifai/wizardlm.generate.wizardLM-70B', - 'clarifai/tiiuae.falcon.falcon-40b-instruct', - 'clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat', - 'clarifai/gcp.generate.code-gecko', - 'clarifai/gcp.generate.code-bison', - 'clarifai/mistralai.completion.mistral-7B-OpenOrca', - 'clarifai/mistralai.completion.openHermes-2-mistral-7B', - 'clarifai/wizardlm.generate.wizardLM-13B', - 'clarifai/huggingface-research.zephyr.zephyr-7B-alpha', - 'clarifai/wizardlm.generate.wizardCoder-15B', - 'clarifai/microsoft.text-generation.phi-1_5', - 'clarifai/databricks.Dolly-v2.dolly-v2-12b', - 'clarifai/bigcode.code.StarCoder', - 'clarifai/salesforce.xgen.xgen-7b-8k-instruct', - 'clarifai/mosaicml.mpt.mpt-7b-instruct', - 'clarifai/anthropic.completion.claude-3-opus', - 'clarifai/anthropic.completion.claude-3-sonnet', - 'clarifai/gcp.generate.gemini-1_5-pro', - 'clarifai/gcp.generate.imagen-2', - 'clarifai/salesforce.blip.general-english-image-caption-blip-2', + "clarifai/meta.Llama-3.Llama-3-8B-Instruct", + "clarifai/gcp.generate.gemma-1_1-7b-it", + "clarifai/mistralai.completion.mixtral-8x22B", + "clarifai/cohere.generate.command-r-plus", + "clarifai/databricks.drbx.dbrx-instruct", + "clarifai/mistralai.completion.mistral-large", + "clarifai/mistralai.completion.mistral-medium", + "clarifai/mistralai.completion.mistral-small", + "clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1", + "clarifai/gcp.generate.gemma-2b-it", + "clarifai/gcp.generate.gemma-7b-it", + "clarifai/deci.decilm.deciLM-7B-instruct", + "clarifai/mistralai.completion.mistral-7B-Instruct", + "clarifai/gcp.generate.gemini-pro", + "clarifai/anthropic.completion.claude-v1", + "clarifai/anthropic.completion.claude-instant-1_2", + "clarifai/anthropic.completion.claude-instant", + "clarifai/anthropic.completion.claude-v2", + "clarifai/anthropic.completion.claude-2_1", + "clarifai/meta.Llama-2.codeLlama-70b-Python", + "clarifai/meta.Llama-2.codeLlama-70b-Instruct", + "clarifai/openai.completion.gpt-3_5-turbo-instruct", + "clarifai/meta.Llama-2.llama2-7b-chat", + "clarifai/meta.Llama-2.llama2-13b-chat", + "clarifai/meta.Llama-2.llama2-70b-chat", + "clarifai/openai.chat-completion.gpt-4-turbo", + "clarifai/microsoft.text-generation.phi-2", + "clarifai/meta.Llama-2.llama2-7b-chat-vllm", + "clarifai/upstage.solar.solar-10_7b-instruct", + "clarifai/openchat.openchat.openchat-3_5-1210", + "clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B", + "clarifai/gcp.generate.text-bison", + "clarifai/meta.Llama-2.llamaGuard-7b", + "clarifai/fblgit.una-cybertron.una-cybertron-7b-v2", + "clarifai/openai.chat-completion.GPT-4", + "clarifai/openai.chat-completion.GPT-3_5-turbo", + "clarifai/ai21.complete.Jurassic2-Grande", + "clarifai/ai21.complete.Jurassic2-Grande-Instruct", + "clarifai/ai21.complete.Jurassic2-Jumbo-Instruct", + "clarifai/ai21.complete.Jurassic2-Jumbo", + "clarifai/ai21.complete.Jurassic2-Large", + "clarifai/cohere.generate.cohere-generate-command", + "clarifai/wizardlm.generate.wizardCoder-Python-34B", + "clarifai/wizardlm.generate.wizardLM-70B", + "clarifai/tiiuae.falcon.falcon-40b-instruct", + "clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat", + "clarifai/gcp.generate.code-gecko", + "clarifai/gcp.generate.code-bison", + "clarifai/mistralai.completion.mistral-7B-OpenOrca", + "clarifai/mistralai.completion.openHermes-2-mistral-7B", + "clarifai/wizardlm.generate.wizardLM-13B", + "clarifai/huggingface-research.zephyr.zephyr-7B-alpha", + "clarifai/wizardlm.generate.wizardCoder-15B", + "clarifai/microsoft.text-generation.phi-1_5", + "clarifai/databricks.Dolly-v2.dolly-v2-12b", + "clarifai/bigcode.code.StarCoder", + "clarifai/salesforce.xgen.xgen-7b-8k-instruct", + "clarifai/mosaicml.mpt.mpt-7b-instruct", + "clarifai/anthropic.completion.claude-3-opus", + "clarifai/anthropic.completion.claude-3-sonnet", + "clarifai/gcp.generate.gemini-1_5-pro", + "clarifai/gcp.generate.imagen-2", + "clarifai/salesforce.blip.general-english-image-caption-blip-2", ] diff --git a/litellm/proxy/_super_secret_config.yaml b/litellm/proxy/_super_secret_config.yaml index 832e351133a..86037caf742 100644 --- a/litellm/proxy/_super_secret_config.yaml +++ b/litellm/proxy/_super_secret_config.yaml @@ -13,10 +13,10 @@ router_settings: redis_host: redis # redis_password: redis_port: 6379 + enable_pre_call_checks: true litellm_settings: set_verbose: True - enable_preview_features: true # service_callback: ["prometheus_system"] # success_callback: ["prometheus"] # failure_callback: ["prometheus"] diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index a9862022f84..bf1ce4720ea 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -3762,6 +3762,7 @@ async def chat_completion( "x-litellm-cache-key": cache_key, "x-litellm-model-api-base": api_base, "x-litellm-version": version, + "x-litellm-model-region": user_api_key_dict.allowed_model_region or "", } selected_data_generator = select_data_generator( response=response, @@ -3778,6 +3779,9 @@ async def chat_completion( fastapi_response.headers["x-litellm-cache-key"] = cache_key fastapi_response.headers["x-litellm-model-api-base"] = api_base fastapi_response.headers["x-litellm-version"] = version + fastapi_response.headers["x-litellm-model-region"] = ( + user_api_key_dict.allowed_model_region or "" + ) ### CALL HOOKS ### - modify outgoing data response = await proxy_logging_obj.post_call_success_hook( @@ -4162,6 +4166,9 @@ async def embeddings( fastapi_response.headers["x-litellm-cache-key"] = cache_key fastapi_response.headers["x-litellm-model-api-base"] = api_base fastapi_response.headers["x-litellm-version"] = version + fastapi_response.headers["x-litellm-model-region"] = ( + user_api_key_dict.allowed_model_region or "" + ) return response except Exception as e: @@ -4331,6 +4338,9 @@ async def image_generation( fastapi_response.headers["x-litellm-cache-key"] = cache_key fastapi_response.headers["x-litellm-model-api-base"] = api_base fastapi_response.headers["x-litellm-version"] = version + fastapi_response.headers["x-litellm-model-region"] = ( + user_api_key_dict.allowed_model_region or "" + ) return response except Exception as e: @@ -4524,6 +4534,9 @@ async def audio_transcriptions( fastapi_response.headers["x-litellm-cache-key"] = cache_key fastapi_response.headers["x-litellm-model-api-base"] = api_base fastapi_response.headers["x-litellm-version"] = version + fastapi_response.headers["x-litellm-model-region"] = ( + user_api_key_dict.allowed_model_region or "" + ) return response except Exception as e: @@ -4699,6 +4712,9 @@ async def moderations( fastapi_response.headers["x-litellm-cache-key"] = cache_key fastapi_response.headers["x-litellm-model-api-base"] = api_base fastapi_response.headers["x-litellm-version"] = version + fastapi_response.headers["x-litellm-model-region"] = ( + user_api_key_dict.allowed_model_region or "" + ) return response except Exception as e: From 5342b3dc05a884f3b82b56ae1e13512f17671a43 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Mon, 13 May 2024 09:04:38 -0700 Subject: [PATCH 5/9] fix(router.py): fix error message to return if pre-call-checks + allowed model region --- litellm/router.py | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index 4c03125005e..ad11dc98e73 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -3259,13 +3259,12 @@ class Router: healthy_deployments.remove(deployment) # filter pre-call checks + _allowed_model_region = ( + request_kwargs.get("allowed_model_region") + if request_kwargs is not None + else None + ) if self.enable_pre_call_checks and messages is not None: - _allowed_model_region = ( - request_kwargs.get("allowed_model_region") - if request_kwargs is not None - else None - ) - if _allowed_model_region == "eu": healthy_deployments = self._pre_call_checks( model=model, @@ -3286,8 +3285,10 @@ class Router: ) if len(healthy_deployments) == 0: + if _allowed_model_region is None: + _allowed_model_region = "n/a" raise ValueError( - f"{RouterErrors.no_deployments_available.value}, passed model={model}" + f"{RouterErrors.no_deployments_available.value}, passed model={model}. Enable pre-call-checks={self.enable_pre_call_checks}, allowed_model_region={_allowed_model_region}" ) if ( From b063ef7a4740a59f93f833564c24ce05232c6655 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Mon, 13 May 2024 09:08:04 -0700 Subject: [PATCH 6/9] =?UTF-8?q?bump:=20version=201.37.5=20=E2=86=92=201.37?= =?UTF-8?q?.6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- pyproject.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index fa525496c97..07003e2f2b3 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [tool.poetry] name = "litellm" -version = "1.37.5" +version = "1.37.6" description = "Library to easily interface with LLM API providers" authors = ["BerriAI"] license = "MIT" @@ -80,7 +80,7 @@ requires = ["poetry-core", "wheel"] build-backend = "poetry.core.masonry.api" [tool.commitizen] -version = "1.37.5" +version = "1.37.6" version_files = [ "pyproject.toml:^version" ] From 13e15777536b4823b3e5cc8c27cc1c1e34583a38 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Mon, 13 May 2024 10:04:43 -0700 Subject: [PATCH 7/9] fix(slack_alerting.py): don't fire spam alerts when backend api call fails --- litellm/integrations/slack_alerting.py | 26 +++++--------- litellm/proxy/_super_secret_config.yaml | 14 +++++++- litellm/router.py | 4 +-- litellm/tests/test_router_fallbacks.py | 46 +++++++++++++++++++++++++ 4 files changed, 69 insertions(+), 21 deletions(-) diff --git a/litellm/integrations/slack_alerting.py b/litellm/integrations/slack_alerting.py index d03922bc1f5..b06a2292084 100644 --- a/litellm/integrations/slack_alerting.py +++ b/litellm/integrations/slack_alerting.py @@ -76,16 +76,14 @@ class SlackAlerting(CustomLogger): internal_usage_cache: Optional[DualCache] = None, alerting_threshold: float = 300, # threshold for slow / hanging llm responses (in seconds) alerting: Optional[List] = [], - alert_types: Optional[ - List[ - Literal[ - "llm_exceptions", - "llm_too_slow", - "llm_requests_hanging", - "budget_alerts", - "db_exceptions", - "daily_reports", - ] + alert_types: List[ + Literal[ + "llm_exceptions", + "llm_too_slow", + "llm_requests_hanging", + "budget_alerts", + "db_exceptions", + "daily_reports", ] ] = [ "llm_exceptions", @@ -812,14 +810,6 @@ Model Info: updated_at=litellm.utils.get_utc_datetime(), ) ) - if "llm_exceptions" in self.alert_types: - original_exception = kwargs.get("exception", None) - - await self.send_alert( - message="LLM API Failure - " + str(original_exception), - level="High", - alert_type="llm_exceptions", - ) async def _run_scheduler_helper(self, llm_router) -> bool: """ diff --git a/litellm/proxy/_super_secret_config.yaml b/litellm/proxy/_super_secret_config.yaml index 86037caf742..b83883bebe7 100644 --- a/litellm/proxy/_super_secret_config.yaml +++ b/litellm/proxy/_super_secret_config.yaml @@ -8,6 +8,16 @@ model_list: base_model: text-embedding-ada-002 mode: embedding model_name: text-embedding-ada-002 +- model_name: gpt-3.5-turbo-012 + litellm_params: + model: gpt-3.5-turbo + api_base: http://0.0.0.0:8080 + api_key: "" +- model_name: gpt-3.5-turbo-0125-preview + litellm_params: + model: azure/chatgpt-v-2 + api_key: os.environ/AZURE_API_KEY + api_base: os.environ/AZURE_API_BASE router_settings: redis_host: redis @@ -17,6 +27,7 @@ router_settings: litellm_settings: set_verbose: True + fallbacks: [{"gpt-3.5-turbo-012": ["gpt-3.5-turbo-0125-preview"]}] # service_callback: ["prometheus_system"] # success_callback: ["prometheus"] # failure_callback: ["prometheus"] @@ -25,4 +36,5 @@ general_settings: enable_jwt_auth: True disable_reset_budget: True proxy_batch_write_at: 60 # 👈 Frequency of batch writing logs to server (in seconds) - routing_strategy: simple-shuffle # Literal["simple-shuffle", "least-busy", "usage-based-routing","latency-based-routing"], default="simple-shuffle" \ No newline at end of file + routing_strategy: simple-shuffle # Literal["simple-shuffle", "least-busy", "usage-based-routing","latency-based-routing"], default="simple-shuffle" + alerting: ["slack"] diff --git a/litellm/router.py b/litellm/router.py index ad11dc98e73..b345a2f25db 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -1413,7 +1413,7 @@ class Router: verbose_router_logger.debug(f"Trying to fallback b/w models") if ( hasattr(e, "status_code") - and e.status_code == 400 + and e.status_code == 400 # type: ignore and not isinstance(e, litellm.ContextWindowExceededError) ): # don't retry a malformed request raise e @@ -3648,7 +3648,7 @@ class Router: ) asyncio.create_task( proxy_logging_obj.slack_alerting_instance.send_alert( - message=f"Router: Cooling down deployment: {_api_base}, for {self.cooldown_time} seconds. Got exception: {str(exception_status)}", + message=f"Router: Cooling down deployment: {_api_base}, for {self.cooldown_time} seconds. Got exception: {str(exception_status)}. Change 'cooldown_time' + 'allowed_failes' under 'Router Settings' on proxy UI, or via config - https://docs.litellm.ai/docs/proxy/reliability#fallbacks--retries--timeouts--cooldowns", alert_type="cooldown_deployment", level="Low", ) diff --git a/litellm/tests/test_router_fallbacks.py b/litellm/tests/test_router_fallbacks.py index c1035e3e005..ce2b014e9cf 100644 --- a/litellm/tests/test_router_fallbacks.py +++ b/litellm/tests/test_router_fallbacks.py @@ -961,3 +961,49 @@ def test_custom_cooldown_times(): except Exception as e: print(e) + + +@pytest.mark.parametrize("sync_mode", [True, False]) +@pytest.mark.asyncio +async def test_service_unavailable_fallbacks(sync_mode): + """ + Initial model - openai + Fallback - azure + + Error - 503, service unavailable + """ + router = Router( + model_list=[ + { + "model_name": "gpt-3.5-turbo-012", + "litellm_params": { + "model": "gpt-3.5-turbo", + "api_key": "anything", + "api_base": "http://0.0.0.0:8080", + }, + }, + { + "model_name": "gpt-3.5-turbo-0125-preview", + "litellm_params": { + "model": "azure/chatgpt-v-2", + "api_key": os.getenv("AZURE_API_KEY"), + "api_version": os.getenv("AZURE_API_VERSION"), + "api_base": os.getenv("AZURE_API_BASE"), + }, + }, + ], + fallbacks=[{"gpt-3.5-turbo-012": ["gpt-3.5-turbo-0125-preview"]}], + ) + + if sync_mode: + response = router.completion( + model="gpt-3.5-turbo-012", + messages=[{"role": "user", "content": "Hey, how's it going?"}], + ) + else: + response = await router.acompletion( + model="gpt-3.5-turbo-012", + messages=[{"role": "user", "content": "Hey, how's it going?"}], + ) + + assert response.model == "gpt-35-turbo" From 7f6e933372ce8f23cf585e8b67ff0880847bface Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Mon, 13 May 2024 10:17:32 -0700 Subject: [PATCH 8/9] fix(router.py): give an 'info' log when fallbacks work successfully --- litellm/router.py | 6 ++++++ litellm/tests/test_router_debug_logs.py | 1 + 2 files changed, 7 insertions(+) diff --git a/litellm/router.py b/litellm/router.py index b345a2f25db..f1e590545c1 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -1444,6 +1444,9 @@ class Router: response = await self.async_function_with_retries( *args, **kwargs ) + verbose_router_logger.info( + "Successful fallback b/w models." + ) return response except Exception as e: pass @@ -1478,6 +1481,9 @@ class Router: response = await self.async_function_with_fallbacks( *args, **kwargs ) + verbose_router_logger.info( + "Successful fallback b/w models." + ) return response except Exception as e: raise e diff --git a/litellm/tests/test_router_debug_logs.py b/litellm/tests/test_router_debug_logs.py index 202038d9798..1d908abe81b 100644 --- a/litellm/tests/test_router_debug_logs.py +++ b/litellm/tests/test_router_debug_logs.py @@ -85,6 +85,7 @@ def test_async_fallbacks(caplog): "litellm.acompletion(model=gpt-3.5-turbo)\x1b[31m Exception OpenAIException - Error code: 401 - {'error': {'message': 'Incorrect API key provided: bad-key. You can find your API key at https://platform.openai.com/account/api-keys.', 'type': 'invalid_request_error', 'param': None, 'code': 'invalid_api_key'}} \nModel: gpt-3.5-turbo\nAPI Base: https://api.openai.com\nMessages: [{'content': 'Hello, how are you?', 'role': 'user'}]\nmodel_group: gpt-3.5-turbo\n\ndeployment: gpt-3.5-turbo\n\x1b[0m", "Falling back to model_group = azure/gpt-3.5-turbo", "litellm.acompletion(model=azure/chatgpt-v-2)\x1b[32m 200 OK\x1b[0m", + "Successful fallback b/w models.", ] # Assert that the captured logs match the expected log messages From 04ae285001d9b7f82ec98ac83ec15e947bea6f77 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Mon, 13 May 2024 10:42:31 -0700 Subject: [PATCH 9/9] fix(vertex_ai.py): support tool call list response async completion --- litellm/llms/vertex_ai.py | 20 ++++++++++++++++--- .../tests/test_amazing_vertex_completion.py | 19 ++++++++++-------- 2 files changed, 28 insertions(+), 11 deletions(-) diff --git a/litellm/llms/vertex_ai.py b/litellm/llms/vertex_ai.py index d3bb2c78ab3..84fec734fd0 100644 --- a/litellm/llms/vertex_ai.py +++ b/litellm/llms/vertex_ai.py @@ -867,6 +867,8 @@ async def async_completion( Add support for acompletion calls for gemini-pro """ try: + import proto # type: ignore + if mode == "vision": print_verbose("\nMaking VertexAI Gemini Pro/Vision Call") print_verbose(f"\nProcessing input messages = {messages}") @@ -901,9 +903,21 @@ async def async_completion( ): function_call = response.candidates[0].content.parts[0].function_call args_dict = {} - for k, v in function_call.args.items(): - args_dict[k] = v - args_str = json.dumps(args_dict) + + # Check if it's a RepeatedComposite instance + for key, val in function_call.args.items(): + if isinstance( + val, proto.marshal.collections.repeated.RepeatedComposite + ): + # If so, convert to list + args_dict[key] = [v for v in val] + else: + args_dict[key] = val + + try: + args_str = json.dumps(args_dict) + except Exception as e: + raise VertexAIError(status_code=422, message=str(e)) message = litellm.Message( content=None, tool_calls=[ diff --git a/litellm/tests/test_amazing_vertex_completion.py b/litellm/tests/test_amazing_vertex_completion.py index a56d7fe5a9a..ce9e6286ff2 100644 --- a/litellm/tests/test_amazing_vertex_completion.py +++ b/litellm/tests/test_amazing_vertex_completion.py @@ -590,19 +590,20 @@ def test_gemini_pro_vision_base64(): pytest.fail(f"An exception occurred - {str(e)}") +@pytest.mark.parametrize("sync_mode", [True, False]) @pytest.mark.asyncio -def test_gemini_pro_function_calling(): +async def test_gemini_pro_function_calling(sync_mode): try: load_vertex_ai_credentials() - response = litellm.completion( - model="vertex_ai/gemini-pro", - messages=[ + data = { + "model": "vertex_ai/gemini-pro", + "messages": [ { "role": "user", "content": "Call the submit_cities function with San Francisco and New York", } ], - tools=[ + "tools": [ { "type": "function", "function": { @@ -618,11 +619,13 @@ def test_gemini_pro_function_calling(): }, } ], - ) + } + if sync_mode: + response = litellm.completion(**data) + else: + response = await litellm.acompletion(**data) print(f"response: {response}") - except litellm.APIError as e: - pass except litellm.RateLimitError as e: pass except Exception as e: