diff --git a/.github/workflows/auto_update_price_and_context_window_file.py b/.github/workflows/auto_update_price_and_context_window_file.py index 3e0731b94bd..461d8d347d9 100644 --- a/.github/workflows/auto_update_price_and_context_window_file.py +++ b/.github/workflows/auto_update_price_and_context_window_file.py @@ -43,8 +43,8 @@ def write_to_file(file_path, data): # Print an error message if writing to file fails print("Error updating JSON file:", e) -# Update the existing models and add the missing models -def transform_remote_data(data): +# Update the existing models and add the missing models for OpenRouter +def transform_openrouter_data(data): transformed = {} for row in data: # Add the fields 'max_tokens' and 'input_cost_per_token' @@ -81,6 +81,34 @@ def transform_remote_data(data): return transformed +# Update the existing models and add the missing models for Vercel AI Gateway +def transform_vercel_ai_gateway_data(data): + transformed = {} + for row in data: + obj = { + "max_tokens": row["context_window"], + "input_cost_per_token": float(row["pricing"]["input"]), + "output_cost_per_token": float(row["pricing"]["output"]), + 'max_output_tokens': row['max_tokens'], + 'max_input_tokens': row["context_window"], + } + + # Handle cache pricing if available + if "pricing" in row: + if "input_cache_read" in row["pricing"] and row["pricing"]["input_cache_read"] is not None: + obj['cache_read_input_token_cost'] = float(f"{float(row['pricing']['input_cache_read']):e}") + + if "input_cache_write" in row["pricing"] and row["pricing"]["input_cache_write"] is not None: + obj['cache_creation_input_token_cost'] = float(f"{float(row['pricing']['input_cache_write']):e}") + + mode = "embedding" if "embedding" in row["id"].lower() else "chat" + + obj.update({"litellm_provider": "vercel_ai_gateway", "mode": mode}) + + transformed[f'vercel_ai_gateway/{row["id"]}'] = obj + + return transformed + # Load local data from a specified file def load_local_data(file_path): @@ -100,22 +128,32 @@ def load_local_data(file_path): def main(): local_file_path = "model_prices_and_context_window.json" # Path to the local data file - url = "https://openrouter.ai/api/v1/models" # URL to fetch remote data + openrouter_url = "https://openrouter.ai/api/v1/models" # URL to fetch OpenRouter data + vercel_ai_gateway_url = "https://ai-gateway.vercel.sh/v1/models" # URL to fetch Vercel AI Gateway data # Load local data from file local_data = load_local_data(local_file_path) - # Fetch remote data asynchronously - remote_data = asyncio.run(fetch_data(url)) - # Transform the fetched remote data - remote_data = transform_remote_data(remote_data) + + # Fetch OpenRouter data + openrouter_data = asyncio.run(fetch_data(openrouter_url)) + # Transform the fetched OpenRouter data + openrouter_data = transform_openrouter_data(openrouter_data) + + # Fetch Vercel AI Gateway data + vercel_data = asyncio.run(fetch_data(vercel_ai_gateway_url)) + # Transform the fetched Vercel AI Gateway data + vercel_data = transform_vercel_ai_gateway_data(vercel_data) + + # Combine both datasets + all_remote_data = {**openrouter_data, **vercel_data} - # If both local and remote data are available, synchronize and save - if local_data and remote_data: - sync_local_data_with_remote(local_data, remote_data) + # If both local and openrouter data are available, synchronize and save + if local_data and all_remote_data: + sync_local_data_with_remote(local_data, all_remote_data) write_to_file(local_file_path, local_data) else: print("Failed to fetch model data from either local file or URL.") # Entry point of the script if __name__ == "__main__": - main() \ No newline at end of file + main() diff --git a/docs/my-website/docs/index.md b/docs/my-website/docs/index.md index 58cabc81b48..d242842c24e 100644 --- a/docs/my-website/docs/index.md +++ b/docs/my-website/docs/index.md @@ -226,6 +226,23 @@ response = completion( + + +```python +from litellm import completion +import os + +## set ENV variables. Visit https://vercel.com/docs/ai-gateway#using-the-ai-gateway-with-an-api-key for insturctions on obtaining a key +os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "your-vercel-api-key" + +response = completion( + model="vercel_ai_gateway/openai/gpt-4o", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) +``` + + + ### Response Format (OpenAI Format) @@ -446,6 +463,24 @@ response = completion( + + +```python +from litellm import completion +import os + +## set ENV variables. Visit https://vercel.com/docs/ai-gateway#using-the-ai-gateway-with-an-api-key for insturctions on obtaining a key +os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "your-vercel-api-key" + +response = completion( + model="vercel_ai_gateway/openai/gpt-4o", + messages = [{ "content": "Hello, how are you?","role": "user"}], + stream=True, +) +``` + + + ### Streaming Response Format (OpenAI Format) diff --git a/docs/my-website/docs/providers/vercel_ai_gateway.md b/docs/my-website/docs/providers/vercel_ai_gateway.md new file mode 100644 index 00000000000..91f0a18ea1c --- /dev/null +++ b/docs/my-website/docs/providers/vercel_ai_gateway.md @@ -0,0 +1,219 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Vercel AI Gateway + +## Overview + +| Property | Details | +|-------|-------| +| Description | Vercel AI Gateway provides a unified interface to access multiple AI providers through a single endpoint, with built-in caching, rate limiting, and analytics. | +| Provider Route on LiteLLM | `vercel_ai_gateway/` | +| Link to Provider Doc | [Vercel AI Gateway Documentation ↗](https://vercel.com/docs/ai-gateway) | +| Base URL | `https://ai-gateway.vercel.sh/v1` | +| Supported Operations | `/chat/completions`, `/models` | + +
+
+ +https://vercel.com/docs/ai-gateway + +**We support ALL models available through Vercel AI Gateway, just set `vercel_ai_gateway/` as a prefix when sending completion requests** + +## Required Variables + +```python showLineNumbers title="Environment Variables" +os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "" # your Vercel AI Gateway API key +# OR +os.environ["VERCEL_OIDC_TOKEN"] = "" # your Vercel OIDC token for authentication +``` + +## Optional Variables + +```python showLineNumbers title="Environment Variables" +os.environ["VERCEL_SITE_URL"] = "" # your site url +# OR +os.environ["VERCEL_APP_NAME"] = "" # your app name +``` + +Note: see the [Vercel AI Gateway docs](https://vercel.com/docs/ai-gateway#using-the-ai-gateway-with-an-api-key) for instructions on obtaining a key. + +## Usage - LiteLLM Python SDK + +### Non-streaming + +```python showLineNumbers title="Vercel AI Gateway Non-streaming Completion" +import os +import litellm +from litellm import completion + +os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "your-api-key" + +messages = [{"content": "Hello, how are you?", "role": "user"}] + +# Vercel AI Gateway call +response = completion( + model="vercel_ai_gateway/openai/gpt-4o", + messages=messages +) + +print(response) +``` + +### Streaming + +```python showLineNumbers title="Vercel AI Gateway Streaming Completion" +import os +import litellm +from litellm import completion + +os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "your-api-key" + +messages = [{"content": "Hello, how are you?", "role": "user"}] + +# Vercel AI Gateway call with streaming +response = completion( + model="vercel_ai_gateway/openai/gpt-4o", + messages=messages, + stream=True +) + +for chunk in response: + print(chunk) +``` + +## Usage - LiteLLM Proxy + +Add the following to your LiteLLM Proxy configuration file: + +```yaml showLineNumbers title="config.yaml" +model_list: + - model_name: gpt-4o-gateway + litellm_params: + model: vercel_ai_gateway/openai/gpt-4o + api_key: os.environ/VERCEL_AI_GATEWAY_API_KEY + + - model_name: claude-4-sonnet-gateway + litellm_params: + model: vercel_ai_gateway/anthropic/claude-4-sonnet + api_key: os.environ/VERCEL_AI_GATEWAY_API_KEY +``` + +Start your LiteLLM Proxy server: + +```bash showLineNumbers title="Start LiteLLM Proxy" +litellm --config config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + + + + +```python showLineNumbers title="Vercel AI Gateway via Proxy - Non-streaming" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-proxy-api-key" # Your proxy API key +) + +# Non-streaming response +response = client.chat.completions.create( + model="gpt-4o-gateway", + messages=[{"role": "user", "content": "Hello, how are you?"}] +) + +print(response.choices[0].message.content) +``` + +```python showLineNumbers title="Vercel AI Gateway via Proxy - Streaming" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-proxy-api-key" # Your proxy API key +) + +# Streaming response +response = client.chat.completions.create( + model="gpt-4o-gateway", + messages=[{"role": "user", "content": "Hello, how are you?"}], + stream=True +) + +for chunk in response: + if chunk.choices[0].delta.content is not None: + print(chunk.choices[0].delta.content, end="") +``` + + + + + +```python showLineNumbers title="Vercel AI Gateway via Proxy - LiteLLM SDK" +import litellm + +# Configure LiteLLM to use your proxy +response = litellm.completion( + model="litellm_proxy/gpt-4o-gateway", + messages=[{"role": "user", "content": "Hello, how are you?"}], + api_base="http://localhost:4000", + api_key="your-proxy-api-key" +) + +print(response.choices[0].message.content) +``` + +```python showLineNumbers title="Vercel AI Gateway via Proxy - LiteLLM SDK Streaming" +import litellm + +# Configure LiteLLM to use your proxy with streaming +response = litellm.completion( + model="litellm_proxy/gpt-4o-gateway", + messages=[{"role": "user", "content": "Hello, how are you?"}], + api_base="http://localhost:4000", + api_key="your-proxy-api-key", + stream=True +) + +for chunk in response: + if hasattr(chunk.choices[0], 'delta') and chunk.choices[0].delta.content is not None: + print(chunk.choices[0].delta.content, end="") +``` + + + + + +```bash showLineNumbers title="Vercel AI Gateway via Proxy - cURL" +curl http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer your-proxy-api-key" \ + -d '{ + "model": "gpt-4o-gateway", + "messages": [{"role": "user", "content": "Hello, how are you?"}] + }' +``` + +```bash showLineNumbers title="Vercel AI Gateway via Proxy - cURL Streaming" +curl http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer your-proxy-api-key" \ + -d '{ + "model": "gpt-4o-gateway", + "messages": [{"role": "user", "content": "Hello, how are you?"}], + "stream": true + }' +``` + + + + +For more detailed information on using the LiteLLM Proxy, see the [LiteLLM Proxy documentation](../providers/litellm_proxy). + +## Additional Resources + +- [Vercel AI Gateway Documentation](https://vercel.com/docs/ai-gateway) diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 49dd0c804ec..19ec9ffc0b2 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -464,6 +464,7 @@ const sidebars = { "providers/replicate", "providers/togetherai", "providers/v0", + "providers/vercel_ai_gateway", "providers/morph", "providers/lambda_ai", "providers/novita", diff --git a/litellm/__init__.py b/litellm/__init__.py index fb3dc37f345..fb280c34101 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -226,6 +226,7 @@ vertex_location: Optional[str] = None predibase_tenant_id: Optional[str] = None togetherai_api_key: Optional[str] = None cloudflare_api_key: Optional[str] = None +vercel_ai_gateway_key: Optional[str] = None baseten_key: Optional[str] = None llama_api_key: Optional[str] = None aleph_alpha_key: Optional[str] = None @@ -542,6 +543,7 @@ hyperbolic_models: Set = set() recraft_models: Set = set() cometapi_models: Set = set() oci_models: Set = set() +vercel_ai_gateway_models: Set = set() def is_bedrock_pricing_only_model(key: str) -> bool: @@ -599,6 +601,8 @@ def add_known_models(): empower_models.add(key) elif value.get("litellm_provider") == "openrouter": openrouter_models.add(key) + elif value.get("litellm_provider") == "vercel_ai_gateway": + vercel_ai_gateway_models.add(key) elif value.get("litellm_provider") == "datarobot": datarobot_models.add(key) elif value.get("litellm_provider") == "vertex_ai-text-models": @@ -835,6 +839,7 @@ model_list = list( | recraft_models | cometapi_models | oci_models + | vercel_ai_gateway_models ) model_list_set = set(model_list) @@ -853,6 +858,7 @@ models_by_provider: dict = { "together_ai": together_ai_models, "baseten": baseten_models, "openrouter": openrouter_models, + "vercel_ai_gateway": vercel_ai_gateway_models, "datarobot": datarobot_models, "vertex_ai": vertex_chat_models | vertex_text_models | vertex_anthropic_models | vertex_vision_models | vertex_language_models | vertex_deepseek_models, "ai21": ai21_models, @@ -1247,6 +1253,7 @@ from .llms.oci.chat.transformation import OCIChatConfig from .llms.morph.chat.transformation import MorphChatConfig from .llms.lambda_ai.chat.transformation import LambdaAIChatConfig from .llms.hyperbolic.chat.transformation import HyperbolicChatConfig +from .llms.vercel_ai_gateway.chat.transformation import VercelAIGatewayConfig from .main import * # type: ignore from .integrations import * from .llms.custom_httpx.async_client_cleanup import close_litellm_async_clients diff --git a/litellm/constants.py b/litellm/constants.py index 34a4f37ad6f..7ddd16c880c 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -289,6 +289,7 @@ LITELLM_CHAT_PROVIDERS = [ "oci", "morph", "lambda_ai", + "vercel_ai_gateway", ] LITELLM_EMBEDDING_PROVIDERS_SUPPORTING_INPUT_ARRAY_OF_TOKENS = [ @@ -421,6 +422,7 @@ openai_compatible_endpoints: List = [ "https://api.morphllm.com/v1", "https://api.lambda.ai/v1", "https://api.hyperbolic.xyz/v1", + "https://ai-gateway.vercel.sh/v1", ] @@ -463,6 +465,7 @@ openai_compatible_providers: List = [ "morph", "lambda_ai", "hyperbolic", + "vercel_ai_gateway", "aiml", ] openai_text_completion_compatible_providers: List = ( diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 2d511741ec6..be00f964f70 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -249,6 +249,9 @@ def get_llm_provider( # noqa: PLR0915 elif endpoint == "https://api.hyperbolic.xyz/v1": custom_llm_provider = "hyperbolic" dynamic_api_key = get_secret_str("HYPERBOLIC_API_KEY") + elif endpoint == "https://ai-gateway.vercel.sh/v1": + custom_llm_provider = "vercel_ai_gateway" + dynamic_api_key = get_secret_str("VERCEL_AI_GATEWAY_API_KEY") if api_base is not None and not isinstance(api_base, str): raise Exception( @@ -742,6 +745,11 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 ) = litellm.HyperbolicChatConfig()._get_openai_compatible_provider_info( api_base, api_key ) + elif custom_llm_provider == "vercel_ai_gateway": + ( + api_base, + dynamic_api_key, + ) = litellm.VercelAIGatewayConfig()._get_openai_compatible_provider_info( elif custom_llm_provider == "aiml": ( api_base, diff --git a/litellm/litellm_core_utils/get_supported_openai_params.py b/litellm/litellm_core_utils/get_supported_openai_params.py index cf39411f663..86535943762 100644 --- a/litellm/litellm_core_utils/get_supported_openai_params.py +++ b/litellm/litellm_core_utils/get_supported_openai_params.py @@ -131,6 +131,8 @@ def get_supported_openai_params( # noqa: PLR0915 return litellm.AzureOpenAIConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "openrouter": return litellm.OpenrouterConfig().get_supported_openai_params(model=model) + elif custom_llm_provider == "vercel_ai_gateway": + return litellm.VercelAIGatewayConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "mistral" or custom_llm_provider == "codestral": # mistal and codestral api have the exact same params if request_type == "chat_completion": diff --git a/litellm/llms/vercel_ai_gateway/chat/transformation.py b/litellm/llms/vercel_ai_gateway/chat/transformation.py new file mode 100644 index 00000000000..13a88377489 --- /dev/null +++ b/litellm/llms/vercel_ai_gateway/chat/transformation.py @@ -0,0 +1,112 @@ +""" +Support for OpenAI's `/v1/chat/completions` endpoint. + +Calls done in OpenAI/openai.py as Vercel AI Gateway is openai-compatible. + +Docs: https://vercel.com/docs/ai-gateway +""" + +from typing import List, Optional, Tuple, Union + +import httpx + +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.types.llms.openai import AllMessageValues +from litellm.secret_managers.main import get_secret_str +import litellm + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig +from ..common_utils import VercelAIGatewayException + + +class VercelAIGatewayConfig(OpenAIGPTConfig): + @property + def custom_llm_provider(self) -> Optional[str]: + return "vercel_ai_gateway" + + def get_supported_openai_params(self, model: str) -> list: + base_params = super().get_supported_openai_params(model) + if "extra_body" not in base_params: + base_params.append("extra_body") + return base_params + + def _get_openai_compatible_provider_info( + self, api_base: Optional[str], api_key: Optional[str] + ) -> Tuple[Optional[str], Optional[str]]: + + api_base = ( + api_base + or get_secret_str("VERCEL_AI_GATEWAY_API_BASE") + or "https://ai-gateway.vercel.sh/v1" + ) + user_api_key = ( + api_key + or get_secret_str("VERCEL_AI_GATEWAY_API_KEY") + or get_secret_str("VERCEL_OIDC_TOKEN") + ) + return api_base, user_api_key + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + mapped_openai_params = super().map_openai_params( + non_default_params, optional_params, model, drop_params + ) + + # Vercel AI Gateway-only parameters + extra_body = {} + provider_options = non_default_params.pop("providerOptions", None) + + if provider_options is not None: + extra_body["providerOptions"] = provider_options + + mapped_openai_params["extra_body"] = extra_body # openai client supports `extra_body` param + return mapped_openai_params + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + """ + Transform the overall request to be sent to the API. + + Returns: + dict: The transformed request. Sent as the body of the API call. + """ + return super().transform_request( + model, messages, optional_params, litellm_params, headers + ) + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] + ) -> BaseLLMException: + return VercelAIGatewayException( + message=error_message, + status_code=status_code, + headers=headers, + ) + + def get_models( + self, api_key: Optional[str] = None, api_base: Optional[str] = None + ) -> List[str]: + api_base, _ = self._get_openai_compatible_provider_info(api_base, api_key) + + if api_base is None: + api_base = "https://ai-gateway.vercel.sh/v1" + + models_url = f"{api_base}/models" + response = litellm.module_level_client.get(url=models_url) + + if response.status_code != 200: + raise Exception(f"Failed to get models: {response.text}") + + models = response.json()["data"] + return [model["id"] for model in models] diff --git a/litellm/llms/vercel_ai_gateway/common_utils.py b/litellm/llms/vercel_ai_gateway/common_utils.py new file mode 100644 index 00000000000..93e792be05e --- /dev/null +++ b/litellm/llms/vercel_ai_gateway/common_utils.py @@ -0,0 +1,5 @@ +from litellm.llms.base_llm.chat.transformation import BaseLLMException + + +class VercelAIGatewayException(BaseLLMException): + pass diff --git a/litellm/main.py b/litellm/main.py index ffc40b17a77..786a0196e5e 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -2668,6 +2668,70 @@ def completion( # type: ignore # noqa: PLR0915 logging.post_call( input=messages, api_key=openai.api_key, original_response=response ) + elif custom_llm_provider == "vercel_ai_gateway": + api_base = ( + api_base + or litellm.api_base + or get_secret_str("VERCEL_AI_GATEWAY_API_BASE") + or "https://ai-gateway.vercel.sh/v1" + ) + + api_key = ( + api_key + or litellm.api_key + or get_secret("VERCEL_AI_GATEWAY_API_KEY") + ) + + vercel_site_url = get_secret("VERCEL_SITE_URL") or "https://litellm.ai" + vercel_app_name = get_secret("VERCEL_APP_NAME") or "liteLLM" + + vercel_headers = { + "http-referer": vercel_site_url, + "x-title": vercel_app_name, + } + + _headers = headers or litellm.headers + if _headers: + vercel_headers.update(_headers) + + headers = vercel_headers + + ## Load Config + config = litellm.VercelAIGatewayConfig.get_config() + for k, v in config.items(): + if k == "extra_body": + # we use openai 'extra_body' to pass vercel specific params - providerOptions + if "extra_body" in optional_params: + optional_params[k].update(v) + else: + optional_params[k] = v + elif k not in optional_params: + optional_params[k] = v + + data = {"model": model, "messages": messages, **optional_params} + + ## COMPLETION CALL + response = base_llm_http_handler.completion( + model=model, + stream=stream, + messages=messages, + acompletion=acompletion, + api_base=api_base, + model_response=model_response, + optional_params=optional_params, + litellm_params=litellm_params, + custom_llm_provider="vercel_ai_gateway", + timeout=timeout, + headers=headers, + encoding=encoding, + api_key=api_key, + logging_obj=logging, # model call logging done inside the class as we make need to modify I/O to fit aleph alpha's requirements + client=client, + ) + ## LOGGING + logging.post_call( + input=messages, api_key=openai.api_key, original_response=response + ) elif ( custom_llm_provider == "together_ai" or ("togethercomputer" in model) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index bcb6fb9fb69..adac065d2d9 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2351,9 +2351,9 @@ class LlmProviders(str, Enum): COMETAPI = "cometapi" OCI = "oci" AUTO_ROUTER = "auto_router" + VERCEL_AI_GATEWAY = "vercel_ai_gateway" DOTPROMPT = "dotprompt" - # Create a set of all provider values for quick lookup LlmProvidersSet = {provider.value for provider in LlmProviders} diff --git a/litellm/utils.py b/litellm/utils.py index 6375c1f2448..69f4603fea0 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -2352,6 +2352,9 @@ def register_model(model_cost: Union[str, dict]): # noqa: PLR0915 split_string = key.split("/", 1) if key not in litellm.openrouter_models: litellm.openrouter_models.add(split_string[1]) + elif value.get("litellm_provider") == "vercel_ai_gateway": + if key not in litellm.vercel_ai_gateway_models: + litellm.vercel_ai_gateway_models.add(key) elif value.get("litellm_provider") == "vertex_ai-text-models": if key not in litellm.vertex_text_models: litellm.vertex_text_models.add(key) @@ -3226,6 +3229,7 @@ def pre_process_optional_params( and custom_llm_provider != "bedrock" and custom_llm_provider != "ollama_chat" and custom_llm_provider != "openrouter" + and custom_llm_provider != "vercel_ai_gateway" and custom_llm_provider != "nebius" and custom_llm_provider not in litellm.openai_compatible_providers ): @@ -3902,7 +3906,6 @@ def get_optional_params( # noqa: PLR0915 else False ), ) - elif custom_llm_provider == "watsonx": optional_params = litellm.IBMWatsonXChatConfig().map_openai_params( non_default_params=non_default_params, @@ -5297,6 +5300,11 @@ def validate_environment( # noqa: PLR0915 keys_in_environment = True else: missing_keys.append("OPENROUTER_API_KEY") + elif custom_llm_provider == "vercel_ai_gateway": + if "VERCEL_AI_GATEWAY_API_KEY" in os.environ: + keys_in_environment = True + else: + missing_keys.append("VERCEL_AI_GATEWAY_API_KEY") elif custom_llm_provider == "datarobot": if "DATAROBOT_API_TOKEN" in os.environ: keys_in_environment = True @@ -5520,6 +5528,12 @@ def validate_environment( # noqa: PLR0915 keys_in_environment = True else: missing_keys.append("OPENROUTER_API_KEY") + ## vercel_ai_gateway + elif model in litellm.vercel_ai_gateway_models: + if "VERCEL_AI_GATEWAY_API_KEY" in os.environ: + keys_in_environment = True + else: + missing_keys.append("VERCEL_AI_GATEWAY_API_KEY") ## datarobot elif model in litellm.datarobot_models: if "DATAROBOT_API_TOKEN" in os.environ: @@ -6904,6 +6918,8 @@ class ProviderConfigManager: return litellm.TogetherAIConfig() elif litellm.LlmProviders.OPENROUTER == provider: return litellm.OpenrouterConfig() + elif litellm.LlmProviders.VERCEL_AI_GATEWAY == provider: + return litellm.VercelAIGatewayConfig() elif litellm.LlmProviders.COMETAPI == provider: return litellm.CometAPIConfig() elif litellm.LlmProviders.DATAROBOT == provider: diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c9658c6cc47..90a64300219 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -19775,6 +19775,848 @@ "supports_system_messages": true, "supports_tool_choice": false }, + "vercel_ai_gateway/alibaba/qwen3-coder": { + "max_tokens": 262144, + "input_cost_per_token": 4e-07, + "output_cost_per_token": 1.6e-06, + "max_output_tokens": 66536, + "max_input_tokens": 262144, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/codestral-embed": { + "max_tokens": 0, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/google/gemini-2.5-pro": { + "max_tokens": 1048576, + "input_cost_per_token": 2.5e-06, + "output_cost_per_token": 1e-05, + "max_output_tokens": 65536, + "max_input_tokens": 1048576, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/deepseek/deepseek-v3": { + "max_tokens": 128000, + "input_cost_per_token": 9e-07, + "output_cost_per_token": 9e-07, + "max_output_tokens": 8192, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/amazon/nova-lite": { + "max_tokens": 300000, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 2.4e-07, + "max_output_tokens": 8192, + "max_input_tokens": 300000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-4-scout": { + "max_tokens": 131072, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, + "max_output_tokens": 8192, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-3.2-1b": { + "max_tokens": 128000, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 1e-07, + "max_output_tokens": 8192, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/mistral-small": { + "max_tokens": 32000, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, + "max_output_tokens": 4000, + "max_input_tokens": 32000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/google/gemini-2.5-flash": { + "max_tokens": 1000000, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 2.5e-06, + "max_output_tokens": 65536, + "max_input_tokens": 1000000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/inception/mercury-coder-small": { + "max_tokens": 32000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 1e-06, + "max_output_tokens": 16384, + "max_input_tokens": 32000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/text-embedding-3-small": { + "max_tokens": 0, + "input_cost_per_token": 2e-08, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "embedding" + }, + "vercel_ai_gateway/xai/grok-2-vision": { + "max_tokens": 32768, + "input_cost_per_token": 2e-06, + "output_cost_per_token": 1e-05, + "max_output_tokens": 32768, + "max_input_tokens": 32768, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/xai/grok-2": { + "max_tokens": 131072, + "input_cost_per_token": 2e-06, + "output_cost_per_token": 1e-05, + "max_output_tokens": 4000, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/deepseek/deepseek-r1-distill-llama-70b": { + "max_tokens": 131072, + "input_cost_per_token": 7.5e-07, + "output_cost_per_token": 9.9e-07, + "max_output_tokens": 131072, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-3.1-70b": { + "max_tokens": 128000, + "input_cost_per_token": 7.2e-07, + "output_cost_per_token": 7.2e-07, + "max_output_tokens": 8192, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/xai/grok-3": { + "max_tokens": 131072, + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "max_output_tokens": 131072, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/alibaba/qwen-3-235b": { + "max_tokens": 40960, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 6e-07, + "max_output_tokens": 16384, + "max_input_tokens": 40960, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/xai/grok-3-fast": { + "max_tokens": 131072, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2.5e-05, + "max_output_tokens": 131072, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/vercel/v0-1.5-md": { + "max_tokens": 128000, + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "max_output_tokens": 32768, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/o4-mini": { + "max_tokens": 200000, + "input_cost_per_token": 1.1e-06, + "output_cost_per_token": 4.4e-06, + "max_output_tokens": 100000, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 2.75e-07, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/magistral-medium": { + "max_tokens": 128000, + "input_cost_per_token": 2e-06, + "output_cost_per_token": 5e-06, + "max_output_tokens": 64000, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/amazon/titan-embed-text-v2": { + "max_tokens": 0, + "input_cost_per_token": 2e-08, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/alibaba/qwen-3-30b": { + "max_tokens": 40960, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, + "max_output_tokens": 16384, + "max_input_tokens": 40960, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/zai/glm-4.5-air": { + "max_tokens": 128000, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 1.1e-06, + "max_output_tokens": 96000, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/gpt-4-turbo": { + "max_tokens": 128000, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 3e-05, + "max_output_tokens": 4096, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/mistral-large": { + "max_tokens": 32000, + "input_cost_per_token": 2e-06, + "output_cost_per_token": 6e-06, + "max_output_tokens": 4000, + "max_input_tokens": 32000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/perplexity/sonar-pro": { + "max_tokens": 200000, + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "max_output_tokens": 8000, + "max_input_tokens": 200000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-3.2-90b": { + "max_tokens": 128000, + "input_cost_per_token": 7.2e-07, + "output_cost_per_token": 7.2e-07, + "max_output_tokens": 8192, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-3-8b": { + "max_tokens": 8192, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 8e-08, + "max_output_tokens": 8192, + "max_input_tokens": 8192, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/google/text-embedding-005": { + "max_tokens": 0, + "input_cost_per_token": 2.5e-08, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "embedding" + }, + "vercel_ai_gateway/mistral/pixtral-large": { + "max_tokens": 128000, + "input_cost_per_token": 2e-06, + "output_cost_per_token": 6e-06, + "max_output_tokens": 4000, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/anthropic/claude-3.5-sonnet": { + "max_tokens": 200000, + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "max_output_tokens": 8192, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 3e-07, + "cache_creation_input_token_cost": 3.75e-06, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/amazon/nova-micro": { + "max_tokens": 128000, + "input_cost_per_token": 3.5e-08, + "output_cost_per_token": 1.4e-07, + "max_output_tokens": 8192, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/cohere/command-r": { + "max_tokens": 128000, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "max_output_tokens": 4096, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/morph/morph-v3-large": { + "max_tokens": 32768, + "input_cost_per_token": 9e-07, + "output_cost_per_token": 1.9e-06, + "max_output_tokens": 16384, + "max_input_tokens": 32768, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/mixtral-8x22b-instruct": { + "max_tokens": 65536, + "input_cost_per_token": 1.2e-06, + "output_cost_per_token": 1.2e-06, + "max_output_tokens": 2048, + "max_input_tokens": 65536, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/xai/grok-4": { + "max_tokens": 256000, + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "max_output_tokens": 256000, + "max_input_tokens": 256000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-3.1-8b": { + "max_tokens": 131000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 8e-08, + "max_output_tokens": 131072, + "max_input_tokens": 131000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/anthropic/claude-3-opus": { + "max_tokens": 200000, + "input_cost_per_token": 1.5e-05, + "output_cost_per_token": 7.5e-05, + "max_output_tokens": 4096, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 1.5e-06, + "cache_creation_input_token_cost": 1.875e-05, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/zai/glm-4.5": { + "max_tokens": 131072, + "input_cost_per_token": 6e-07, + "output_cost_per_token": 2.2e-06, + "max_output_tokens": 131072, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/gpt-4o": { + "max_tokens": 128000, + "input_cost_per_token": 2.5e-06, + "output_cost_per_token": 1e-05, + "max_output_tokens": 16384, + "max_input_tokens": 128000, + "cache_read_input_token_cost": 1.25e-06, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/o3-mini": { + "max_tokens": 200000, + "input_cost_per_token": 1.1e-06, + "output_cost_per_token": 4.4e-06, + "max_output_tokens": 100000, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 5.5e-07, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/ministral-8b": { + "max_tokens": 128000, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 1e-07, + "max_output_tokens": 4000, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/o3": { + "max_tokens": 200000, + "input_cost_per_token": 2e-06, + "output_cost_per_token": 8e-06, + "max_output_tokens": 100000, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 5e-07, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/vercel/v0-1.0-md": { + "max_tokens": 128000, + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "max_output_tokens": 32000, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/google/text-multilingual-embedding-002": { + "max_tokens": 0, + "input_cost_per_token": 2.5e-08, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "embedding" + }, + "vercel_ai_gateway/amazon/nova-pro": { + "max_tokens": 300000, + "input_cost_per_token": 8e-07, + "output_cost_per_token": 3.2e-06, + "max_output_tokens": 8192, + "max_input_tokens": 300000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/morph/morph-v3-fast": { + "max_tokens": 32768, + "input_cost_per_token": 8e-07, + "output_cost_per_token": 1.2e-06, + "max_output_tokens": 16384, + "max_input_tokens": 32768, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/gpt-3.5-turbo": { + "max_tokens": 16385, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 1.5e-06, + "max_output_tokens": 4096, + "max_input_tokens": 16385, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/codestral": { + "max_tokens": 256000, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 9e-07, + "max_output_tokens": 4000, + "max_input_tokens": 256000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-3.2-11b": { + "max_tokens": 128000, + "input_cost_per_token": 1.6e-07, + "output_cost_per_token": 1.6e-07, + "max_output_tokens": 8192, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-3-70b": { + "max_tokens": 8192, + "input_cost_per_token": 5.9e-07, + "output_cost_per_token": 7.9e-07, + "max_output_tokens": 8192, + "max_input_tokens": 8192, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/xai/grok-3-mini-fast": { + "max_tokens": 131072, + "input_cost_per_token": 6e-07, + "output_cost_per_token": 4e-06, + "max_output_tokens": 131072, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/text-embedding-3-large": { + "max_tokens": 0, + "input_cost_per_token": 1.3e-07, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "embedding" + }, + "vercel_ai_gateway/google/gemini-2.0-flash-lite": { + "max_tokens": 1048576, + "input_cost_per_token": 7.5e-08, + "output_cost_per_token": 3e-07, + "max_output_tokens": 8192, + "max_input_tokens": 1048576, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/ministral-3b": { + "max_tokens": 128000, + "input_cost_per_token": 4e-08, + "output_cost_per_token": 4e-08, + "max_output_tokens": 4000, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/perplexity/sonar-reasoning-pro": { + "max_tokens": 127000, + "input_cost_per_token": 2e-06, + "output_cost_per_token": 8e-06, + "max_output_tokens": 8000, + "max_input_tokens": 127000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/google/gemini-embedding-001": { + "max_tokens": 0, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "embedding" + }, + "vercel_ai_gateway/anthropic/claude-3-haiku": { + "max_tokens": 200000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 1.25e-06, + "max_output_tokens": 4096, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 3e-08, + "cache_creation_input_token_cost": 3e-07, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/o1": { + "max_tokens": 200000, + "input_cost_per_token": 1.5e-05, + "output_cost_per_token": 6e-05, + "max_output_tokens": 100000, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 7.5e-06, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/deepseek/deepseek-r1": { + "max_tokens": 128000, + "input_cost_per_token": 5.5e-07, + "output_cost_per_token": 2.19e-06, + "max_output_tokens": 8192, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/mistral-embed": { + "max_tokens": 0, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/gpt-4.1-mini": { + "max_tokens": 1047576, + "input_cost_per_token": 4e-07, + "output_cost_per_token": 1.6e-06, + "max_output_tokens": 32768, + "max_input_tokens": 1047576, + "cache_read_input_token_cost": 1e-07, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/gpt-4o-mini": { + "max_tokens": 128000, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "max_output_tokens": 16384, + "max_input_tokens": 128000, + "cache_read_input_token_cost": 7.5e-08, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/alibaba/qwen-3-14b": { + "max_tokens": 40960, + "input_cost_per_token": 8e-08, + "output_cost_per_token": 2.4e-07, + "max_output_tokens": 16384, + "max_input_tokens": 40960, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/anthropic/claude-4-opus": { + "max_tokens": 200000, + "input_cost_per_token": 1.5e-05, + "output_cost_per_token": 7.5e-05, + "max_output_tokens": 32000, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 1.5e-06, + "cache_creation_input_token_cost": 1.875e-05, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/mistral-saba-24b": { + "max_tokens": 32768, + "input_cost_per_token": 7.9e-07, + "output_cost_per_token": 7.9e-07, + "max_output_tokens": 32768, + "max_input_tokens": 32768, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/perplexity/sonar-reasoning": { + "max_tokens": 127000, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 5e-06, + "max_output_tokens": 8000, + "max_input_tokens": 127000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/anthropic/claude-3.5-haiku": { + "max_tokens": 200000, + "input_cost_per_token": 8e-07, + "output_cost_per_token": 4e-06, + "max_output_tokens": 8192, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 8e-08, + "cache_creation_input_token_cost": 1e-06, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/cohere/command-a": { + "max_tokens": 256000, + "input_cost_per_token": 2.5e-06, + "output_cost_per_token": 1e-05, + "max_output_tokens": 8000, + "max_input_tokens": 256000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/google/gemma-2-9b": { + "max_tokens": 8192, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 2e-07, + "max_output_tokens": 8192, + "max_input_tokens": 8192, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-3.2-3b": { + "max_tokens": 128000, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 1.5e-07, + "max_output_tokens": 8192, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/gpt-4.1-nano": { + "max_tokens": 1047576, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 4e-07, + "max_output_tokens": 32768, + "max_input_tokens": 1047576, + "cache_read_input_token_cost": 2.5e-08, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/anthropic/claude-4-sonnet": { + "max_tokens": 200000, + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "max_output_tokens": 64000, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 3e-07, + "cache_creation_input_token_cost": 3.75e-06, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/perplexity/sonar": { + "max_tokens": 127000, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 1e-06, + "max_output_tokens": 8000, + "max_input_tokens": 127000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-4-maverick": { + "max_tokens": 131072, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 6e-07, + "max_output_tokens": 8192, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/text-embedding-ada-002": { + "max_tokens": 0, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "embedding" + }, + "vercel_ai_gateway/xai/grok-3-mini": { + "max_tokens": 131072, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 5e-07, + "max_output_tokens": 131072, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/cohere/embed-v4.0": { + "max_tokens": 0, + "input_cost_per_token": 1.2e-07, + "output_cost_per_token": 0.0, + "max_output_tokens": 0, + "max_input_tokens": 0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/meta/llama-3.3-70b": { + "max_tokens": 128000, + "input_cost_per_token": 7.2e-07, + "output_cost_per_token": 7.2e-07, + "max_output_tokens": 8192, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/cohere/command-r-plus": { + "max_tokens": 128000, + "input_cost_per_token": 2.5e-06, + "output_cost_per_token": 1e-05, + "max_output_tokens": 4096, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/gpt-3.5-turbo-instruct": { + "max_tokens": 8192, + "input_cost_per_token": 1.5e-06, + "output_cost_per_token": 2e-06, + "max_output_tokens": 4096, + "max_input_tokens": 8192, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/devstral-small": { + "max_tokens": 128000, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 2.8e-07, + "max_output_tokens": 128000, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/anthropic/claude-3.7-sonnet": { + "max_tokens": 200000, + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "max_output_tokens": 64000, + "max_input_tokens": 200000, + "cache_read_input_token_cost": 3e-07, + "cache_creation_input_token_cost": 3.75e-06, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/google/gemini-2.0-flash": { + "max_tokens": 1048576, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "max_output_tokens": 8192, + "max_input_tokens": 1048576, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/pixtral-12b": { + "max_tokens": 128000, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 1.5e-07, + "max_output_tokens": 4000, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/mistral/magistral-small": { + "max_tokens": 128000, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 1.5e-06, + "max_output_tokens": 64000, + "max_input_tokens": 128000, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/moonshotai/kimi-k2": { + "max_tokens": 131072, + "input_cost_per_token": 5.5e-07, + "output_cost_per_token": 2.2e-06, + "max_output_tokens": 16384, + "max_input_tokens": 131072, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/alibaba/qwen-3-32b": { + "max_tokens": 40960, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, + "max_output_tokens": 16384, + "max_input_tokens": 40960, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, + "vercel_ai_gateway/openai/gpt-4.1": { + "max_tokens": 1047576, + "input_cost_per_token": 2e-06, + "output_cost_per_token": 8e-06, + "max_output_tokens": 32768, + "max_input_tokens": 1047576, + "cache_read_input_token_cost": 5e-07, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "vercel_ai_gateway", + "mode": "chat" + }, "oci/meta.llama-4-maverick-17b-128e-instruct-fp8": { "max_tokens": 512000, "max_input_tokens": 512000, @@ -20009,4 +20851,4 @@ "notes": "DALL-E 2 via AI/ML API - Reliable text-to-image generation" } } -} +} \ No newline at end of file diff --git a/tests/test_litellm/llms/vercel_ai_gateway/chat/test_vercel_ai_gateway_transformation.py b/tests/test_litellm/llms/vercel_ai_gateway/chat/test_vercel_ai_gateway_transformation.py new file mode 100644 index 00000000000..2121473d95c --- /dev/null +++ b/tests/test_litellm/llms/vercel_ai_gateway/chat/test_vercel_ai_gateway_transformation.py @@ -0,0 +1,112 @@ +import os +import sys +from unittest.mock import patch + +import pytest + +sys.path.insert( + 0, os.path.abspath("../../../../..") +) # Adds the parent directory to the system path + +from litellm.llms.vercel_ai_gateway.chat.transformation import ( + VercelAIGatewayConfig, +) +from litellm.llms.vercel_ai_gateway.common_utils import VercelAIGatewayException + + +def test_vercel_ai_gateway_extra_body_transformation(): + """Test that providerOptions is correctly moved to extra_body""" + transformed_request = VercelAIGatewayConfig().transform_request( + model="vercel_ai_gateway/openai/gpt-4o", + messages=[{"role": "user", "content": "Hello, world!"}], + optional_params={ + "extra_body": { + "providerOptions": { + "gateway": {"order": ["azure", "openai"]} + } + } + }, + litellm_params={}, + headers={}, + ) + + assert transformed_request["extra_body"]["providerOptions"]["gateway"]["order"] == ["azure", "openai"] + assert transformed_request["messages"] == [ + {"role": "user", "content": "Hello, world!"} + ] + + +def test_vercel_ai_gateway_provider_options_mapping(): + """Test that providerOptions from non_default_params is moved to extra_body""" + config = VercelAIGatewayConfig() + + non_default_params = { + "providerOptions": { + "gateway": {"order": ["azure", "openai"]} + } + } + optional_params = {} + model = "vercel_ai_gateway/openai/gpt-4o" + + result = config.map_openai_params( + non_default_params, optional_params, model, drop_params=False + ) + + assert result["extra_body"]["providerOptions"]["gateway"]["order"] == ["azure", "openai"] + assert "providerOptions" not in result + + +def test_vercel_ai_gateway_get_supported_openai_params(): + """Test that extra_body is included in supported params""" + config = VercelAIGatewayConfig() + supported_params = config.get_supported_openai_params("vercel_ai_gateway/openai/gpt-4o") + + assert "extra_body" in supported_params + assert "temperature" in supported_params + assert "max_tokens" in supported_params + assert "stream" in supported_params + + +def test_vercel_ai_gateway_get_openai_compatible_provider_info(): + """Test provider info retrieval with environment variables""" + config = VercelAIGatewayConfig() + + with patch.dict( + "os.environ", + { + "VERCEL_AI_GATEWAY_API_BASE": "https://env.vercel.sh/v1", + "VERCEL_AI_GATEWAY_API_KEY": "env_api_key", + }, + ): + api_base, api_key = config._get_openai_compatible_provider_info(None, None) + assert api_base == "https://env.vercel.sh/v1" + assert api_key == "env_api_key" + + +def test_vercel_ai_gateway_error_class(): + """Test error class creation""" + config = VercelAIGatewayConfig() + + error_message = "Test error" + status_code = 400 + headers = {"Content-Type": "application/json"} + + error_class = config.get_error_class(error_message, status_code, headers) + + assert isinstance(error_class, VercelAIGatewayException) + assert error_class.message == error_message + assert error_class.status_code == status_code + assert error_class.headers == headers + + +def test_vercel_ai_gateway_exception_inheritance(): + """Test that VercelAIGatewayException inherits from BaseLLMException""" + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + exception = VercelAIGatewayException( + message="test", + status_code=500, + headers={} + ) + + assert isinstance(exception, BaseLLMException) diff --git a/tests/test_litellm/llms/vercel_ai_gateway/test_vercel_ai_gateway.py b/tests/test_litellm/llms/vercel_ai_gateway/test_vercel_ai_gateway.py new file mode 100755 index 00000000000..7a2d992ec89 --- /dev/null +++ b/tests/test_litellm/llms/vercel_ai_gateway/test_vercel_ai_gateway.py @@ -0,0 +1,228 @@ +""" +Mock tests for vercel_ai_gateway provider +""" +import json +from unittest.mock import MagicMock, patch + +import pytest +import respx + +import litellm +from litellm import completion +from litellm.llms.vercel_ai_gateway.chat.transformation import VercelAIGatewayConfig + + +@pytest.fixture +def vercel_ai_gateway_response(): + """Mock response from Vercel AI Gateway API""" + return { + "id": "chatcmpl-vercel-123", + "object": "chat.completion", + "created": 1677652288, + "model": "openai/gpt-3.5-turbo", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "Hello! This is a test response from Vercel AI Gateway."}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 10, "completion_tokens": 15, "total_tokens": 25}, + } + + +def test_vercel_ai_gateway_config_initialization(): + """Test VercelAIGatewayConfig initializes correctly""" + config = VercelAIGatewayConfig() + assert config.custom_llm_provider == "vercel_ai_gateway" + + +def test_get_llm_provider_vercel_ai_gateway(): + """Test that get_llm_provider correctly identifies vercel_ai_gateway""" + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + # Test with vercel_ai_gateway/provider/model-name format + model, provider, api_key, api_base = get_llm_provider("vercel_ai_gateway/openai/gpt-4o") + assert model == "openai/gpt-4o" + assert provider == "vercel_ai_gateway" + + # Test with api_base containing vercel ai gateway endpoint + model, provider, api_key, api_base = get_llm_provider("gpt-4o", api_base="https://ai-gateway.vercel.sh/v1") + assert model == "gpt-4o" + assert provider == "vercel_ai_gateway" + assert api_base == "https://ai-gateway.vercel.sh/v1" + + +def test_vercel_ai_gateway_in_provider_lists(): + """Test that vercel_ai_gateway is registered in all necessary provider lists""" + assert "vercel_ai_gateway" in litellm.openai_compatible_providers + assert "vercel_ai_gateway" in litellm.provider_list + assert "https://ai-gateway.vercel.sh/v1" in litellm.openai_compatible_endpoints + + +@pytest.mark.asyncio +async def test_vercel_ai_gateway_completion_call(respx_mock, vercel_ai_gateway_response, monkeypatch): + """Test completion call with vercel_ai_gateway provider using mocked response""" + monkeypatch.setenv("VERCEL_AI_GATEWAY_API_KEY", "test-api-key") + litellm.disable_aiohttp_transport = True + + respx_mock.post("https://ai-gateway.vercel.sh/v1/chat/completions").respond(json=vercel_ai_gateway_response) + + response = await litellm.acompletion( + model="vercel_ai_gateway/openai/gpt-3.5-turbo", + messages=[{"role": "user", "content": "Hello, this is a test"}], + max_tokens=20, + ) + + assert response.choices[0].message.content == "Hello! This is a test response from Vercel AI Gateway." + assert response.model == "vercel_ai_gateway/openai/gpt-3.5-turbo" + assert response.usage.total_tokens == 25 + + assert len(respx_mock.calls) == 1 + request = respx_mock.calls[0].request + assert request.method == "POST" + assert "ai-gateway.vercel.sh" in str(request.url) + + assert "Authorization" in request.headers + assert request.headers["Authorization"] == "Bearer test-api-key" + + +@pytest.mark.asyncio +async def test_vercel_ai_gateway_with_oidc_token(respx_mock, vercel_ai_gateway_response, monkeypatch): + """Test completion call with vercel_ai_gateway provider using VERCEL_OIDC_TOKEN""" + monkeypatch.setenv("VERCEL_OIDC_TOKEN", "test-oidc-token") + litellm.disable_aiohttp_transport = True + + respx_mock.post("https://ai-gateway.vercel.sh/v1/chat/completions").respond(json=vercel_ai_gateway_response) + + response = await litellm.acompletion( + model="vercel_ai_gateway/openai/gpt-3.5-turbo", + messages=[{"role": "user", "content": "Hello, this is a test"}], + max_tokens=20, + ) + + assert response.choices[0].message.content == "Hello! This is a test response from Vercel AI Gateway." + assert response.model == "vercel_ai_gateway/openai/gpt-3.5-turbo" + assert response.usage.total_tokens == 25 + + assert len(respx_mock.calls) == 1 + request = respx_mock.calls[0].request + + assert "Authorization" in request.headers + assert request.headers["Authorization"] == "Bearer test-oidc-token" + + +def test_vercel_ai_gateway_supported_params(): + """Test that vercel_ai_gateway returns the supported parameters""" + config = VercelAIGatewayConfig() + supported_params = config.get_supported_openai_params("vercel_ai_gateway/openai/gpt-3.5-turbo") + + # vercel_ai_gateway should include all base OpenAI params plus extra_body + expected_base_params = [ + "frequency_penalty", + "logit_bias", + "logprobs", + "top_logprobs", + "max_tokens", + "max_completion_tokens", + "modalities", + "prediction", + "n", + "presence_penalty", + "seed", + "stop", + "stream", + "stream_options", + "temperature", + "top_p", + "tools", + "tool_choice", + "function_call", + "functions", + "max_retries", + "extra_headers", + "parallel_tool_calls", + "audio", + "web_search_options", + "extra_body", + ] + + for param in expected_base_params: + assert param in supported_params, f"Expected parameter '{param}' not found in supported params" + + assert "extra_body" in supported_params + + +def test_vercel_ai_gateway_sync_completion(respx_mock, vercel_ai_gateway_response, monkeypatch): + """Test synchronous completion call""" + monkeypatch.setenv("VERCEL_AI_GATEWAY_API_KEY", "test-api-key") + litellm.disable_aiohttp_transport = True + + respx_mock.post("https://ai-gateway.vercel.sh/v1/chat/completions").respond(json=vercel_ai_gateway_response) + + response = completion( + model="vercel_ai_gateway/openai/gpt-3.5-turbo", + messages=[{"role": "user", "content": "Hello"}], + max_tokens=20, + ) + + assert response.choices[0].message.content == "Hello! This is a test response from Vercel AI Gateway." + assert response.model == "vercel_ai_gateway/openai/gpt-3.5-turbo" + assert response.usage.total_tokens == 25 + + +def test_vercel_ai_gateway_with_provider_options(respx_mock, vercel_ai_gateway_response, monkeypatch): + """Test vercel_ai_gateway with providerOptions parameter""" + monkeypatch.setenv("VERCEL_AI_GATEWAY_API_KEY", "test-api-key") + litellm.disable_aiohttp_transport = True + + respx_mock.post("https://ai-gateway.vercel.sh/v1/chat/completions").respond(json=vercel_ai_gateway_response) + + response = completion( + model="vercel_ai_gateway/openai/gpt-3.5-turbo", + messages=[{"role": "user", "content": "Hello"}], + providerOptions={"gateway": {"order": ["azure", "openai"]}}, + max_tokens=20, + ) + + assert response.choices[0].message.content == "Hello! This is a test response from Vercel AI Gateway." + assert response.model == "vercel_ai_gateway/openai/gpt-3.5-turbo" + assert response.usage.total_tokens == 25 + + assert len(respx_mock.calls) == 1 + request = respx_mock.calls[0].request + request_data = json.loads(request.content.decode("utf-8")) + assert "providerOptions" in request_data + assert request_data["providerOptions"]["gateway"]["order"] == ["azure", "openai"] + + +def test_vercel_ai_gateway_models_endpoint(): + """Test the get_models functionality""" + config = VercelAIGatewayConfig() + + with patch("litellm.module_level_client.get") as mock_get: + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "data": [{"id": "openai/gpt-4o"}, {"id": "openai/gpt-3.5-turbo"}, {"id": "anthropic/claude-3-sonnet"}] + } + mock_get.return_value = mock_response + + models = config.get_models() + + assert models == ["openai/gpt-4o", "openai/gpt-3.5-turbo", "anthropic/claude-3-sonnet"] + mock_get.assert_called_once_with(url="https://ai-gateway.vercel.sh/v1/models") + + +def test_vercel_ai_gateway_models_endpoint_failure(): + """Test the get_models functionality with failure""" + config = VercelAIGatewayConfig() + + with patch("litellm.module_level_client.get") as mock_get: + mock_response = MagicMock() + mock_response.status_code = 404 + mock_response.text = "Not found" + mock_get.return_value = mock_response + + with pytest.raises(Exception, match="Failed to get models: Not found"): + config.get_models()