diff --git a/.github/workflows/interpret_load_test.py b/.github/workflows/interpret_load_test.py
index b52d4d2b395..2eea46778aa 100644
--- a/.github/workflows/interpret_load_test.py
+++ b/.github/workflows/interpret_load_test.py
@@ -77,6 +77,9 @@ if __name__ == "__main__":
new_release_body = (
existing_release_body
+ "\n\n"
+ + "### Don't want to maintain your internal proxy? get in touch 🎉"
+ + "Hosted Proxy Alpha: https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat"
+ + "\n\n"
+ "## Load Test LiteLLM Proxy Results"
+ "\n\n"
+ markdown_table
diff --git a/docs/my-website/docs/completion/vision.md b/docs/my-website/docs/completion/vision.md
new file mode 100644
index 00000000000..ea04b1e1e11
--- /dev/null
+++ b/docs/my-website/docs/completion/vision.md
@@ -0,0 +1,45 @@
+# Using Vision Models
+
+## Quick Start
+Example passing images to a model
+
+```python
+import os
+from litellm import completion
+
+os.environ["OPENAI_API_KEY"] = "your-api-key"
+
+# openai call
+response = completion(
+ model = "gpt-4-vision-preview",
+ messages=[
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What’s in this image?"
+ },
+ {
+ "type": "image_url",
+ "image_url": {
+ "url": "https://upload.wikimedia.org/wikipedia/commons/thumb/d/dd/Gfp-wisconsin-madison-the-nature-boardwalk.jpg/2560px-Gfp-wisconsin-madison-the-nature-boardwalk.jpg"
+ }
+ }
+ ]
+ }
+ ],
+)
+
+```
+
+## Checking if a model supports `vision`
+
+Use `litellm.supports_vision(model="")` -> returns `True` if model supports `vision` and `False` if not
+
+```python
+assert litellm.supports_vision(model="gpt-4-vision-preview") == True
+assert litellm.supports_vision(model="gemini-1.0-pro-visionn") == True
+assert litellm.supports_vision(model="gpt-3.5-turbo") == False
+```
+
diff --git a/docs/my-website/docs/embedding/supported_embedding.md b/docs/my-website/docs/embedding/supported_embedding.md
index 7e2374d16df..52cfee7ac32 100644
--- a/docs/my-website/docs/embedding/supported_embedding.md
+++ b/docs/my-website/docs/embedding/supported_embedding.md
@@ -339,6 +339,8 @@ All models listed [here](https://github.com/BerriAI/litellm/blob/57f37f743886a02
| textembedding-gecko-multilingual@001 | `embedding(model="vertex_ai/textembedding-gecko-multilingual@001", input)` |
| textembedding-gecko@001 | `embedding(model="vertex_ai/textembedding-gecko@001", input)` |
| textembedding-gecko@003 | `embedding(model="vertex_ai/textembedding-gecko@003", input)` |
+| text-embedding-preview-0409 | `embedding(model="vertex_ai/text-embedding-preview-0409", input)` |
+| text-multilingual-embedding-preview-0409 | `embedding(model="vertex_ai/text-multilingual-embedding-preview-0409", input)` |
## Voyage AI Embedding Models
diff --git a/docs/my-website/docs/enterprise.md b/docs/my-website/docs/enterprise.md
index 7b623e40756..68091fe2ed3 100644
--- a/docs/my-website/docs/enterprise.md
+++ b/docs/my-website/docs/enterprise.md
@@ -8,12 +8,13 @@ For companies that need SSO, user management and professional support for LiteLL
:::
This covers:
-- ✅ **Features under the [LiteLLM Commercial License](https://docs.litellm.ai/docs/proxy/enterprise):**
+- ✅ **Features under the [LiteLLM Commercial License (Content Mod, Custom Tags, etc.)](https://docs.litellm.ai/docs/proxy/enterprise)**
- ✅ **Feature Prioritization**
- ✅ **Custom Integrations**
- ✅ **Professional Support - Dedicated discord + slack**
- ✅ **Custom SLAs**
-- ✅ **Secure access with Single Sign-On**
+- ✅ [**Secure UI access with Single Sign-On**](../docs/proxy/ui.md#setup-ssoauth-for-ui)
+- ✅ [**JWT-Auth**](../docs/proxy/token_auth.md)
## Frequently Asked Questions
diff --git a/docs/my-website/docs/observability/callbacks.md b/docs/my-website/docs/observability/callbacks.md
index fbc0733e53e..af745e84552 100644
--- a/docs/my-website/docs/observability/callbacks.md
+++ b/docs/my-website/docs/observability/callbacks.md
@@ -8,6 +8,7 @@ liteLLM supports:
- [Custom Callback Functions](https://docs.litellm.ai/docs/observability/custom_callback)
- [Lunary](https://lunary.ai/docs)
+- [Langfuse](https://langfuse.com/docs)
- [Helicone](https://docs.helicone.ai/introduction)
- [Traceloop](https://traceloop.com/docs)
- [Athina](https://docs.athina.ai/)
@@ -22,8 +23,8 @@ from litellm import completion
# set callbacks
litellm.input_callback=["sentry"] # for sentry breadcrumbing - logs the input being sent to the api
-litellm.success_callback=["posthog", "helicone", "lunary", "athina"]
-litellm.failure_callback=["sentry", "lunary"]
+litellm.success_callback=["posthog", "helicone", "langfuse", "lunary", "athina"]
+litellm.failure_callback=["sentry", "lunary", "langfuse"]
## set env variables
os.environ['SENTRY_DSN'], os.environ['SENTRY_API_TRACE_RATE']= ""
@@ -32,6 +33,9 @@ os.environ["HELICONE_API_KEY"] = ""
os.environ["TRACELOOP_API_KEY"] = ""
os.environ["LUNARY_PUBLIC_KEY"] = ""
os.environ["ATHINA_API_KEY"] = ""
+os.environ["LANGFUSE_PUBLIC_KEY"] = ""
+os.environ["LANGFUSE_SECRET_KEY"] = ""
+os.environ["LANGFUSE_HOST"] = ""
response = completion(model="gpt-3.5-turbo", messages=messages)
```
diff --git a/docs/my-website/docs/providers/azure_ai.md b/docs/my-website/docs/providers/azure_ai.md
index b8dbe16ba9e..ed13c56641b 100644
--- a/docs/my-website/docs/providers/azure_ai.md
+++ b/docs/my-website/docs/providers/azure_ai.md
@@ -3,8 +3,6 @@ import TabItem from '@theme/TabItem';
# Azure AI Studio
-## Sample Usage
-
**Ensure the following:**
1. The API Base passed ends in the `/v1/` prefix
example:
@@ -14,8 +12,11 @@ import TabItem from '@theme/TabItem';
2. The `model` passed is listed in [supported models](#supported-models). You **DO NOT** Need to pass your deployment name to litellm. Example `model=azure/Mistral-large-nmefg`
+## Usage
+
+",
+ "eos_token": "",
+ },
+ "status": "success",
+ }
+}
+
+
def hf_chat_template(model: str, messages: list, chat_template: Optional[Any] = None):
# Define Jinja2 environment
env = ImmutableSandboxedEnvironment()
@@ -246,20 +258,23 @@ def hf_chat_template(model: str, messages: list, chat_template: Optional[Any] =
else:
return {"status": "failure"}
- tokenizer_config = _get_tokenizer_config(model)
+ if model in known_tokenizer_config:
+ tokenizer_config = known_tokenizer_config[model]
+ else:
+ tokenizer_config = _get_tokenizer_config(model)
if (
tokenizer_config["status"] == "failure"
or "chat_template" not in tokenizer_config["tokenizer"]
):
raise Exception("No chat template found")
## read the bos token, eos token and chat template from the json
- tokenizer_config = tokenizer_config["tokenizer"]
- bos_token = tokenizer_config["bos_token"]
- eos_token = tokenizer_config["eos_token"]
- chat_template = tokenizer_config["chat_template"]
+ tokenizer_config = tokenizer_config["tokenizer"] # type: ignore
+ bos_token = tokenizer_config["bos_token"] # type: ignore
+ eos_token = tokenizer_config["eos_token"] # type: ignore
+ chat_template = tokenizer_config["chat_template"] # type: ignore
try:
- template = env.from_string(chat_template)
+ template = env.from_string(chat_template) # type: ignore
except Exception as e:
raise e
@@ -705,7 +720,7 @@ def anthropic_messages_pt_xml(messages: list):
if assistant_content:
new_messages.append({"role": "assistant", "content": assistant_content})
- if new_messages[0]["role"] != "user":
+ if not new_messages or new_messages[0]["role"] != "user":
if litellm.modify_params:
new_messages.insert(
0, {"role": "user", "content": [{"type": "text", "text": "."}]}
@@ -884,7 +899,7 @@ def anthropic_messages_pt(messages: list):
if assistant_content:
new_messages.append({"role": "assistant", "content": assistant_content})
- if new_messages[0]["role"] != "user":
+ if not new_messages or new_messages[0]["role"] != "user":
if litellm.modify_params:
new_messages.insert(
0, {"role": "user", "content": [{"type": "text", "text": "."}]}
@@ -1286,7 +1301,11 @@ def prompt_factory(
messages=messages, prompt_format=prompt_format, chat_template=chat_template
)
elif custom_llm_provider == "gemini":
- if model == "gemini-pro-vision":
+ if (
+ model == "gemini-pro-vision"
+ or litellm.supports_vision(model=model)
+ or litellm.supports_vision(model=custom_llm_provider + "/" + model)
+ ):
return _gemini_vision_convert_messages(messages=messages)
else:
return gemini_text_image_pt(messages=messages)
diff --git a/litellm/llms/vertex_ai.py b/litellm/llms/vertex_ai.py
index 176902e1a93..fc2d882afd8 100644
--- a/litellm/llms/vertex_ai.py
+++ b/litellm/llms/vertex_ai.py
@@ -87,6 +87,60 @@ class VertexAIConfig:
and v is not None
}
+ def get_supported_openai_params(self):
+ return [
+ "temperature",
+ "top_p",
+ "max_tokens",
+ "stream",
+ "tools",
+ "tool_choice",
+ "response_format",
+ "n",
+ "stop",
+ ]
+
+ def map_openai_params(self, non_default_params: dict, optional_params: dict):
+ for param, value in non_default_params.items():
+ if param == "temperature":
+ optional_params["temperature"] = value
+ if param == "top_p":
+ optional_params["top_p"] = value
+ if param == "stream":
+ optional_params["stream"] = value
+ if param == "n":
+ optional_params["candidate_count"] = value
+ if param == "stop":
+ if isinstance(value, str):
+ optional_params["stop_sequences"] = [value]
+ elif isinstance(value, list):
+ optional_params["stop_sequences"] = value
+ if param == "max_tokens":
+ optional_params["max_output_tokens"] = value
+ if param == "response_format" and value["type"] == "json_object":
+ optional_params["response_mime_type"] = "application/json"
+ if param == "tools" and isinstance(value, list):
+ from vertexai.preview import generative_models
+
+ gtool_func_declarations = []
+ for tool in value:
+ gtool_func_declaration = generative_models.FunctionDeclaration(
+ name=tool["function"]["name"],
+ description=tool["function"].get("description", ""),
+ parameters=tool["function"].get("parameters", {}),
+ )
+ gtool_func_declarations.append(gtool_func_declaration)
+ optional_params["tools"] = [
+ generative_models.Tool(
+ function_declarations=gtool_func_declarations
+ )
+ ]
+ if param == "tool_choice" and (
+ isinstance(value, str) or isinstance(value, dict)
+ ):
+ pass
+ return optional_params
+
import asyncio
@@ -349,8 +403,17 @@ def completion(
print_verbose(
f"VERTEX AI: vertex_project={vertex_project}; vertex_location={vertex_location}"
)
+ if vertex_credentials is not None and isinstance(vertex_credentials, str):
+ import google.oauth2.service_account
- creds, _ = google.auth.default(quota_project_id=vertex_project)
+ json_obj = json.loads(vertex_credentials)
+
+ creds = google.oauth2.service_account.Credentials.from_service_account_info(
+ json_obj,
+ scopes=["https://www.googleapis.com/auth/cloud-platform"],
+ )
+ else:
+ creds, _ = google.auth.default(quota_project_id=vertex_project)
print_verbose(
f"VERTEX AI: creds={creds}; google application credentials: {os.getenv('GOOGLE_APPLICATION_CREDENTIALS')}"
)
@@ -813,8 +876,8 @@ async def async_completion(
tools=tools,
)
- if tools is not None and hasattr(
- response.candidates[0].content.parts[0], "function_call"
+ if tools is not None and bool(
+ getattr(response.candidates[0].content.parts[0], "function_call", None)
):
function_call = response.candidates[0].content.parts[0].function_call
args_dict = {}
@@ -1171,6 +1234,7 @@ def embedding(
encoding=None,
vertex_project=None,
vertex_location=None,
+ vertex_credentials=None,
aembedding=False,
print_verbose=None,
):
@@ -1191,7 +1255,17 @@ def embedding(
print_verbose(
f"VERTEX AI: vertex_project={vertex_project}; vertex_location={vertex_location}"
)
- creds, _ = google.auth.default(quota_project_id=vertex_project)
+ if vertex_credentials is not None and isinstance(vertex_credentials, str):
+ import google.oauth2.service_account
+
+ json_obj = json.loads(vertex_credentials)
+
+ creds = google.oauth2.service_account.Credentials.from_service_account_info(
+ json_obj,
+ scopes=["https://www.googleapis.com/auth/cloud-platform"],
+ )
+ else:
+ creds, _ = google.auth.default(quota_project_id=vertex_project)
print_verbose(
f"VERTEX AI: creds={creds}; google application credentials: {os.getenv('GOOGLE_APPLICATION_CREDENTIALS')}"
)
diff --git a/litellm/main.py b/litellm/main.py
index df9eb3e55a0..b1e75f744c9 100644
--- a/litellm/main.py
+++ b/litellm/main.py
@@ -12,6 +12,7 @@ from typing import Any, Literal, Union, BinaryIO
from functools import partial
import dotenv, traceback, random, asyncio, time, contextvars
from copy import deepcopy
+
import httpx
import litellm
from ._logging import verbose_logger
@@ -341,6 +342,7 @@ async def acompletion(
custom_llm_provider=custom_llm_provider,
original_exception=e,
completion_kwargs=completion_kwargs,
+ extra_kwargs=kwargs,
)
@@ -1709,6 +1711,7 @@ def completion(
encoding=encoding,
vertex_location=vertex_ai_location,
vertex_project=vertex_ai_project,
+ vertex_credentials=vertex_credentials,
logging_obj=logging,
acompletion=acompletion,
)
@@ -2136,6 +2139,7 @@ def completion(
custom_llm_provider=custom_llm_provider,
original_exception=e,
completion_kwargs=args,
+ extra_kwargs=kwargs,
)
@@ -2497,6 +2501,7 @@ async def aembedding(*args, **kwargs):
custom_llm_provider=custom_llm_provider,
original_exception=e,
completion_kwargs=args,
+ extra_kwargs=kwargs,
)
@@ -2806,6 +2811,11 @@ def embedding(
or litellm.vertex_location
or get_secret("VERTEXAI_LOCATION")
)
+ vertex_credentials = (
+ optional_params.pop("vertex_credentials", None)
+ or optional_params.pop("vertex_ai_credentials", None)
+ or get_secret("VERTEXAI_CREDENTIALS")
+ )
response = vertex_ai.embedding(
model=model,
@@ -2816,6 +2826,7 @@ def embedding(
model_response=EmbeddingResponse(),
vertex_project=vertex_ai_project,
vertex_location=vertex_ai_location,
+ vertex_credentials=vertex_credentials,
aembedding=aembedding,
print_verbose=print_verbose,
)
@@ -2932,7 +2943,10 @@ def embedding(
)
## Map to OpenAI Exception
raise exception_type(
- model=model, original_exception=e, custom_llm_provider=custom_llm_provider
+ model=model,
+ original_exception=e,
+ custom_llm_provider=custom_llm_provider,
+ extra_kwargs=kwargs,
)
@@ -3026,6 +3040,7 @@ async def atext_completion(*args, **kwargs):
custom_llm_provider=custom_llm_provider,
original_exception=e,
completion_kwargs=args,
+ extra_kwargs=kwargs,
)
@@ -3363,6 +3378,7 @@ async def aimage_generation(*args, **kwargs):
custom_llm_provider=custom_llm_provider,
original_exception=e,
completion_kwargs=args,
+ extra_kwargs=kwargs,
)
@@ -3562,6 +3578,7 @@ def image_generation(
custom_llm_provider=custom_llm_provider,
original_exception=e,
completion_kwargs=locals(),
+ extra_kwargs=kwargs,
)
@@ -3611,6 +3628,7 @@ async def atranscription(*args, **kwargs):
custom_llm_provider=custom_llm_provider,
original_exception=e,
completion_kwargs=args,
+ extra_kwargs=kwargs,
)
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index 4298ed28b3e..eedcaa49320 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -10,9 +10,9 @@
"supports_function_calling": true
},
"gpt-4-turbo-preview": {
- "max_tokens": 4096,
- "max_input_tokens": 8192,
- "max_output_tokens": 4096,
+ "max_tokens": 4096,
+ "max_input_tokens": 128000,
+ "max_output_tokens": 4096,
"input_cost_per_token": 0.00001,
"output_cost_per_token": 0.00003,
"litellm_provider": "openai",
@@ -75,7 +75,8 @@
"litellm_provider": "openai",
"mode": "chat",
"supports_function_calling": true,
- "supports_parallel_function_calling": true
+ "supports_parallel_function_calling": true,
+ "supports_vision": true
},
"gpt-4-turbo-2024-04-09": {
"max_tokens": 4096,
@@ -86,7 +87,8 @@
"litellm_provider": "openai",
"mode": "chat",
"supports_function_calling": true,
- "supports_parallel_function_calling": true
+ "supports_parallel_function_calling": true,
+ "supports_vision": true
},
"gpt-4-1106-preview": {
"max_tokens": 4096,
@@ -117,7 +119,8 @@
"input_cost_per_token": 0.00001,
"output_cost_per_token": 0.00003,
"litellm_provider": "openai",
- "mode": "chat"
+ "mode": "chat",
+ "supports_vision": true
},
"gpt-4-1106-vision-preview": {
"max_tokens": 4096,
@@ -126,7 +129,8 @@
"input_cost_per_token": 0.00001,
"output_cost_per_token": 0.00003,
"litellm_provider": "openai",
- "mode": "chat"
+ "mode": "chat",
+ "supports_vision": true
},
"gpt-3.5-turbo": {
"max_tokens": 4097,
@@ -209,6 +213,7 @@
"text-embedding-3-large": {
"max_tokens": 8191,
"max_input_tokens": 8191,
+ "output_vector_size": 3072,
"input_cost_per_token": 0.00000013,
"output_cost_per_token": 0.000000,
"litellm_provider": "openai",
@@ -217,6 +222,7 @@
"text-embedding-3-small": {
"max_tokens": 8191,
"max_input_tokens": 8191,
+ "output_vector_size": 1536,
"input_cost_per_token": 0.00000002,
"output_cost_per_token": 0.000000,
"litellm_provider": "openai",
@@ -225,6 +231,7 @@
"text-embedding-ada-002": {
"max_tokens": 8191,
"max_input_tokens": 8191,
+ "output_vector_size": 1536,
"input_cost_per_token": 0.0000001,
"output_cost_per_token": 0.000000,
"litellm_provider": "openai",
@@ -409,7 +416,8 @@
"input_cost_per_token": 0.00001,
"output_cost_per_token": 0.00003,
"litellm_provider": "azure",
- "mode": "chat"
+ "mode": "chat",
+ "supports_vision": true
},
"azure/gpt-35-turbo-16k-0613": {
"max_tokens": 4096,
@@ -700,6 +708,16 @@
"mode": "chat",
"supports_function_calling": true
},
+ "mistral/open-mixtral-8x7b": {
+ "max_tokens": 8191,
+ "max_input_tokens": 32000,
+ "max_output_tokens": 8191,
+ "input_cost_per_token": 0.000002,
+ "output_cost_per_token": 0.000006,
+ "litellm_provider": "mistral",
+ "mode": "chat",
+ "supports_function_calling": true
+ },
"mistral/mistral-embed": {
"max_tokens": 8192,
"max_input_tokens": 8192,
@@ -1004,6 +1022,7 @@
"litellm_provider": "vertex_ai-language-models",
"mode": "chat",
"supports_function_calling": true,
+ "supports_tool_choice": true,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
"gemini-1.5-pro-preview-0215": {
@@ -1015,6 +1034,7 @@
"litellm_provider": "vertex_ai-language-models",
"mode": "chat",
"supports_function_calling": true,
+ "supports_tool_choice": true,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
"gemini-1.5-pro-preview-0409": {
@@ -1026,6 +1046,7 @@
"litellm_provider": "vertex_ai-language-models",
"mode": "chat",
"supports_function_calling": true,
+ "supports_tool_choice": true,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
"gemini-experimental": {
@@ -1037,6 +1058,7 @@
"litellm_provider": "vertex_ai-language-models",
"mode": "chat",
"supports_function_calling": false,
+ "supports_tool_choice": true,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
"gemini-pro-vision": {
@@ -1051,6 +1073,7 @@
"litellm_provider": "vertex_ai-vision-models",
"mode": "chat",
"supports_function_calling": true,
+ "supports_vision": true,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
"gemini-1.0-pro-vision": {
@@ -1065,6 +1088,7 @@
"litellm_provider": "vertex_ai-vision-models",
"mode": "chat",
"supports_function_calling": true,
+ "supports_vision": true,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
"gemini-1.0-pro-vision-001": {
@@ -1079,6 +1103,7 @@
"litellm_provider": "vertex_ai-vision-models",
"mode": "chat",
"supports_function_calling": true,
+ "supports_vision": true,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
"vertex_ai/claude-3-sonnet@20240229": {
@@ -1158,6 +1183,27 @@
"mode": "embedding",
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
+ "text-embedding-preview-0409": {
+ "max_tokens": 3072,
+ "max_input_tokens": 3072,
+ "output_vector_size": 768,
+ "input_cost_per_token": 0.00000000625,
+ "input_cost_per_token_batch_requests": 0.000000005,
+ "output_cost_per_token": 0,
+ "litellm_provider": "vertex_ai-embedding-models",
+ "mode": "embedding",
+ "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing"
+ },
+ "text-multilingual-embedding-preview-0409":{
+ "max_tokens": 3072,
+ "max_input_tokens": 3072,
+ "output_vector_size": 768,
+ "input_cost_per_token": 0.00000000625,
+ "output_cost_per_token": 0,
+ "litellm_provider": "vertex_ai-embedding-models",
+ "mode": "embedding",
+ "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
+ },
"palm/chat-bison": {
"max_tokens": 4096,
"max_input_tokens": 8192,
@@ -1238,8 +1284,23 @@
"litellm_provider": "gemini",
"mode": "chat",
"supports_function_calling": true,
+ "supports_vision": true,
+ "supports_tool_choice": true,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
+ "gemini/gemini-1.5-pro-latest": {
+ "max_tokens": 8192,
+ "max_input_tokens": 1048576,
+ "max_output_tokens": 8192,
+ "input_cost_per_token": 0,
+ "output_cost_per_token": 0,
+ "litellm_provider": "gemini",
+ "mode": "chat",
+ "supports_function_calling": true,
+ "supports_vision": true,
+ "supports_tool_choice": true,
+ "source": "https://ai.google.dev/models/gemini"
+ },
"gemini/gemini-pro-vision": {
"max_tokens": 2048,
"max_input_tokens": 30720,
@@ -1249,6 +1310,7 @@
"litellm_provider": "gemini",
"mode": "chat",
"supports_function_calling": true,
+ "supports_vision": true,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models"
},
"command-r": {
@@ -1711,6 +1773,15 @@
"litellm_provider": "bedrock",
"mode": "chat"
},
+ "anthropic.claude-3-opus-20240229-v1:0": {
+ "max_tokens": 4096,
+ "max_input_tokens": 200000,
+ "max_output_tokens": 4096,
+ "input_cost_per_token": 0.000015,
+ "output_cost_per_token": 0.000075,
+ "litellm_provider": "bedrock",
+ "mode": "chat"
+ },
"anthropic.claude-v1": {
"max_tokens": 8191,
"max_input_tokens": 100000,
diff --git a/litellm/proxy/_experimental/out/404.html b/litellm/proxy/_experimental/out/404.html
index 98382f706ec..3ace44f6825 100644
--- a/litellm/proxy/_experimental/out/404.html
+++ b/litellm/proxy/_experimental/out/404.html
@@ -1 +1 @@
-