Added vertex_ai/qwen models and azure/gpt-5-codex (#14844)

* added qwen models and gpt-5-codex

* fix flaky test

* fix failing test
This commit is contained in:
Mubashir Osmani 2025-09-24 13:40:00 -04:00 • committed by GitHub
parent 22eef373eb
commit 0cd91a82d2
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 159 additions and 6 deletions

View file

@ -2032,6 +2032,36 @@
"supports_tool_choice": false,
"supports_vision": true
},
"azure/gpt-5-codex": {
"cache_read_input_token_cost": 1.25e-07,
"input_cost_per_token": 1.25e-06,
"litellm_provider": "azure",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "responses",
"output_cost_per_token": 1e-05,
"supported_endpoints": [
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true
},
"azure/gpt-5-mini": {
"cache_read_input_token_cost": 2.5e-08,
"input_cost_per_token": 2.5e-07,
@ -5282,6 +5312,49 @@
"supports_tool_choice": true,
"supports_vision": true
},
"deepseek-chat": {
"cache_read_input_token_cost": 6e-08,
"input_cost_per_token": 6e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 131072,
"max_output_tokens": 8192,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 1.7e-06,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"deepseek-reasoner": {
"cache_read_input_token_cost": 6e-08,
"input_cost_per_token": 6e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 1.7e-06,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_function_calling": false,
"supports_native_streaming": true,
"supports_parallel_function_calling": false,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": false
},
"dashscope/qwen-coder": {
"input_cost_per_token": 3e-07,
"litellm_provider": "dashscope",
@ -12427,6 +12500,36 @@
"supports_tool_choice": false,
"supports_vision": true
},
"gpt-5-codex": {
"cache_read_input_token_cost": 1.25e-07,
"input_cost_per_token": 1.25e-06,
"litellm_provider": "openai",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "responses",
"output_cost_per_token": 1e-05,
"supported_endpoints": [
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_native_streaming": false,
"supports_parallel_function_calling": true,
"supports_pdf_input": false,
"supports_prompt_caching": true,
"supports_reasoning": false,
"supports_response_schema": true,
"supports_system_messages": false,
"supports_tool_choice": true,
"supports_vision": true
},
"gpt-5-mini": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
@ -20527,6 +20630,24 @@
"supports_function_calling": true,
"supports_tool_choice": true
},
"vertex_ai/deepseek-ai/deepseek-v3.1-maas": {
"input_cost_per_token": 1.35e-06,
"litellm_provider": "vertex_ai-deepseek_models",
"max_input_tokens": 163840,
"max_output_tokens": 32768,
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 5.4e-06,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
"supported_regions": [
"us-west2"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_tool_choice": true
},
"vertex_ai/deepseek-ai/deepseek-r1-0528-maas": {
"input_cost_per_token": 1.35e-06,
"litellm_provider": "vertex_ai-deepseek_models",
@ -20940,6 +21061,30 @@
"supports_function_calling": true,
"supports_tool_choice": true
},
"vertex_ai/qwen/qwen3-next-80b-a3b-instruct-maas": {
"input_cost_per_token": 1.5e-07,
"litellm_provider": "vertex_ai-qwen_models",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"supports_function_calling": true,
"supports_tool_choice": true
},
"vertex_ai/qwen/qwen3-next-80b-a3b-thinking-maas": {
"input_cost_per_token": 1.5e-07,
"litellm_provider": "vertex_ai-qwen_models",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"supports_function_calling": true,
"supports_tool_choice": true
},
"vertex_ai/veo-2.0-generate-001": {
"litellm_provider": "vertex_ai-video-models",
"max_input_tokens": 1024,

View file

@ -27,6 +27,7 @@ import litellm
from litellm import (
ModelResponse,
RateLimitError,
ServiceUnavailableError,
Timeout,
completion,
completion_cost,
@ -2020,10 +2021,16 @@ def test_bedrock_context_window_error():
def test_bedrock_converse_route():
litellm.set_verbose = True
litellm.completion(
model="bedrock/converse/us.amazon.nova-pro-v1:0",
messages=[{"role": "user", "content": "Hello, world!"}],
)
try:
litellm.completion(
model="bedrock/converse/us.amazon.nova-pro-v1:0",
messages=[{"role": "user", "content": "Hello, world!"}],
)
except ServiceUnavailableError as e:
if "Too many requests" in str(e):
pytest.skip("Skipping test due to AWS Bedrock rate limiting")
else:
raise
def test_bedrock_mapped_converse_models():

View file

@ -43,8 +43,9 @@ async def test_anthropic_basic_completion_with_headers():
json.dumps(response_json, indent=4, default=str),
)
reported_usage = response_json.get("usage", None)
anthropic_api_input_tokens = reported_usage.get("input_tokens", None)
anthropic_api_output_tokens = reported_usage.get("output_tokens", None)
# fix null checks for reported_usage
anthropic_api_input_tokens = reported_usage.get("input_tokens", None) if reported_usage else None
anthropic_api_output_tokens = reported_usage.get("output_tokens", None) if reported_usage else None
litellm_call_id = response_headers.get("x-litellm-call-id")
print(f"LiteLLM Call ID: {litellm_call_id}")