diff --git a/litellm/constants.py b/litellm/constants.py index 8713bd49f57..a6b6310d6bc 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -792,6 +792,7 @@ openai_compatible_endpoints: Final[list] = [ "https://api.meta.ai/v1", "https://api.cognition.ai/v1", "https://api.scx.ai/v1", + "https://api.flex.ai/v1", ] @@ -861,6 +862,7 @@ openai_compatible_providers: Final[list] = [ "meta", # Meta Model API (Muse Spark) - JSON-configured provider "cognition", "scx-ai", + "flexai", # FlexAI Token Service - JSON-configured provider ] openai_text_completion_compatible_providers: Final[list] = [ # providers that support `/v1/completions` "together_ai", diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index a458a209ea9..f9f39661ba0 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -200,5 +200,12 @@ "temperature_max": 1.99 }, "supported_endpoints": ["/v1/chat/completions"] + }, + "flexai": { + "base_url": "https://api.flex.ai/v1", + "api_key_env": "FLEXAI_API_KEY", + "api_base_env": "FLEXAI_API_BASE", + "base_class": "openai_gpt", + "supported_endpoints": ["/v1/chat/completions", "/v1/responses"] } } diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index a6da2c1fb09..918552f46e9 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -51157,6 +51157,276 @@ ], "supports_audio_output": true }, + "flexai/bge-m3": { + "input_cost_per_token": 1e-08, + "litellm_provider": "flexai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "embedding", + "output_cost_per_token": 0.0, + "output_vector_size": 1024, + "source": "https://platform.flex.ai/models" + }, + "flexai/DeepSeek-V4-Flash-0731": { + "cache_read_input_token_cost": 1.2e-08, + "input_cost_per_token": 8e-08, + "litellm_provider": "flexai", + "max_input_tokens": 786432, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 1.8e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/FLUX.1-schnell": { + "input_cost_per_image": 0.0005, + "litellm_provider": "flexai", + "mode": "image_generation", + "source": "https://platform.flex.ai/models" + }, + "flexai/gemma-4-26B-A4B-it": { + "cache_read_input_token_cost": 9e-09, + "input_cost_per_token": 6e-08, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "flexai/gemma-4-31b-it": { + "cache_read_input_token_cost": 1.35e-08, + "input_cost_per_token": 9e-08, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3.4e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "flexai/GLM-4.5-Air-FP8": { + "cache_read_input_token_cost": 1.87e-08, + "input_cost_per_token": 1.25e-07, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 8.5e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/GLM-5.2": { + "cache_read_input_token_cost": 6.03e-08, + "input_cost_per_token": 4.018e-07, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.2628e-06, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "flexai/gpt-oss-120b": { + "cache_read_input_token_cost": 5.9e-09, + "input_cost_per_token": 3.9e-08, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/gpt-oss-20b": { + "cache_read_input_token_cost": 4.5e-09, + "input_cost_per_token": 3e-08, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1.3e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Kokoro-82M": { + "input_cost_per_character": 1e-05, + "litellm_provider": "flexai", + "mode": "audio_speech", + "source": "https://platform.flex.ai/models", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, + "flexai/Llama-3.3-70B-Instruct-FP8": { + "cache_read_input_token_cost": 1.5e-08, + "input_cost_per_token": 1e-07, + "litellm_provider": "flexai", + "max_input_tokens": 65536, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 3.2e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Meta-Llama-3.1-8B-Instruct-FP8": { + "cache_read_input_token_cost": 3e-09, + "input_cost_per_token": 2e-08, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 3e-08, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Mistral-Nemo-Instruct-2407-FP8": { + "cache_read_input_token_cost": 2.7e-09, + "input_cost_per_token": 1.8e-08, + "litellm_provider": "flexai", + "max_input_tokens": 32768, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 3e-08, + "source": "https://platform.flex.ai/models", + "supports_response_schema": true + }, + "flexai/PaddleOCR-VL": { + "input_cost_per_token": 1.4e-07, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 8e-07, + "source": "https://platform.flex.ai/models", + "supports_vision": true + }, + "flexai/parakeet-tdt-0.6b-v3": { + "input_cost_per_second": 2.5e-05, + "litellm_provider": "flexai", + "mode": "audio_transcription", + "output_cost_per_second": 0.0, + "source": "https://platform.flex.ai/models", + "supported_endpoints": [ + "/v1/audio/transcriptions" + ] + }, + "flexai/Qwen3-30B-A3B-Thinking-2507-FP8": { + "cache_read_input_token_cost": 1.2e-08, + "input_cost_per_token": 8e-08, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Qwen3-8B-FP8": { + "cache_read_input_token_cost": 3e-09, + "input_cost_per_token": 2e-08, + "litellm_provider": "flexai", + "max_input_tokens": 40960, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Qwen3-Coder-30B-A3B-Instruct-FP8": { + "cache_read_input_token_cost": 1.05e-08, + "input_cost_per_token": 7e-08, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 2.6e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Qwen3.5-9B": { + "cache_read_input_token_cost": 1.5e-08, + "input_cost_per_token": 1e-07, + "litellm_provider": "flexai", + "max_input_tokens": 256000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.5e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "flexai/Qwen3.6-35B-A3B-FP8": { + "cache_read_input_token_cost": 1.5e-08, + "input_cost_per_token": 1e-07, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.5e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/whisper-large-v3-turbo": { + "input_cost_per_second": 1.1667e-05, + "litellm_provider": "flexai", + "mode": "audio_transcription", + "output_cost_per_second": 0.0, + "source": "https://platform.flex.ai/models", + "supported_endpoints": [ + "/v1/audio/transcriptions" + ] + }, "fallback_generalizations": { "rules": [ { diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 73f46bd2181..c1db795e7e0 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3809,6 +3809,7 @@ class LlmProviders(str, Enum): SCX_AI = "scx-ai" DARKBLOOM = "darkbloom" META = "meta" + FLEXAI = "flexai" LITELLM_AGENT = "litellm_agent" CURSOR = "cursor" BEDROCK_MANTLE = "bedrock_mantle" diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a6da2c1fb09..918552f46e9 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -51157,6 +51157,276 @@ ], "supports_audio_output": true }, + "flexai/bge-m3": { + "input_cost_per_token": 1e-08, + "litellm_provider": "flexai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "embedding", + "output_cost_per_token": 0.0, + "output_vector_size": 1024, + "source": "https://platform.flex.ai/models" + }, + "flexai/DeepSeek-V4-Flash-0731": { + "cache_read_input_token_cost": 1.2e-08, + "input_cost_per_token": 8e-08, + "litellm_provider": "flexai", + "max_input_tokens": 786432, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 1.8e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/FLUX.1-schnell": { + "input_cost_per_image": 0.0005, + "litellm_provider": "flexai", + "mode": "image_generation", + "source": "https://platform.flex.ai/models" + }, + "flexai/gemma-4-26B-A4B-it": { + "cache_read_input_token_cost": 9e-09, + "input_cost_per_token": 6e-08, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "flexai/gemma-4-31b-it": { + "cache_read_input_token_cost": 1.35e-08, + "input_cost_per_token": 9e-08, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3.4e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "flexai/GLM-4.5-Air-FP8": { + "cache_read_input_token_cost": 1.87e-08, + "input_cost_per_token": 1.25e-07, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 8.5e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/GLM-5.2": { + "cache_read_input_token_cost": 6.03e-08, + "input_cost_per_token": 4.018e-07, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.2628e-06, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "flexai/gpt-oss-120b": { + "cache_read_input_token_cost": 5.9e-09, + "input_cost_per_token": 3.9e-08, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/gpt-oss-20b": { + "cache_read_input_token_cost": 4.5e-09, + "input_cost_per_token": 3e-08, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1.3e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Kokoro-82M": { + "input_cost_per_character": 1e-05, + "litellm_provider": "flexai", + "mode": "audio_speech", + "source": "https://platform.flex.ai/models", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, + "flexai/Llama-3.3-70B-Instruct-FP8": { + "cache_read_input_token_cost": 1.5e-08, + "input_cost_per_token": 1e-07, + "litellm_provider": "flexai", + "max_input_tokens": 65536, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 3.2e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Meta-Llama-3.1-8B-Instruct-FP8": { + "cache_read_input_token_cost": 3e-09, + "input_cost_per_token": 2e-08, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 3e-08, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Mistral-Nemo-Instruct-2407-FP8": { + "cache_read_input_token_cost": 2.7e-09, + "input_cost_per_token": 1.8e-08, + "litellm_provider": "flexai", + "max_input_tokens": 32768, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 3e-08, + "source": "https://platform.flex.ai/models", + "supports_response_schema": true + }, + "flexai/PaddleOCR-VL": { + "input_cost_per_token": 1.4e-07, + "litellm_provider": "flexai", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 8e-07, + "source": "https://platform.flex.ai/models", + "supports_vision": true + }, + "flexai/parakeet-tdt-0.6b-v3": { + "input_cost_per_second": 2.5e-05, + "litellm_provider": "flexai", + "mode": "audio_transcription", + "output_cost_per_second": 0.0, + "source": "https://platform.flex.ai/models", + "supported_endpoints": [ + "/v1/audio/transcriptions" + ] + }, + "flexai/Qwen3-30B-A3B-Thinking-2507-FP8": { + "cache_read_input_token_cost": 1.2e-08, + "input_cost_per_token": 8e-08, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Qwen3-8B-FP8": { + "cache_read_input_token_cost": 3e-09, + "input_cost_per_token": 2e-08, + "litellm_provider": "flexai", + "max_input_tokens": 40960, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Qwen3-Coder-30B-A3B-Instruct-FP8": { + "cache_read_input_token_cost": 1.05e-08, + "input_cost_per_token": 7e-08, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 2.6e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/Qwen3.5-9B": { + "cache_read_input_token_cost": 1.5e-08, + "input_cost_per_token": 1e-07, + "litellm_provider": "flexai", + "max_input_tokens": 256000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.5e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "flexai/Qwen3.6-35B-A3B-FP8": { + "cache_read_input_token_cost": 1.5e-08, + "input_cost_per_token": 1e-07, + "litellm_provider": "flexai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.5e-07, + "source": "https://platform.flex.ai/models", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "flexai/whisper-large-v3-turbo": { + "input_cost_per_second": 1.1667e-05, + "litellm_provider": "flexai", + "mode": "audio_transcription", + "output_cost_per_second": 0.0, + "source": "https://platform.flex.ai/models", + "supported_endpoints": [ + "/v1/audio/transcriptions" + ] + }, "fallback_generalizations": { "rules": [ { diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 1d8d374c2c4..0fff8ba61b5 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -2989,6 +2989,23 @@ "rerank": false, "a2a": false } + }, + "flexai": { + "display_name": "FlexAI (`flexai`)", + "url": "https://docs.litellm.ai/docs/providers/flexai", + "endpoints": { + "chat_completions": true, + "messages": false, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false + } } }, "endpoints": { diff --git a/tests/test_litellm/llms/openai_like/test_flexai_provider.py b/tests/test_litellm/llms/openai_like/test_flexai_provider.py new file mode 100644 index 00000000000..dc4c65b3ab1 --- /dev/null +++ b/tests/test_litellm/llms/openai_like/test_flexai_provider.py @@ -0,0 +1,332 @@ +""" +Tests for FlexAI provider configuration and integration. +""" + +import json +import os + +import litellm + +# a flagship model FlexAI pins always-on; used only for provider-resolution +# assertions, which do not require the model to be reachable +FLEXAI_MODEL = "gpt-oss-120b" +FLEXAI_BASE_URL = "https://api.flex.ai/v1" + + +class TestFlexAIProviderConfig: + """Test FlexAI provider configuration""" + + def test_flexai_in_provider_list(self): + """Test that flexai is in the provider list""" + from litellm import LlmProviders + + assert hasattr(LlmProviders, "FLEXAI") + assert LlmProviders.FLEXAI.value == "flexai" + assert "flexai" in litellm.provider_list + + def test_flexai_json_config_exists(self): + """Test that flexai is configured in providers.json""" + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + assert JSONProviderRegistry.exists("flexai") + + flexai = JSONProviderRegistry.get("flexai") + assert flexai is not None + assert flexai.base_url == FLEXAI_BASE_URL + assert flexai.api_key_env == "FLEXAI_API_KEY" + assert flexai.api_base_env == "FLEXAI_API_BASE" + + def test_flexai_supports_responses_api(self): + """Test that flexai declares Responses API support""" + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + assert JSONProviderRegistry.supports_responses_api("flexai") + + def test_flexai_in_openai_compatible_providers(self): + """Test that flexai is in the openai_compatible_providers list""" + from litellm.constants import ( + openai_compatible_endpoints, + openai_compatible_providers, + ) + + assert "flexai" in openai_compatible_providers + assert FLEXAI_BASE_URL in openai_compatible_endpoints + + def test_flexai_provider_resolution(self): + """Test that provider resolution finds flexai and returns the default base URL""" + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + model, provider, api_key, api_base = get_llm_provider( + model=f"flexai/{FLEXAI_MODEL}", + custom_llm_provider=None, + api_base=None, + api_key=None, + ) + + assert model == FLEXAI_MODEL + assert provider == "flexai" + assert api_base == FLEXAI_BASE_URL + + def test_flexai_api_base_override(self): + """Test that an explicit api_base / api_key overrides the default""" + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + model, provider, api_key, api_base = get_llm_provider( + model=f"flexai/{FLEXAI_MODEL}", + custom_llm_provider=None, + api_base="https://tokens.staging.flex.ai/v1", + api_key="sk-test", + ) + + assert provider == "flexai" + assert api_base == "https://tokens.staging.flex.ai/v1" + assert api_key == "sk-test" + + def test_flexai_env_key_resolution(self, monkeypatch): + """Test that the key is read from FLEXAI_API_KEY when not passed explicitly""" + from litellm.llms.openai_like.dynamic_config import create_config_class + from litellm.llms.openai_like.json_loader import JSONProviderRegistry + + monkeypatch.setenv("FLEXAI_API_KEY", "sk-flexai-env") + + config = create_config_class(JSONProviderRegistry.get("flexai"))() + api_base, api_key = config._get_openai_compatible_provider_info(None, None) + + assert api_base == FLEXAI_BASE_URL + assert api_key == "sk-flexai-env" + + def test_flexai_url_autodetection(self): + """Test that api_base=api.flex.ai/v1 auto-sets custom_llm_provider=flexai""" + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + model, provider, api_key, api_base = get_llm_provider( + model=FLEXAI_MODEL, + custom_llm_provider=None, + api_base=FLEXAI_BASE_URL, + api_key=None, + ) + + assert provider == "flexai" + assert api_base == FLEXAI_BASE_URL + + def test_flexai_url_autodetection_prefers_caller_key(self, monkeypatch): + """An explicit api_key must win over the server's FLEXAI_API_KEY. + + `completion()` overwrites api_key with dynamic_api_key whenever it is + set, so reading the environment unconditionally here would spend the + server's credential on a caller-supplied api_base. + """ + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("FLEXAI_API_KEY", "sk-server-env") + + _, provider, api_key, _ = get_llm_provider( + model=FLEXAI_MODEL, + custom_llm_provider=None, + api_base=FLEXAI_BASE_URL, + api_key="sk-caller", + ) + + assert provider == "flexai" + assert api_key == "sk-caller" + + def test_flexai_url_autodetection_falls_back_to_env_key(self, monkeypatch): + """With no explicit key, the server environment is still used""" + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("FLEXAI_API_KEY", "sk-server-env") + + _, provider, api_key, _ = get_llm_provider( + model=FLEXAI_MODEL, + custom_llm_provider=None, + api_base=FLEXAI_BASE_URL, + api_key=None, + ) + + assert provider == "flexai" + assert api_key == "sk-server-env" + + def test_flexai_router_config(self): + """Test that flexai can be used in Router configuration""" + from litellm import Router + + router = Router( + model_list=[ + { + "model_name": "flexai-chat", + "litellm_params": { + "model": f"flexai/{FLEXAI_MODEL}", + "api_key": "test-key", + }, + } + ] + ) + + assert len(router.model_list) == 1 + assert router.model_list[0]["model_name"] == "flexai-chat" + + +class TestFlexAICostMap: + """Test the FlexAI entries in the model cost map""" + + @staticmethod + def _cost_map(): + repo_root = os.path.abspath( + os.path.join(os.path.dirname(__file__), "../../../..") + ) + with open(os.path.join(repo_root, "model_prices_and_context_window.json")) as f: + return json.load(f) + + @staticmethod + def _backup_cost_map(): + repo_root = os.path.abspath( + os.path.join(os.path.dirname(__file__), "../../../..") + ) + path = os.path.join( + repo_root, "litellm", "model_prices_and_context_window_backup.json" + ) + with open(path) as f: + return json.load(f) + + def test_flexai_entries_present(self): + """Test that flexai models are priced in the cost map""" + entries = { + k: v for k, v in self._cost_map().items() if k.startswith("flexai/") + } + + assert entries, "no flexai/* entries in the cost map" + for key, entry in entries.items(): + assert entry["litellm_provider"] == "flexai", key + assert entry.get("mode"), key + + def test_flexai_chat_entries_are_priced(self): + """Test that every flexai chat model carries per-token pricing + context""" + entries = { + k: v + for k, v in self._cost_map().items() + if k.startswith("flexai/") and v.get("mode") == "chat" + } + + assert entries, "no flexai/* chat entries in the cost map" + for key, entry in entries.items(): + assert entry["input_cost_per_token"] > 0, key + assert entry["output_cost_per_token"] > 0, key + assert entry["max_input_tokens"] > 0, key + assert entry["max_output_tokens"] > 0, key + + def test_flexai_entries_match_packaged_backup(self): + """Test that the packaged backup carries the same flexai entries. + + litellm loads the packaged backup when the remote cost map is + unavailable (or when LITELLM_LOCAL_MODEL_COST_MAP=True), so the two + files must not drift. + """ + root = {k: v for k, v in self._cost_map().items() if k.startswith("flexai/")} + backup = { + k: v for k, v in self._backup_cost_map().items() if k.startswith("flexai/") + } + + assert root == backup + + def test_flexai_non_chat_entries_use_calculator_readable_fields(self): + """Non-chat entries must carry the cost fields the generic calculators read. + + `default_image_cost_calculator` reads `input_cost_per_image` / + `input_cost_per_pixel`, and `select_cost_metric_for_model` requires + `input_cost_per_character` (or `input_cost_per_token`) for speech. Using + the `output_*` variants instead makes cost calculation raise rather than + return the configured cost. + """ + readable_fields = { + "image_generation": ("input_cost_per_image", "input_cost_per_pixel"), + "audio_speech": ("input_cost_per_character", "input_cost_per_token"), + "audio_transcription": ("input_cost_per_second", "input_cost_per_token"), + "embedding": ("input_cost_per_token",), + } + + entries = { + k: v + for k, v in self._cost_map().items() + if k.startswith("flexai/") and v.get("mode") != "chat" + } + assert entries, "no non-chat flexai/* entries in the cost map" + + for key, entry in entries.items(): + options = readable_fields.get(entry["mode"]) + assert options, f"{key}: unhandled mode {entry['mode']}" + assert any(field in entry for field in options), ( + f"{key} (mode={entry['mode']}) has none of {options}" + ) + + def test_flexai_image_cost_calculation(self): + """Test that image cost resolves through the generic image calculator""" + from litellm.cost_calculator import default_image_cost_calculator + + cost_map = self._cost_map() + key = next( + k + for k, v in cost_map.items() + if k.startswith("flexai/") and v.get("mode") == "image_generation" + ) + per_image = cost_map[key]["input_cost_per_image"] + + for n in (1, 3): + cost = default_image_cost_calculator( + model=key, custom_llm_provider="flexai", n=n, quality=None, size=None + ) + assert cost == per_image * n + + def test_flexai_speech_cost_calculation(self): + """Test that TTS cost resolves per input character""" + from litellm.cost_calculator import cost_per_token + from litellm.litellm_core_utils.llm_cost_calc.utils import ( + select_cost_metric_for_model, + ) + + cost_map = self._cost_map() + key = next( + k + for k, v in cost_map.items() + if k.startswith("flexai/") and v.get("mode") == "audio_speech" + ) + model = key.split("/", 1)[1] + per_character = cost_map[key]["input_cost_per_character"] + + model_info = litellm.get_model_info(model=model, custom_llm_provider="flexai") + assert select_cost_metric_for_model(model_info) == "cost_per_character" + + prompt_cost, completion_cost = cost_per_token( + model=model, + custom_llm_provider="flexai", + call_type="speech", + prompt_characters=1000, + ) + assert prompt_cost + completion_cost == per_character * 1000 + + def test_flexai_cost_calculation(self): + """Test that completion_cost resolves for a flexai chat model""" + from litellm.types.utils import Choices, Message, ModelResponse, Usage + + cost_map = self._cost_map() + key = next( + k + for k, v in cost_map.items() + if k.startswith("flexai/") and v.get("mode") == "chat" + ) + entry = cost_map[key] + + response = ModelResponse( + model=key.split("/", 1)[1], + choices=[Choices(message=Message(role="assistant", content="ok"))], + usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500), + ) + response._hidden_params = {"custom_llm_provider": "flexai"} + + cost = litellm.completion_cost( + completion_response=response, custom_llm_provider="flexai", model=key + ) + + expected = ( + 1000 * entry["input_cost_per_token"] + 500 * entry["output_cost_per_token"] + ) + assert cost == expected