diff --git a/README.md b/README.md index 98c5343daee..e154099cdd9 100644 --- a/README.md +++ b/README.md @@ -292,6 +292,7 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse | [Bytez (`bytez`)](https://docs.litellm.ai/docs/providers/bytez) | ✅ | ✅ | ✅ | | | | | | | | | [Cerebras (`cerebras`)](https://docs.litellm.ai/docs/providers/cerebras) | ✅ | ✅ | ✅ | | | | | | | | | [Clarifai (`clarifai`)](https://docs.litellm.ai/docs/providers/clarifai) | ✅ | ✅ | ✅ | | | | | | | | +| [CLF AI Gateway (`clf_ai_gateway`)](https://docs.litellm.ai/docs/providers/clf_ai_gateway) | ✅ | ✅ | ✅ | | | | | | | | | [Cloudflare AI Workers (`cloudflare`)](https://docs.litellm.ai/docs/providers/cloudflare_workers) | ✅ | ✅ | ✅ | | | | | | | | | [Codestral (`codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | | | [Cognition (`cognition`)](https://docs.litellm.ai/docs/providers/cognition) | ✅ | ✅ | ✅ | | | | | | | | diff --git a/litellm/__init__.py b/litellm/__init__.py index e1da202b9ee..c2f090823c1 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -637,6 +637,7 @@ fal_ai_models: Set = set() fireworks_ai_models: Set = set() fireworks_ai_embedding_models: Set = set() deepinfra_models: Set = set() +clf_ai_gateway_models: set[str] = set() # mutable-ok: filled at import like sibling sets perplexity_models: Set = set() watsonx_models: Set = set() gemini_models: Set = set() @@ -839,6 +840,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: bedrock_converse_models.add(key) elif value.get("litellm_provider") == "deepinfra": deepinfra_models.add(key) + elif value.get("litellm_provider") == "clf_ai_gateway": # pyright: ignore[reportUnknownMemberType] # raw cost map + clf_ai_gateway_models.add(key) # pyright: ignore[reportUnknownArgumentType] # raw cost map key elif value.get("litellm_provider") == "perplexity": perplexity_models.add(key) elif value.get("litellm_provider") == "watsonx": @@ -1065,6 +1068,7 @@ model_list = list( | set(ollama_models) | bedrock_models | deepinfra_models + | clf_ai_gateway_models | perplexity_models | set(maritalk_models) | runwayml_models @@ -1170,6 +1174,7 @@ def _build_models_by_provider() -> dict: "ollama": ollama_models, "ollama_chat": ollama_models, "deepinfra": deepinfra_models, + "clf_ai_gateway": clf_ai_gateway_models, "perplexity": perplexity_models, "maritalk": maritalk_models, "watsonx": watsonx_models, @@ -1965,6 +1970,9 @@ if TYPE_CHECKING: LiteLLMProxyChatConfig as _LiteLLMProxyChatConfig, ) from .llms.deepinfra.chat.transformation import DeepInfraConfig as _DeepInfraConfig + from .llms.clf_ai_gateway.chat.transformation import ( + ClfAiGatewayConfig as _ClfAiGatewayConfig, + ) from .llms.llamafile.chat.transformation import ( LlamafileChatConfig as _LlamafileChatConfig, ) @@ -1994,6 +2002,7 @@ if TYPE_CHECKING: IBMWatsonXAIConfig: Type[_IBMWatsonXAIConfig] LiteLLMProxyChatConfig: Type[_LiteLLMProxyChatConfig] DeepInfraConfig: Type[_DeepInfraConfig] + ClfAiGatewayConfig: type[_ClfAiGatewayConfig] LlamafileChatConfig: Type[_LlamafileChatConfig] LMStudioChatConfig: Type[_LMStudioChatConfig] LmStudioEmbeddingConfig: Type[_LmStudioEmbeddingConfig] diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index aef3cbd9414..0f14426d536 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -219,6 +219,7 @@ LLM_CONFIG_NAMES: Final = ( "MistralEmbeddingConfig", "OpenAIImageVariationConfig", "DeepInfraConfig", + "ClfAiGatewayConfig", "DeepgramAudioTranscriptionConfig", "TopazImageVariationConfig", "OpenAITextCompletionConfig", @@ -908,6 +909,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { "OpenAIImageVariationConfig", ), "DeepInfraConfig": (".llms.deepinfra.chat.transformation", "DeepInfraConfig"), + "ClfAiGatewayConfig": ( + ".llms.clf_ai_gateway.chat.transformation", + "ClfAiGatewayConfig", + ), "DeepgramAudioTranscriptionConfig": ( ".llms.deepgram.audio_transcription.transformation", "DeepgramAudioTranscriptionConfig", diff --git a/litellm/constants.py b/litellm/constants.py index 39c10d71709..7a0de7129bd 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -708,6 +708,7 @@ LITELLM_CHAT_PROVIDERS: Final = [ "ollama", "ollama_chat", "deepinfra", + "clf_ai_gateway", "perplexity", "mistral", "groq", @@ -903,6 +904,7 @@ openai_compatible_endpoints: Final[list] = [ "api.perplexity.ai", "api.endpoints.anyscale.com/v1", "api.deepinfra.com/v1/openai", + "api.clfaigateway.dev/v1", "api.mistral.ai/v1", "codestral.mistral.ai/v1/chat/completions", "codestral.mistral.ai/v1/fim/completions", @@ -968,6 +970,7 @@ openai_compatible_providers: Final[list] = [ "deepseek", "tencent", "deepinfra", + "clf_ai_gateway", "perplexity", "xinference", "xai", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index d4642ae2aad..7391399ea54 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -269,6 +269,10 @@ def get_llm_provider( elif endpoint == "api.deepinfra.com/v1/openai": custom_llm_provider = "deepinfra" dynamic_api_key = get_secret_str("DEEPINFRA_API_KEY") + elif endpoint == "api.clfaigateway.dev/v1": + custom_llm_provider = "clf_ai_gateway" # rebind-ok: endpoint inference, like every branch + clf_env_key = get_secret_str("CLF_AI_GATEWAY_API_KEY") + dynamic_api_key = api_key or clf_env_key # rebind-ok: caller key wins, like every branch elif endpoint == "api.mistral.ai/v1": custom_llm_provider = "mistral" dynamic_api_key = get_secret_str("MISTRAL_API_KEY") @@ -622,6 +626,12 @@ def _get_openai_compatible_provider_info( api_base, dynamic_api_key, ) = litellm.DeepInfraConfig()._get_openai_compatible_provider_info(api_base, api_key) + elif custom_llm_provider == "clf_ai_gateway": + cfg = litellm.ClfAiGatewayConfig() + ( + api_base, # rebind-ok: resolved base/key are what this function returns, like every branch + dynamic_api_key, # rebind-ok: see above + ) = cfg._get_openai_compatible_provider_info(api_base, api_key) # pyright: ignore[reportPrivateUsage] # hook elif custom_llm_provider == "empower": api_base = api_base or get_secret("EMPOWER_API_BASE") or "https://app.empower.dev/api/v1" dynamic_api_key = api_key or get_secret_str("EMPOWER_API_KEY") diff --git a/litellm/litellm_core_utils/get_supported_openai_params.py b/litellm/litellm_core_utils/get_supported_openai_params.py index c635cf828eb..2dcdb592aa8 100644 --- a/litellm/litellm_core_utils/get_supported_openai_params.py +++ b/litellm/litellm_core_utils/get_supported_openai_params.py @@ -222,6 +222,8 @@ def get_supported_openai_params( return litellm.PetalsConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "deepinfra": return litellm.DeepInfraConfig().get_supported_openai_params(model=model) + elif custom_llm_provider == "clf_ai_gateway": + return litellm.ClfAiGatewayConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "perplexity": return litellm.PerplexityChatConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "nscale": diff --git a/litellm/llms/clf_ai_gateway/chat/transformation.py b/litellm/llms/clf_ai_gateway/chat/transformation.py new file mode 100644 index 00000000000..3201c14aeea --- /dev/null +++ b/litellm/llms/clf_ai_gateway/chat/transformation.py @@ -0,0 +1,63 @@ +from typing import Final + +import litellm +from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig +from litellm.secret_managers.main import get_secret_str + +# Narrower than the generic OpenAI surface on purpose: the gateway rejects legacy +# top-level fields (``functions``, ``function_call``, ``logit_bias``) with a 400. +SUPPORTED_OPENAI_PARAMS: Final = ( + "stream", + "stream_options", + "frequency_penalty", + "presence_penalty", + "max_tokens", + "max_completion_tokens", + "n", + "stop", + "temperature", + "top_p", + "seed", + "response_format", + "tools", + "tool_choice", + "parallel_tool_calls", + "user", +) + + +class ClfAiGatewayConfig(OpenAIGPTConfig): + """ + Reference: https://clfaigateway.dev/docs + + CLF AI Gateway is an OpenAI-compatible gateway (chat completions, streaming, tool + calling, structured output, ``reasoning_effort``) serving open-weight models + (GLM, Kimi, DeepSeek, Qwen) on Cloudflare Workers AI upstream. Model ids are the + gateway's canonical names, e.g. ``clf_ai_gateway/glm-5.3``. + + The gateway validates request bodies strictly and rejects unknown or legacy + top-level fields with a 400 (``functions``, ``function_call``, ``logit_bias``), + so the supported-params list below is deliberately narrower than the generic + OpenAI surface. + """ + + @property + def custom_llm_provider(self) -> str | None: + return "clf_ai_gateway" + + def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: matches OpenAIGPTConfig's list return + supports_reasoning: Final = litellm.supports_reasoning( + model=model, + custom_llm_provider=self.custom_llm_provider, + ) + extra: Final = ("reasoning_effort",) if supports_reasoning else () + return [*SUPPORTED_OPENAI_PARAMS, *extra] # mutable-ok: one-shot build, list per base contract + + def _get_openai_compatible_provider_info( + self, api_base: str | None, api_key: str | None + ) -> tuple[str | None, str | None]: + resolved_api_base: Final = ( + api_base or get_secret_str("CLF_AI_GATEWAY_API_BASE") or "https://api.clfaigateway.dev/v1" + ) + dynamic_api_key: Final = api_key or get_secret_str("CLF_AI_GATEWAY_API_KEY") + return resolved_api_base, dynamic_api_key diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index fc48c17b506..4a694e0ad14 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -78773,5 +78773,179 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true + }, + "clf_ai_gateway/glm-5.3": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_max_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-5.3-flash": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 3e-07, + "cache_read_input_token_cost": 1.8e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-5.2": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-4.7-flash": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.6e-08, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/deepseek-v4-pro": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 7.92e-07, + "output_cost_per_token": 2.376e-06, + "cache_read_input_token_cost": 2.6e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/deepseek-v4-flash": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 2.64e-07, + "output_cost_per_token": 7.92e-07, + "cache_read_input_token_cost": 8e-09, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/kimi-k2.6": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 5.7e-07, + "output_cost_per_token": 2.4e-06, + "cache_read_input_token_cost": 9.6e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/kimi-k2.7-code": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 5.7e-07, + "output_cost_per_token": 2.4e-06, + "cache_read_input_token_cost": 1.14e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/qwen3.8-27b": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 2.7e-07, + "output_cost_per_token": 1.92e-06, + "cache_read_input_token_cost": 2.7e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" } } diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index ad6e5857218..8e74214082c 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -492,6 +492,24 @@ "interactions": true } }, + "clf_ai_gateway": { + "display_name": "CLF AI Gateway (`clf_ai_gateway`)", + "url": "https://docs.litellm.ai/docs/providers/clf_ai_gateway", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true, + "interactions": true + } + }, "cloudflare": { "display_name": "Cloudflare AI Workers (`cloudflare`)", "url": "https://docs.litellm.ai/docs/providers/cloudflare_workers", diff --git a/litellm/types/utils.py b/litellm/types/utils.py index dba15bc99a5..1cbbfcbd30d 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4033,6 +4033,7 @@ class LlmProviders(str, Enum): OLLAMA = "ollama" OLLAMA_CHAT = "ollama_chat" DEEPINFRA = "deepinfra" + CLF_AI_GATEWAY = "clf_ai_gateway" PERPLEXITY = "perplexity" MISTRAL = "mistral" MILVUS = "milvus" diff --git a/litellm/utils.py b/litellm/utils.py index 09b5067339d..e8f61e34db0 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -6787,6 +6787,11 @@ def validate_environment( keys_in_environment = True else: missing_keys.append("DEEPINFRA_API_KEY") + elif custom_llm_provider == "clf_ai_gateway": + if "CLF_AI_GATEWAY_API_KEY" in os.environ: + keys_in_environment = True + else: + missing_keys.append("CLF_AI_GATEWAY_API_KEY") elif custom_llm_provider == "featherless_ai": if "FEATHERLESS_AI_API_KEY" in os.environ: keys_in_environment = True @@ -8493,6 +8498,7 @@ class ProviderConfigManager: LlmProviders.OOBABOOGA: (lambda: litellm.OobaboogaConfig(), False), LlmProviders.OLLAMA_CHAT: (lambda: litellm.OllamaChatConfig(), False), LlmProviders.DEEPINFRA: (lambda: litellm.DeepInfraConfig(), False), + LlmProviders.CLF_AI_GATEWAY: (lambda: litellm.ClfAiGatewayConfig(), False), LlmProviders.PERPLEXITY: (lambda: litellm.PerplexityChatConfig(), False), LlmProviders.MISTRAL: (lambda: litellm.MistralConfig(), False), LlmProviders.CODESTRAL: (lambda: litellm.MistralConfig(), False), diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index fc48c17b506..4a694e0ad14 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -78773,5 +78773,179 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true + }, + "clf_ai_gateway/glm-5.3": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_max_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-5.3-flash": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 3e-07, + "cache_read_input_token_cost": 1.8e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-5.2": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-4.7-flash": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.6e-08, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/deepseek-v4-pro": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 7.92e-07, + "output_cost_per_token": 2.376e-06, + "cache_read_input_token_cost": 2.6e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/deepseek-v4-flash": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 2.64e-07, + "output_cost_per_token": 7.92e-07, + "cache_read_input_token_cost": 8e-09, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/kimi-k2.6": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 5.7e-07, + "output_cost_per_token": 2.4e-06, + "cache_read_input_token_cost": 9.6e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/kimi-k2.7-code": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 5.7e-07, + "output_cost_per_token": 2.4e-06, + "cache_read_input_token_cost": 1.14e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/qwen3.8-27b": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 2.7e-07, + "output_cost_per_token": 1.92e-06, + "cache_read_input_token_cost": 2.7e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" } } diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 44ef9363b64..72386c87bad 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -545,6 +545,24 @@ "interactions": true } }, + "clf_ai_gateway": { + "display_name": "CLF AI Gateway (`clf_ai_gateway`)", + "url": "https://docs.litellm.ai/docs/providers/clf_ai_gateway", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true, + "interactions": true + } + }, "cloudflare": { "display_name": "Cloudflare AI Workers (`cloudflare`)", "url": "https://docs.litellm.ai/docs/providers/cloudflare_workers", diff --git a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py new file mode 100644 index 00000000000..587103e1077 --- /dev/null +++ b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py @@ -0,0 +1,128 @@ +import pytest + +import litellm +from litellm.llms.clf_ai_gateway.chat.transformation import ClfAiGatewayConfig +from litellm.types.utils import LlmProviders + +MODELS = [ + "glm-5.3", + "glm-5.3-flash", + "glm-5.2", + "glm-4.7-flash", + "deepseek-v4-pro", + "deepseek-v4-flash", + "kimi-k2.6", + "kimi-k2.7-code", + "qwen3.8-27b", +] + + +@pytest.fixture(autouse=True) +def _local_cost_map(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + +def test_provider_is_registered() -> None: + assert LlmProviders.CLF_AI_GATEWAY.value == "clf_ai_gateway" + assert "clf_ai_gateway" in litellm.openai_compatible_providers + assert "api.clfaigateway.dev/v1" in litellm.openai_compatible_endpoints + assert "clf_ai_gateway" in litellm.provider_list + assert isinstance( + litellm.ProviderConfigManager.get_provider_chat_config(model="glm-5.3", provider=LlmProviders.CLF_AI_GATEWAY), + ClfAiGatewayConfig, + ) + + +def test_default_api_base_and_env_key(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + monkeypatch.delenv("CLF_AI_GATEWAY_API_BASE", raising=False) + api_base, api_key = ClfAiGatewayConfig()._get_openai_compatible_provider_info(api_base=None, api_key=None) + assert api_base == "https://api.clfaigateway.dev/v1" + assert api_key == "sk-gw-test" + + api_base, api_key = ClfAiGatewayConfig()._get_openai_compatible_provider_info( + api_base="https://example.test/v1", api_key="explicit" + ) + assert api_base == "https://example.test/v1" + assert api_key == "explicit" + + +def test_get_llm_provider_resolves_prefix(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + model, provider, key, api_base = litellm.get_llm_provider(model="clf_ai_gateway/glm-5.3") + assert model == "glm-5.3" + assert provider == "clf_ai_gateway" + assert key == "sk-gw-test" + assert api_base == "https://api.clfaigateway.dev/v1" + + +def test_get_llm_provider_infers_from_api_base(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + _, provider, _, _ = litellm.get_llm_provider(model="glm-5.3", api_base="https://api.clfaigateway.dev/v1") + assert provider == "clf_ai_gateway" + + _, provider, key, _ = litellm.get_llm_provider( + model="glm-5.3", api_base="https://api.clfaigateway.dev/v1", api_key="explicit" + ) + assert provider == "clf_ai_gateway" + assert key == "explicit" + + +def test_supported_params_match_gateway_surface() -> None: + params = ClfAiGatewayConfig().get_supported_openai_params(model="glm-5.3") + assert "reasoning_effort" in params + assert "tools" in params and "tool_choice" in params + # the gateway rejects these legacy/unknown fields with a 400 — never advertise them + for legacy in ("functions", "function_call", "logit_bias"): + assert legacy not in params + + +@pytest.mark.parametrize("model", MODELS) +def test_cost_map_entries(model: str) -> None: + key = f"clf_ai_gateway/{model}" + assert key in litellm.model_cost, f"missing cost map entry for {key}" + entry = litellm.model_cost[key] + assert entry["litellm_provider"] == "clf_ai_gateway" + assert entry["mode"] == "chat" + assert entry["max_output_tokens"] == 131072 + assert entry["input_cost_per_token"] > 0 + assert entry["output_cost_per_token"] > 0 + assert entry["supports_function_calling"] is True + assert entry["supports_reasoning"] is True + + +def test_vision_flags_match_measured_surface() -> None: + vision = {"glm-5.3-flash", "kimi-k2.6", "kimi-k2.7-code", "qwen3.8-27b"} + for model in MODELS: + entry = litellm.model_cost[f"clf_ai_gateway/{model}"] + assert bool(entry.get("supports_vision")) is (model in vision), model + + +def test_validate_environment_reports_env_key(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv("CLF_AI_GATEWAY_API_KEY", raising=False) + missing = litellm.validate_environment(model="clf_ai_gateway/glm-5.3") + assert missing["keys_in_environment"] is False + assert "CLF_AI_GATEWAY_API_KEY" in missing["missing_keys"] + + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + present = litellm.validate_environment(model="clf_ai_gateway/glm-5.3") + assert present["keys_in_environment"] is True + assert present["missing_keys"] == [] + + +def test_get_supported_openai_params_dispatches_to_provider_config() -> None: + params = litellm.get_supported_openai_params(model="glm-5.3", custom_llm_provider="clf_ai_gateway") + assert params is not None + assert "reasoning_effort" in params + assert "logit_bias" not in params + + +def test_completion_routes_without_network(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + response = litellm.completion( + model="clf_ai_gateway/glm-5.3", + messages=[{"role": "user", "content": "hi"}], + mock_response="hello from mock", + ) + assert response.choices[0].message.content == "hello from mock"