From 935e8178ab954f94f861c54765d8df36334a391a Mon Sep 17 00:00:00 2001 From: bap1106 Date: Wed, 2 Sep 2026 20:56:29 +0700 Subject: [PATCH 1/8] feat(providers): add CLF AI Gateway (clf_ai_gateway), OpenAI-compatible, 9 models OpenAI-compatible gateway at https://api.clfaigateway.dev/v1 serving open-weight models (GLM, Kimi, DeepSeek, Qwen) on Cloudflare Workers AI upstream. Follows the DeepInfra pattern: ClfAiGatewayConfig(OpenAIGPTConfig), CLF_AI_GATEWAY_API_KEY / _API_BASE, provider registered in enum, provider lists, endpoint inference, lazy-import registry, ProviderConfigManager, validate_environment. Supported params deliberately omit functions/function_call/logit_bias (gateway returns 400 for them). 9 cost-map entries (root + backup) with measured context/output limits and currently billed prices; endpoint-support entry; README row. Mocked tests only. Co-Authored-By: Claude Fable 5 --- README.md | 1 + litellm/__init__.py | 9 + litellm/_lazy_imports_registry.py | 5 + litellm/constants.py | 3 + .../get_llm_provider_logic.py | 10 + .../get_supported_openai_params.py | 2 + .../clf_ai_gateway/chat/transformation.py | 63 +++++++ ...odel_prices_and_context_window_backup.json | 174 ++++++++++++++++++ .../provider_endpoints_support_backup.json | 18 ++ litellm/types/utils.py | 1 + litellm/utils.py | 6 + model_prices_and_context_window.json | 174 ++++++++++++++++++ provider_endpoints_support.json | 18 ++ ...test_clf_ai_gateway_chat_transformation.py | 128 +++++++++++++ 14 files changed, 612 insertions(+) create mode 100644 litellm/llms/clf_ai_gateway/chat/transformation.py create mode 100644 tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py diff --git a/README.md b/README.md index 98c5343daee..e154099cdd9 100644 --- a/README.md +++ b/README.md @@ -292,6 +292,7 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse | [Bytez (`bytez`)](https://docs.litellm.ai/docs/providers/bytez) | ✅ | ✅ | ✅ | | | | | | | | | [Cerebras (`cerebras`)](https://docs.litellm.ai/docs/providers/cerebras) | ✅ | ✅ | ✅ | | | | | | | | | [Clarifai (`clarifai`)](https://docs.litellm.ai/docs/providers/clarifai) | ✅ | ✅ | ✅ | | | | | | | | +| [CLF AI Gateway (`clf_ai_gateway`)](https://docs.litellm.ai/docs/providers/clf_ai_gateway) | ✅ | ✅ | ✅ | | | | | | | | | [Cloudflare AI Workers (`cloudflare`)](https://docs.litellm.ai/docs/providers/cloudflare_workers) | ✅ | ✅ | ✅ | | | | | | | | | [Codestral (`codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | | | [Cognition (`cognition`)](https://docs.litellm.ai/docs/providers/cognition) | ✅ | ✅ | ✅ | | | | | | | | diff --git a/litellm/__init__.py b/litellm/__init__.py index e1da202b9ee..c2f090823c1 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -637,6 +637,7 @@ fal_ai_models: Set = set() fireworks_ai_models: Set = set() fireworks_ai_embedding_models: Set = set() deepinfra_models: Set = set() +clf_ai_gateway_models: set[str] = set() # mutable-ok: filled at import like sibling sets perplexity_models: Set = set() watsonx_models: Set = set() gemini_models: Set = set() @@ -839,6 +840,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None: bedrock_converse_models.add(key) elif value.get("litellm_provider") == "deepinfra": deepinfra_models.add(key) + elif value.get("litellm_provider") == "clf_ai_gateway": # pyright: ignore[reportUnknownMemberType] # raw cost map + clf_ai_gateway_models.add(key) # pyright: ignore[reportUnknownArgumentType] # raw cost map key elif value.get("litellm_provider") == "perplexity": perplexity_models.add(key) elif value.get("litellm_provider") == "watsonx": @@ -1065,6 +1068,7 @@ model_list = list( | set(ollama_models) | bedrock_models | deepinfra_models + | clf_ai_gateway_models | perplexity_models | set(maritalk_models) | runwayml_models @@ -1170,6 +1174,7 @@ def _build_models_by_provider() -> dict: "ollama": ollama_models, "ollama_chat": ollama_models, "deepinfra": deepinfra_models, + "clf_ai_gateway": clf_ai_gateway_models, "perplexity": perplexity_models, "maritalk": maritalk_models, "watsonx": watsonx_models, @@ -1965,6 +1970,9 @@ if TYPE_CHECKING: LiteLLMProxyChatConfig as _LiteLLMProxyChatConfig, ) from .llms.deepinfra.chat.transformation import DeepInfraConfig as _DeepInfraConfig + from .llms.clf_ai_gateway.chat.transformation import ( + ClfAiGatewayConfig as _ClfAiGatewayConfig, + ) from .llms.llamafile.chat.transformation import ( LlamafileChatConfig as _LlamafileChatConfig, ) @@ -1994,6 +2002,7 @@ if TYPE_CHECKING: IBMWatsonXAIConfig: Type[_IBMWatsonXAIConfig] LiteLLMProxyChatConfig: Type[_LiteLLMProxyChatConfig] DeepInfraConfig: Type[_DeepInfraConfig] + ClfAiGatewayConfig: type[_ClfAiGatewayConfig] LlamafileChatConfig: Type[_LlamafileChatConfig] LMStudioChatConfig: Type[_LMStudioChatConfig] LmStudioEmbeddingConfig: Type[_LmStudioEmbeddingConfig] diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index aef3cbd9414..0f14426d536 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -219,6 +219,7 @@ LLM_CONFIG_NAMES: Final = ( "MistralEmbeddingConfig", "OpenAIImageVariationConfig", "DeepInfraConfig", + "ClfAiGatewayConfig", "DeepgramAudioTranscriptionConfig", "TopazImageVariationConfig", "OpenAITextCompletionConfig", @@ -908,6 +909,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { "OpenAIImageVariationConfig", ), "DeepInfraConfig": (".llms.deepinfra.chat.transformation", "DeepInfraConfig"), + "ClfAiGatewayConfig": ( + ".llms.clf_ai_gateway.chat.transformation", + "ClfAiGatewayConfig", + ), "DeepgramAudioTranscriptionConfig": ( ".llms.deepgram.audio_transcription.transformation", "DeepgramAudioTranscriptionConfig", diff --git a/litellm/constants.py b/litellm/constants.py index 39c10d71709..7a0de7129bd 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -708,6 +708,7 @@ LITELLM_CHAT_PROVIDERS: Final = [ "ollama", "ollama_chat", "deepinfra", + "clf_ai_gateway", "perplexity", "mistral", "groq", @@ -903,6 +904,7 @@ openai_compatible_endpoints: Final[list] = [ "api.perplexity.ai", "api.endpoints.anyscale.com/v1", "api.deepinfra.com/v1/openai", + "api.clfaigateway.dev/v1", "api.mistral.ai/v1", "codestral.mistral.ai/v1/chat/completions", "codestral.mistral.ai/v1/fim/completions", @@ -968,6 +970,7 @@ openai_compatible_providers: Final[list] = [ "deepseek", "tencent", "deepinfra", + "clf_ai_gateway", "perplexity", "xinference", "xai", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index d4642ae2aad..7391399ea54 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -269,6 +269,10 @@ def get_llm_provider( elif endpoint == "api.deepinfra.com/v1/openai": custom_llm_provider = "deepinfra" dynamic_api_key = get_secret_str("DEEPINFRA_API_KEY") + elif endpoint == "api.clfaigateway.dev/v1": + custom_llm_provider = "clf_ai_gateway" # rebind-ok: endpoint inference, like every branch + clf_env_key = get_secret_str("CLF_AI_GATEWAY_API_KEY") + dynamic_api_key = api_key or clf_env_key # rebind-ok: caller key wins, like every branch elif endpoint == "api.mistral.ai/v1": custom_llm_provider = "mistral" dynamic_api_key = get_secret_str("MISTRAL_API_KEY") @@ -622,6 +626,12 @@ def _get_openai_compatible_provider_info( api_base, dynamic_api_key, ) = litellm.DeepInfraConfig()._get_openai_compatible_provider_info(api_base, api_key) + elif custom_llm_provider == "clf_ai_gateway": + cfg = litellm.ClfAiGatewayConfig() + ( + api_base, # rebind-ok: resolved base/key are what this function returns, like every branch + dynamic_api_key, # rebind-ok: see above + ) = cfg._get_openai_compatible_provider_info(api_base, api_key) # pyright: ignore[reportPrivateUsage] # hook elif custom_llm_provider == "empower": api_base = api_base or get_secret("EMPOWER_API_BASE") or "https://app.empower.dev/api/v1" dynamic_api_key = api_key or get_secret_str("EMPOWER_API_KEY") diff --git a/litellm/litellm_core_utils/get_supported_openai_params.py b/litellm/litellm_core_utils/get_supported_openai_params.py index c635cf828eb..2dcdb592aa8 100644 --- a/litellm/litellm_core_utils/get_supported_openai_params.py +++ b/litellm/litellm_core_utils/get_supported_openai_params.py @@ -222,6 +222,8 @@ def get_supported_openai_params( return litellm.PetalsConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "deepinfra": return litellm.DeepInfraConfig().get_supported_openai_params(model=model) + elif custom_llm_provider == "clf_ai_gateway": + return litellm.ClfAiGatewayConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "perplexity": return litellm.PerplexityChatConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "nscale": diff --git a/litellm/llms/clf_ai_gateway/chat/transformation.py b/litellm/llms/clf_ai_gateway/chat/transformation.py new file mode 100644 index 00000000000..3201c14aeea --- /dev/null +++ b/litellm/llms/clf_ai_gateway/chat/transformation.py @@ -0,0 +1,63 @@ +from typing import Final + +import litellm +from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig +from litellm.secret_managers.main import get_secret_str + +# Narrower than the generic OpenAI surface on purpose: the gateway rejects legacy +# top-level fields (``functions``, ``function_call``, ``logit_bias``) with a 400. +SUPPORTED_OPENAI_PARAMS: Final = ( + "stream", + "stream_options", + "frequency_penalty", + "presence_penalty", + "max_tokens", + "max_completion_tokens", + "n", + "stop", + "temperature", + "top_p", + "seed", + "response_format", + "tools", + "tool_choice", + "parallel_tool_calls", + "user", +) + + +class ClfAiGatewayConfig(OpenAIGPTConfig): + """ + Reference: https://clfaigateway.dev/docs + + CLF AI Gateway is an OpenAI-compatible gateway (chat completions, streaming, tool + calling, structured output, ``reasoning_effort``) serving open-weight models + (GLM, Kimi, DeepSeek, Qwen) on Cloudflare Workers AI upstream. Model ids are the + gateway's canonical names, e.g. ``clf_ai_gateway/glm-5.3``. + + The gateway validates request bodies strictly and rejects unknown or legacy + top-level fields with a 400 (``functions``, ``function_call``, ``logit_bias``), + so the supported-params list below is deliberately narrower than the generic + OpenAI surface. + """ + + @property + def custom_llm_provider(self) -> str | None: + return "clf_ai_gateway" + + def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: matches OpenAIGPTConfig's list return + supports_reasoning: Final = litellm.supports_reasoning( + model=model, + custom_llm_provider=self.custom_llm_provider, + ) + extra: Final = ("reasoning_effort",) if supports_reasoning else () + return [*SUPPORTED_OPENAI_PARAMS, *extra] # mutable-ok: one-shot build, list per base contract + + def _get_openai_compatible_provider_info( + self, api_base: str | None, api_key: str | None + ) -> tuple[str | None, str | None]: + resolved_api_base: Final = ( + api_base or get_secret_str("CLF_AI_GATEWAY_API_BASE") or "https://api.clfaigateway.dev/v1" + ) + dynamic_api_key: Final = api_key or get_secret_str("CLF_AI_GATEWAY_API_KEY") + return resolved_api_base, dynamic_api_key diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index fc48c17b506..4a694e0ad14 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -78773,5 +78773,179 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true + }, + "clf_ai_gateway/glm-5.3": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_max_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-5.3-flash": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 3e-07, + "cache_read_input_token_cost": 1.8e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-5.2": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-4.7-flash": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.6e-08, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/deepseek-v4-pro": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 7.92e-07, + "output_cost_per_token": 2.376e-06, + "cache_read_input_token_cost": 2.6e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/deepseek-v4-flash": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 2.64e-07, + "output_cost_per_token": 7.92e-07, + "cache_read_input_token_cost": 8e-09, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/kimi-k2.6": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 5.7e-07, + "output_cost_per_token": 2.4e-06, + "cache_read_input_token_cost": 9.6e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/kimi-k2.7-code": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 5.7e-07, + "output_cost_per_token": 2.4e-06, + "cache_read_input_token_cost": 1.14e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/qwen3.8-27b": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 2.7e-07, + "output_cost_per_token": 1.92e-06, + "cache_read_input_token_cost": 2.7e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" } } diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index ad6e5857218..8e74214082c 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -492,6 +492,24 @@ "interactions": true } }, + "clf_ai_gateway": { + "display_name": "CLF AI Gateway (`clf_ai_gateway`)", + "url": "https://docs.litellm.ai/docs/providers/clf_ai_gateway", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true, + "interactions": true + } + }, "cloudflare": { "display_name": "Cloudflare AI Workers (`cloudflare`)", "url": "https://docs.litellm.ai/docs/providers/cloudflare_workers", diff --git a/litellm/types/utils.py b/litellm/types/utils.py index dba15bc99a5..1cbbfcbd30d 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4033,6 +4033,7 @@ class LlmProviders(str, Enum): OLLAMA = "ollama" OLLAMA_CHAT = "ollama_chat" DEEPINFRA = "deepinfra" + CLF_AI_GATEWAY = "clf_ai_gateway" PERPLEXITY = "perplexity" MISTRAL = "mistral" MILVUS = "milvus" diff --git a/litellm/utils.py b/litellm/utils.py index 09b5067339d..e8f61e34db0 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -6787,6 +6787,11 @@ def validate_environment( keys_in_environment = True else: missing_keys.append("DEEPINFRA_API_KEY") + elif custom_llm_provider == "clf_ai_gateway": + if "CLF_AI_GATEWAY_API_KEY" in os.environ: + keys_in_environment = True + else: + missing_keys.append("CLF_AI_GATEWAY_API_KEY") elif custom_llm_provider == "featherless_ai": if "FEATHERLESS_AI_API_KEY" in os.environ: keys_in_environment = True @@ -8493,6 +8498,7 @@ class ProviderConfigManager: LlmProviders.OOBABOOGA: (lambda: litellm.OobaboogaConfig(), False), LlmProviders.OLLAMA_CHAT: (lambda: litellm.OllamaChatConfig(), False), LlmProviders.DEEPINFRA: (lambda: litellm.DeepInfraConfig(), False), + LlmProviders.CLF_AI_GATEWAY: (lambda: litellm.ClfAiGatewayConfig(), False), LlmProviders.PERPLEXITY: (lambda: litellm.PerplexityChatConfig(), False), LlmProviders.MISTRAL: (lambda: litellm.MistralConfig(), False), LlmProviders.CODESTRAL: (lambda: litellm.MistralConfig(), False), diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index fc48c17b506..4a694e0ad14 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -78773,5 +78773,179 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true + }, + "clf_ai_gateway/glm-5.3": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_max_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-5.3-flash": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 3e-07, + "cache_read_input_token_cost": 1.8e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-5.2": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/glm-4.7-flash": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.6e-08, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/deepseek-v4-pro": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 7.92e-07, + "output_cost_per_token": 2.376e-06, + "cache_read_input_token_cost": 2.6e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/deepseek-v4-flash": { + "max_tokens": 131072, + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "input_cost_per_token": 2.64e-07, + "output_cost_per_token": 7.92e-07, + "cache_read_input_token_cost": 8e-09, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/kimi-k2.6": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 5.7e-07, + "output_cost_per_token": 2.4e-06, + "cache_read_input_token_cost": 9.6e-08, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/kimi-k2.7-code": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 5.7e-07, + "output_cost_per_token": 2.4e-06, + "cache_read_input_token_cost": 1.14e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" + }, + "clf_ai_gateway/qwen3.8-27b": { + "max_tokens": 131072, + "max_input_tokens": 262144, + "max_output_tokens": 131072, + "input_cost_per_token": 2.7e-07, + "output_cost_per_token": 1.92e-06, + "cache_read_input_token_cost": 2.7e-07, + "litellm_provider": "clf_ai_gateway", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_low_reasoning_effort": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_prompt_caching": true, + "supports_vision": true, + "source": "https://clfaigateway.dev/models" } } diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 44ef9363b64..72386c87bad 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -545,6 +545,24 @@ "interactions": true } }, + "clf_ai_gateway": { + "display_name": "CLF AI Gateway (`clf_ai_gateway`)", + "url": "https://docs.litellm.ai/docs/providers/clf_ai_gateway", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true, + "interactions": true + } + }, "cloudflare": { "display_name": "Cloudflare AI Workers (`cloudflare`)", "url": "https://docs.litellm.ai/docs/providers/cloudflare_workers", diff --git a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py new file mode 100644 index 00000000000..587103e1077 --- /dev/null +++ b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py @@ -0,0 +1,128 @@ +import pytest + +import litellm +from litellm.llms.clf_ai_gateway.chat.transformation import ClfAiGatewayConfig +from litellm.types.utils import LlmProviders + +MODELS = [ + "glm-5.3", + "glm-5.3-flash", + "glm-5.2", + "glm-4.7-flash", + "deepseek-v4-pro", + "deepseek-v4-flash", + "kimi-k2.6", + "kimi-k2.7-code", + "qwen3.8-27b", +] + + +@pytest.fixture(autouse=True) +def _local_cost_map(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + +def test_provider_is_registered() -> None: + assert LlmProviders.CLF_AI_GATEWAY.value == "clf_ai_gateway" + assert "clf_ai_gateway" in litellm.openai_compatible_providers + assert "api.clfaigateway.dev/v1" in litellm.openai_compatible_endpoints + assert "clf_ai_gateway" in litellm.provider_list + assert isinstance( + litellm.ProviderConfigManager.get_provider_chat_config(model="glm-5.3", provider=LlmProviders.CLF_AI_GATEWAY), + ClfAiGatewayConfig, + ) + + +def test_default_api_base_and_env_key(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + monkeypatch.delenv("CLF_AI_GATEWAY_API_BASE", raising=False) + api_base, api_key = ClfAiGatewayConfig()._get_openai_compatible_provider_info(api_base=None, api_key=None) + assert api_base == "https://api.clfaigateway.dev/v1" + assert api_key == "sk-gw-test" + + api_base, api_key = ClfAiGatewayConfig()._get_openai_compatible_provider_info( + api_base="https://example.test/v1", api_key="explicit" + ) + assert api_base == "https://example.test/v1" + assert api_key == "explicit" + + +def test_get_llm_provider_resolves_prefix(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + model, provider, key, api_base = litellm.get_llm_provider(model="clf_ai_gateway/glm-5.3") + assert model == "glm-5.3" + assert provider == "clf_ai_gateway" + assert key == "sk-gw-test" + assert api_base == "https://api.clfaigateway.dev/v1" + + +def test_get_llm_provider_infers_from_api_base(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + _, provider, _, _ = litellm.get_llm_provider(model="glm-5.3", api_base="https://api.clfaigateway.dev/v1") + assert provider == "clf_ai_gateway" + + _, provider, key, _ = litellm.get_llm_provider( + model="glm-5.3", api_base="https://api.clfaigateway.dev/v1", api_key="explicit" + ) + assert provider == "clf_ai_gateway" + assert key == "explicit" + + +def test_supported_params_match_gateway_surface() -> None: + params = ClfAiGatewayConfig().get_supported_openai_params(model="glm-5.3") + assert "reasoning_effort" in params + assert "tools" in params and "tool_choice" in params + # the gateway rejects these legacy/unknown fields with a 400 — never advertise them + for legacy in ("functions", "function_call", "logit_bias"): + assert legacy not in params + + +@pytest.mark.parametrize("model", MODELS) +def test_cost_map_entries(model: str) -> None: + key = f"clf_ai_gateway/{model}" + assert key in litellm.model_cost, f"missing cost map entry for {key}" + entry = litellm.model_cost[key] + assert entry["litellm_provider"] == "clf_ai_gateway" + assert entry["mode"] == "chat" + assert entry["max_output_tokens"] == 131072 + assert entry["input_cost_per_token"] > 0 + assert entry["output_cost_per_token"] > 0 + assert entry["supports_function_calling"] is True + assert entry["supports_reasoning"] is True + + +def test_vision_flags_match_measured_surface() -> None: + vision = {"glm-5.3-flash", "kimi-k2.6", "kimi-k2.7-code", "qwen3.8-27b"} + for model in MODELS: + entry = litellm.model_cost[f"clf_ai_gateway/{model}"] + assert bool(entry.get("supports_vision")) is (model in vision), model + + +def test_validate_environment_reports_env_key(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv("CLF_AI_GATEWAY_API_KEY", raising=False) + missing = litellm.validate_environment(model="clf_ai_gateway/glm-5.3") + assert missing["keys_in_environment"] is False + assert "CLF_AI_GATEWAY_API_KEY" in missing["missing_keys"] + + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + present = litellm.validate_environment(model="clf_ai_gateway/glm-5.3") + assert present["keys_in_environment"] is True + assert present["missing_keys"] == [] + + +def test_get_supported_openai_params_dispatches_to_provider_config() -> None: + params = litellm.get_supported_openai_params(model="glm-5.3", custom_llm_provider="clf_ai_gateway") + assert params is not None + assert "reasoning_effort" in params + assert "logit_bias" not in params + + +def test_completion_routes_without_network(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test") + response = litellm.completion( + model="clf_ai_gateway/glm-5.3", + messages=[{"role": "user", "content": "hi"}], + mock_response="hello from mock", + ) + assert response.choices[0].message.content == "hello from mock" From da8494e2f06f441e4a32ae93bc3ebb940aef2201 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:03:27 -0700 Subject: [PATCH 2/8] fix(clf_ai_gateway): declare each model's real reasoning_effort levels The per-level supports_*_reasoning_effort flags cannot express the sets this gateway actually takes, so all nine models advertised a wrong one: every model offered "minimal" and eight offered "none", neither of which the gateway accepts, five dropped "xhigh", which it does accept, and qwen3.8-27b offered "high", which it rejects. That set surfaces through /model_group/info and the model picker, so a user picking a level off it gets a 400 from the gateway. Declare reasoning_effort_levels instead, taken from the gateway's own GET /v1/public/models, which resolve_supported_reasoning_efforts reads first and uses whole. Declaring it also stops clf_ai_gateway/deepseek-v4-pro and clf_ai_gateway/deepseek-v4-flash inheriting effort metadata from the unprefixed deepseek entries of the same name. --- ...odel_prices_and_context_window_backup.json | 61 ++++++++++++++++--- model_prices_and_context_window.json | 61 ++++++++++++++++--- ...test_clf_ai_gateway_chat_transformation.py | 20 ++++++ 3 files changed, 122 insertions(+), 20 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 4a694e0ad14..a678359210f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -78787,8 +78787,13 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, - "supports_max_reasoning_effort": true, + "reasoning_effort_levels": [ + "none", + "low", + "medium", + "high", + "max" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78807,7 +78812,12 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78827,7 +78837,12 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78845,7 +78860,11 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high" + ], "supports_response_schema": true, "supports_system_messages": true, "source": "https://clfaigateway.dev/models" @@ -78863,7 +78882,12 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78882,7 +78906,12 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78901,7 +78930,11 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78921,7 +78954,11 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78941,7 +78978,11 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 4a694e0ad14..a678359210f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -78787,8 +78787,13 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, - "supports_max_reasoning_effort": true, + "reasoning_effort_levels": [ + "none", + "low", + "medium", + "high", + "max" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78807,7 +78812,12 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78827,7 +78837,12 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78845,7 +78860,11 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high" + ], "supports_response_schema": true, "supports_system_messages": true, "source": "https://clfaigateway.dev/models" @@ -78863,7 +78882,12 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78882,7 +78906,12 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78901,7 +78930,11 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78921,7 +78954,11 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "high" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, @@ -78941,7 +78978,11 @@ "supports_parallel_function_calling": true, "supports_tool_choice": true, "supports_reasoning": true, - "supports_low_reasoning_effort": true, + "reasoning_effort_levels": [ + "low", + "medium", + "xhigh" + ], "supports_response_schema": true, "supports_system_messages": true, "supports_prompt_caching": true, diff --git a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py index 587103e1077..7cec1d65f21 100644 --- a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py +++ b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py @@ -2,6 +2,7 @@ import pytest import litellm from litellm.llms.clf_ai_gateway.chat.transformation import ClfAiGatewayConfig +from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts from litellm.types.utils import LlmProviders MODELS = [ @@ -16,6 +17,18 @@ MODELS = [ "qwen3.8-27b", ] +GATEWAY_REASONING_EFFORTS = { + "glm-5.3": ("none", "low", "medium", "high", "max"), + "glm-5.3-flash": ("low", "medium", "high", "xhigh"), + "glm-5.2": ("low", "medium", "high", "xhigh"), + "glm-4.7-flash": ("low", "medium", "high"), + "deepseek-v4-pro": ("low", "medium", "high", "xhigh"), + "deepseek-v4-flash": ("low", "medium", "high", "xhigh"), + "kimi-k2.6": ("low", "medium", "high"), + "kimi-k2.7-code": ("low", "medium", "high"), + "qwen3.8-27b": ("low", "medium", "xhigh"), +} + @pytest.fixture(autouse=True) def _local_cost_map(monkeypatch: pytest.MonkeyPatch) -> None: @@ -126,3 +139,10 @@ def test_completion_routes_without_network(monkeypatch: pytest.MonkeyPatch) -> N mock_response="hello from mock", ) assert response.choices[0].message.content == "hello from mock" + + +@pytest.mark.parametrize("model", MODELS) +def test_advertised_reasoning_efforts_match_the_gateway(model: str) -> None: + key = f"clf_ai_gateway/{model}" + entry = {**litellm.model_cost[key], "key": key} + assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == GATEWAY_REASONING_EFFORTS[model] From 7acf816586ea978f0fc424298cd02fee37868cc3 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:43:41 -0700 Subject: [PATCH 3/8] test(clf_ai_gateway): name the gateway-rejected params instead of commenting them --- .../test_clf_ai_gateway_chat_transformation.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py index 7cec1d65f21..6415d77e13d 100644 --- a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py +++ b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py @@ -86,9 +86,8 @@ def test_supported_params_match_gateway_surface() -> None: params = ClfAiGatewayConfig().get_supported_openai_params(model="glm-5.3") assert "reasoning_effort" in params assert "tools" in params and "tool_choice" in params - # the gateway rejects these legacy/unknown fields with a 400 — never advertise them - for legacy in ("functions", "function_call", "logit_bias"): - assert legacy not in params + for rejected_by_gateway_with_400 in ("functions", "function_call", "logit_bias"): + assert rejected_by_gateway_with_400 not in params @pytest.mark.parametrize("model", MODELS) From e009e4ffea5babae2bf9b03ab1a408d1c91d9d01 Mon Sep 17 00:00:00 2001 From: bap1106 Date: Thu, 17 Sep 2026 03:44:27 +0700 Subject: [PATCH 4/8] fix(clf_ai_gateway): list the provider in the proxy Add Model form main now requires every LlmProviders value to have an entry in provider_create_fields.json (or be frozen as unlisted), so the Add Model dropdown cannot silently miss a backend provider. Add the clf_ai_gateway entry in the same shape as the SCX.ai registration: required API key, optional API base with https://api.clfaigateway.dev/v1 as placeholder, and a cost-map model as the default model placeholder. A provider test pins the entry. Co-Authored-By: Claude Opus 5 --- .../provider_create_fields.json | 28 +++++++++++++++++++ ...test_clf_ai_gateway_chat_transformation.py | 21 ++++++++++++++ 2 files changed, 49 insertions(+) diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index 67a8c356a4a..ddb5fd2d925 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -829,6 +829,34 @@ ], "default_model_placeholder": "gpt-3.5-turbo" }, + { + "provider": "CLF_AI_GATEWAY", + "provider_display_name": "CLF AI Gateway", + "litellm_provider": "clf_ai_gateway", + "credential_fields": [ + { + "key": "api_base", + "label": "API Base", + "placeholder": "https://api.clfaigateway.dev/v1", + "tooltip": null, + "required": false, + "field_type": "text", + "options": null, + "default_value": null + }, + { + "key": "api_key", + "label": "API Key", + "placeholder": null, + "tooltip": null, + "required": true, + "field_type": "password", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "clf_ai_gateway/glm-5.3" + }, { "provider": "CLOUDFLARE", "provider_display_name": "Cloudflare", diff --git a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py index 6415d77e13d..c43a938ec75 100644 --- a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py +++ b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py @@ -1,3 +1,6 @@ +import json +from pathlib import Path + import pytest import litellm @@ -145,3 +148,21 @@ def test_advertised_reasoning_efforts_match_the_gateway(model: str) -> None: key = f"clf_ai_gateway/{model}" entry = {**litellm.model_cost[key], "key": key} assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == GATEWAY_REASONING_EFFORTS[model] + + +def test_selectable_in_the_proxy_add_model_form() -> None: + fields_path = Path(litellm.__file__).parent / "proxy" / "public_endpoints" / "provider_create_fields.json" + provider_fields = json.loads(fields_path.read_text(encoding="utf-8")) + entries = [entry for entry in provider_fields if entry["litellm_provider"] == "clf_ai_gateway"] + assert len(entries) == 1 + + entry = entries[0] + assert entry["provider"] == "CLF_AI_GATEWAY" + assert entry["provider_display_name"] == "CLF AI Gateway" + assert entry["default_model_placeholder"] in litellm.model_cost + + fields = {field["key"]: field for field in entry["credential_fields"]} + assert fields["api_key"]["required"] is True + assert fields["api_key"]["field_type"] == "password" + assert fields["api_base"]["required"] is False + assert fields["api_base"]["placeholder"] == "https://api.clfaigateway.dev/v1" From a9ac442857a5ce21fb0980e0db9f12b7a0b2a49d Mon Sep 17 00:00:00 2001 From: bap1106 Date: Mon, 28 Sep 2026 15:29:19 +0700 Subject: [PATCH 5/8] fix(clf_ai_gateway): sync prices and reasoning_effort levels with the live catalog CLF AI Gateway raised its launch discount from 40% to 60% off list on 2026-09-26. It also re-measured every model's reasoning_effort levels against the current Workers AI behaviour. - input/output/cached per-token prices for all 9 models now match GET https://api.clfaigateway.dev/v1/public/models to the nano; qwen3.8-27b cached input also follows the upstream cached rate published on 2026-09-25 - reasoning_effort_levels: add "max" to deepseek-v4-flash, deepseek-v4-pro, glm-5.2 and glm-5.3-flash; add "none" to kimi-k2.6 (it really disables thinking there); drop "none" from glm-5.3 (upstream runs it as "max") Context windows, output limits and vision flags are unchanged and were checked against the same endpoint. Co-Authored-By: Claude Opus 5.5 --- ...odel_prices_and_context_window_backup.json | 66 ++++++++++--------- model_prices_and_context_window.json | 66 ++++++++++--------- 2 files changed, 70 insertions(+), 62 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index a678359210f..fcd2726b6dc 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -78778,9 +78778,9 @@ "max_tokens": 131072, "max_input_tokens": 1048576, "max_output_tokens": 131072, - "input_cost_per_token": 8.4e-07, - "output_cost_per_token": 2.64e-06, - "cache_read_input_token_cost": 1.56e-07, + "input_cost_per_token": 5.6e-07, + "output_cost_per_token": 1.76e-06, + "cache_read_input_token_cost": 1.04e-07, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78788,7 +78788,6 @@ "supports_tool_choice": true, "supports_reasoning": true, "reasoning_effort_levels": [ - "none", "low", "medium", "high", @@ -78803,9 +78802,9 @@ "max_tokens": 131072, "max_input_tokens": 1048576, "max_output_tokens": 131072, - "input_cost_per_token": 9e-08, - "output_cost_per_token": 3e-07, - "cache_read_input_token_cost": 1.8e-08, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 2e-07, + "cache_read_input_token_cost": 1.2e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78816,7 +78815,8 @@ "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "supports_response_schema": true, "supports_system_messages": true, @@ -78828,9 +78828,9 @@ "max_tokens": 131072, "max_input_tokens": 262144, "max_output_tokens": 131072, - "input_cost_per_token": 8.4e-07, - "output_cost_per_token": 2.64e-06, - "cache_read_input_token_cost": 1.56e-07, + "input_cost_per_token": 5.6e-07, + "output_cost_per_token": 1.76e-06, + "cache_read_input_token_cost": 1.04e-07, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78841,7 +78841,8 @@ "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "supports_response_schema": true, "supports_system_messages": true, @@ -78852,8 +78853,8 @@ "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 3.6e-08, - "output_cost_per_token": 2.4e-07, + "input_cost_per_token": 2.4e-08, + "output_cost_per_token": 1.6e-07, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78873,9 +78874,9 @@ "max_tokens": 131072, "max_input_tokens": 1048576, "max_output_tokens": 131072, - "input_cost_per_token": 7.92e-07, - "output_cost_per_token": 2.376e-06, - "cache_read_input_token_cost": 2.6e-08, + "input_cost_per_token": 5.28e-07, + "output_cost_per_token": 1.584e-06, + "cache_read_input_token_cost": 1.7e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78886,7 +78887,8 @@ "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "supports_response_schema": true, "supports_system_messages": true, @@ -78897,9 +78899,9 @@ "max_tokens": 131072, "max_input_tokens": 1048576, "max_output_tokens": 131072, - "input_cost_per_token": 2.64e-07, - "output_cost_per_token": 7.92e-07, - "cache_read_input_token_cost": 8e-09, + "input_cost_per_token": 1.76e-07, + "output_cost_per_token": 5.28e-07, + "cache_read_input_token_cost": 5e-09, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78910,7 +78912,8 @@ "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "supports_response_schema": true, "supports_system_messages": true, @@ -78921,9 +78924,9 @@ "max_tokens": 131072, "max_input_tokens": 262144, "max_output_tokens": 131072, - "input_cost_per_token": 5.7e-07, - "output_cost_per_token": 2.4e-06, - "cache_read_input_token_cost": 9.6e-08, + "input_cost_per_token": 3.8e-07, + "output_cost_per_token": 1.6e-06, + "cache_read_input_token_cost": 6.4e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78931,6 +78934,7 @@ "supports_tool_choice": true, "supports_reasoning": true, "reasoning_effort_levels": [ + "none", "low", "medium", "high" @@ -78945,9 +78949,9 @@ "max_tokens": 131072, "max_input_tokens": 262144, "max_output_tokens": 131072, - "input_cost_per_token": 5.7e-07, - "output_cost_per_token": 2.4e-06, - "cache_read_input_token_cost": 1.14e-07, + "input_cost_per_token": 3.8e-07, + "output_cost_per_token": 1.6e-06, + "cache_read_input_token_cost": 7.6e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78969,9 +78973,9 @@ "max_tokens": 131072, "max_input_tokens": 262144, "max_output_tokens": 131072, - "input_cost_per_token": 2.7e-07, - "output_cost_per_token": 1.92e-06, - "cache_read_input_token_cost": 2.7e-07, + "input_cost_per_token": 1.8e-07, + "output_cost_per_token": 1.28e-06, + "cache_read_input_token_cost": 2e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a678359210f..fcd2726b6dc 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -78778,9 +78778,9 @@ "max_tokens": 131072, "max_input_tokens": 1048576, "max_output_tokens": 131072, - "input_cost_per_token": 8.4e-07, - "output_cost_per_token": 2.64e-06, - "cache_read_input_token_cost": 1.56e-07, + "input_cost_per_token": 5.6e-07, + "output_cost_per_token": 1.76e-06, + "cache_read_input_token_cost": 1.04e-07, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78788,7 +78788,6 @@ "supports_tool_choice": true, "supports_reasoning": true, "reasoning_effort_levels": [ - "none", "low", "medium", "high", @@ -78803,9 +78802,9 @@ "max_tokens": 131072, "max_input_tokens": 1048576, "max_output_tokens": 131072, - "input_cost_per_token": 9e-08, - "output_cost_per_token": 3e-07, - "cache_read_input_token_cost": 1.8e-08, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 2e-07, + "cache_read_input_token_cost": 1.2e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78816,7 +78815,8 @@ "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "supports_response_schema": true, "supports_system_messages": true, @@ -78828,9 +78828,9 @@ "max_tokens": 131072, "max_input_tokens": 262144, "max_output_tokens": 131072, - "input_cost_per_token": 8.4e-07, - "output_cost_per_token": 2.64e-06, - "cache_read_input_token_cost": 1.56e-07, + "input_cost_per_token": 5.6e-07, + "output_cost_per_token": 1.76e-06, + "cache_read_input_token_cost": 1.04e-07, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78841,7 +78841,8 @@ "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "supports_response_schema": true, "supports_system_messages": true, @@ -78852,8 +78853,8 @@ "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 3.6e-08, - "output_cost_per_token": 2.4e-07, + "input_cost_per_token": 2.4e-08, + "output_cost_per_token": 1.6e-07, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78873,9 +78874,9 @@ "max_tokens": 131072, "max_input_tokens": 1048576, "max_output_tokens": 131072, - "input_cost_per_token": 7.92e-07, - "output_cost_per_token": 2.376e-06, - "cache_read_input_token_cost": 2.6e-08, + "input_cost_per_token": 5.28e-07, + "output_cost_per_token": 1.584e-06, + "cache_read_input_token_cost": 1.7e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78886,7 +78887,8 @@ "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "supports_response_schema": true, "supports_system_messages": true, @@ -78897,9 +78899,9 @@ "max_tokens": 131072, "max_input_tokens": 1048576, "max_output_tokens": 131072, - "input_cost_per_token": 2.64e-07, - "output_cost_per_token": 7.92e-07, - "cache_read_input_token_cost": 8e-09, + "input_cost_per_token": 1.76e-07, + "output_cost_per_token": 5.28e-07, + "cache_read_input_token_cost": 5e-09, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78910,7 +78912,8 @@ "low", "medium", "high", - "xhigh" + "xhigh", + "max" ], "supports_response_schema": true, "supports_system_messages": true, @@ -78921,9 +78924,9 @@ "max_tokens": 131072, "max_input_tokens": 262144, "max_output_tokens": 131072, - "input_cost_per_token": 5.7e-07, - "output_cost_per_token": 2.4e-06, - "cache_read_input_token_cost": 9.6e-08, + "input_cost_per_token": 3.8e-07, + "output_cost_per_token": 1.6e-06, + "cache_read_input_token_cost": 6.4e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78931,6 +78934,7 @@ "supports_tool_choice": true, "supports_reasoning": true, "reasoning_effort_levels": [ + "none", "low", "medium", "high" @@ -78945,9 +78949,9 @@ "max_tokens": 131072, "max_input_tokens": 262144, "max_output_tokens": 131072, - "input_cost_per_token": 5.7e-07, - "output_cost_per_token": 2.4e-06, - "cache_read_input_token_cost": 1.14e-07, + "input_cost_per_token": 3.8e-07, + "output_cost_per_token": 1.6e-06, + "cache_read_input_token_cost": 7.6e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, @@ -78969,9 +78973,9 @@ "max_tokens": 131072, "max_input_tokens": 262144, "max_output_tokens": 131072, - "input_cost_per_token": 2.7e-07, - "output_cost_per_token": 1.92e-06, - "cache_read_input_token_cost": 2.7e-07, + "input_cost_per_token": 1.8e-07, + "output_cost_per_token": 1.28e-06, + "cache_read_input_token_cost": 2e-08, "litellm_provider": "clf_ai_gateway", "mode": "chat", "supports_function_calling": true, From e2b7e216ec0c5811c3467e5d877f08b159042811 Mon Sep 17 00:00:00 2001 From: bap1106 Date: Mon, 28 Sep 2026 15:48:47 +0700 Subject: [PATCH 6/8] fix(clf_ai_gateway): drop a rebind-ok that no longer suppresses anything After rebasing onto main, the type-discipline gate's new LIT013 rule flags the `# rebind-ok` on the endpoint-inference `dynamic_api_key` assignment. The rebind rule no longer fires on that line, so the suppression is inert. The sibling endpoint branches carry no comment there either. Co-Authored-By: Claude Opus 5.5 --- litellm/litellm_core_utils/get_llm_provider_logic.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 7391399ea54..2cbeefc5d18 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -272,7 +272,7 @@ def get_llm_provider( elif endpoint == "api.clfaigateway.dev/v1": custom_llm_provider = "clf_ai_gateway" # rebind-ok: endpoint inference, like every branch clf_env_key = get_secret_str("CLF_AI_GATEWAY_API_KEY") - dynamic_api_key = api_key or clf_env_key # rebind-ok: caller key wins, like every branch + dynamic_api_key = api_key or clf_env_key elif endpoint == "api.mistral.ai/v1": custom_llm_provider = "mistral" dynamic_api_key = get_secret_str("MISTRAL_API_KEY") From 4dbe4aa51691e64f836feeec957a1698c5c4cdf9 Mon Sep 17 00:00:00 2001 From: bap1106 Date: Mon, 28 Sep 2026 15:51:19 +0700 Subject: [PATCH 7/8] test(clf_ai_gateway): expect the re-measured reasoning_effort levels Follows the catalog sync: max on deepseek-v4-flash, deepseek-v4-pro, glm-5.2 and glm-5.3-flash, none on kimi-k2.6, and no none on glm-5.3. Co-Authored-By: Claude Opus 5.5 --- .../test_clf_ai_gateway_chat_transformation.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py index c43a938ec75..c1818352a3e 100644 --- a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py +++ b/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py @@ -21,13 +21,13 @@ MODELS = [ ] GATEWAY_REASONING_EFFORTS = { - "glm-5.3": ("none", "low", "medium", "high", "max"), - "glm-5.3-flash": ("low", "medium", "high", "xhigh"), - "glm-5.2": ("low", "medium", "high", "xhigh"), + "glm-5.3": ("low", "medium", "high", "max"), + "glm-5.3-flash": ("low", "medium", "high", "xhigh", "max"), + "glm-5.2": ("low", "medium", "high", "xhigh", "max"), "glm-4.7-flash": ("low", "medium", "high"), - "deepseek-v4-pro": ("low", "medium", "high", "xhigh"), - "deepseek-v4-flash": ("low", "medium", "high", "xhigh"), - "kimi-k2.6": ("low", "medium", "high"), + "deepseek-v4-pro": ("low", "medium", "high", "xhigh", "max"), + "deepseek-v4-flash": ("low", "medium", "high", "xhigh", "max"), + "kimi-k2.6": ("none", "low", "medium", "high"), "kimi-k2.7-code": ("low", "medium", "high"), "qwen3.8-27b": ("low", "medium", "xhigh"), } From e1b4d744d722b7775aa339aaebd6a84e8f3ad129 Mon Sep 17 00:00:00 2001 From: bap1106 Date: Tue, 29 Sep 2026 22:45:04 +0700 Subject: [PATCH 8/8] test(clf_ai_gateway): move the provider tests to tests/unit/llms main moved every per-provider test from tests/test_litellm/llms// to tests/unit/llms// (DeepInfra lives at tests/unit/llms/deepinfra/). This branch still added the old path, so no CI shard claimed it. That failed assert-ci-coverage and assert-shard-coverage, and the file never ran (hence the low codecov/patch). Co-Authored-By: Claude Opus 5.5 --- tests/unit/llms/clf_ai_gateway/__init__.py | 0 .../clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py | 0 2 files changed, 0 insertions(+), 0 deletions(-) create mode 100644 tests/unit/llms/clf_ai_gateway/__init__.py rename tests/{test_litellm => unit}/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py (100%) diff --git a/tests/unit/llms/clf_ai_gateway/__init__.py b/tests/unit/llms/clf_ai_gateway/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py b/tests/unit/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py similarity index 100% rename from tests/test_litellm/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py rename to tests/unit/llms/clf_ai_gateway/test_clf_ai_gateway_chat_transformation.py