mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
feat(providers): add CLF AI Gateway (clf_ai_gateway), OpenAI-compatible, 9 models
OpenAI-compatible gateway at https://api.clfaigateway.dev/v1 serving open-weight models (GLM, Kimi, DeepSeek, Qwen) on Cloudflare Workers AI upstream. Follows the DeepInfra pattern: ClfAiGatewayConfig(OpenAIGPTConfig), CLF_AI_GATEWAY_API_KEY / _API_BASE, provider registered in enum, provider lists, endpoint inference, lazy-import registry, ProviderConfigManager, validate_environment. Supported params deliberately omit functions/function_call/logit_bias (gateway returns 400 for them). 9 cost-map entries (root + backup) with measured context/output limits and currently billed prices; endpoint-support entry; README row. Mocked tests only. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
684a1edd44
commit
935e8178ab
14 changed files with 612 additions and 0 deletions
|
|
@ -292,6 +292,7 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse
|
|||
| [Bytez (`bytez`)](https://docs.litellm.ai/docs/providers/bytez) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Cerebras (`cerebras`)](https://docs.litellm.ai/docs/providers/cerebras) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Clarifai (`clarifai`)](https://docs.litellm.ai/docs/providers/clarifai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [CLF AI Gateway (`clf_ai_gateway`)](https://docs.litellm.ai/docs/providers/clf_ai_gateway) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Cloudflare AI Workers (`cloudflare`)](https://docs.litellm.ai/docs/providers/cloudflare_workers) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Codestral (`codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Cognition (`cognition`)](https://docs.litellm.ai/docs/providers/cognition) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
|
|
|
|||
|
|
@ -637,6 +637,7 @@ fal_ai_models: Set = set()
|
|||
fireworks_ai_models: Set = set()
|
||||
fireworks_ai_embedding_models: Set = set()
|
||||
deepinfra_models: Set = set()
|
||||
clf_ai_gateway_models: set[str] = set() # mutable-ok: filled at import like sibling sets
|
||||
perplexity_models: Set = set()
|
||||
watsonx_models: Set = set()
|
||||
gemini_models: Set = set()
|
||||
|
|
@ -839,6 +840,8 @@ def _populate_provider_model_sets(model_cost_map: Dict) -> None:
|
|||
bedrock_converse_models.add(key)
|
||||
elif value.get("litellm_provider") == "deepinfra":
|
||||
deepinfra_models.add(key)
|
||||
elif value.get("litellm_provider") == "clf_ai_gateway": # pyright: ignore[reportUnknownMemberType] # raw cost map
|
||||
clf_ai_gateway_models.add(key) # pyright: ignore[reportUnknownArgumentType] # raw cost map key
|
||||
elif value.get("litellm_provider") == "perplexity":
|
||||
perplexity_models.add(key)
|
||||
elif value.get("litellm_provider") == "watsonx":
|
||||
|
|
@ -1065,6 +1068,7 @@ model_list = list(
|
|||
| set(ollama_models)
|
||||
| bedrock_models
|
||||
| deepinfra_models
|
||||
| clf_ai_gateway_models
|
||||
| perplexity_models
|
||||
| set(maritalk_models)
|
||||
| runwayml_models
|
||||
|
|
@ -1170,6 +1174,7 @@ def _build_models_by_provider() -> dict:
|
|||
"ollama": ollama_models,
|
||||
"ollama_chat": ollama_models,
|
||||
"deepinfra": deepinfra_models,
|
||||
"clf_ai_gateway": clf_ai_gateway_models,
|
||||
"perplexity": perplexity_models,
|
||||
"maritalk": maritalk_models,
|
||||
"watsonx": watsonx_models,
|
||||
|
|
@ -1965,6 +1970,9 @@ if TYPE_CHECKING:
|
|||
LiteLLMProxyChatConfig as _LiteLLMProxyChatConfig,
|
||||
)
|
||||
from .llms.deepinfra.chat.transformation import DeepInfraConfig as _DeepInfraConfig
|
||||
from .llms.clf_ai_gateway.chat.transformation import (
|
||||
ClfAiGatewayConfig as _ClfAiGatewayConfig,
|
||||
)
|
||||
from .llms.llamafile.chat.transformation import (
|
||||
LlamafileChatConfig as _LlamafileChatConfig,
|
||||
)
|
||||
|
|
@ -1994,6 +2002,7 @@ if TYPE_CHECKING:
|
|||
IBMWatsonXAIConfig: Type[_IBMWatsonXAIConfig]
|
||||
LiteLLMProxyChatConfig: Type[_LiteLLMProxyChatConfig]
|
||||
DeepInfraConfig: Type[_DeepInfraConfig]
|
||||
ClfAiGatewayConfig: type[_ClfAiGatewayConfig]
|
||||
LlamafileChatConfig: Type[_LlamafileChatConfig]
|
||||
LMStudioChatConfig: Type[_LMStudioChatConfig]
|
||||
LmStudioEmbeddingConfig: Type[_LmStudioEmbeddingConfig]
|
||||
|
|
|
|||
|
|
@ -219,6 +219,7 @@ LLM_CONFIG_NAMES: Final = (
|
|||
"MistralEmbeddingConfig",
|
||||
"OpenAIImageVariationConfig",
|
||||
"DeepInfraConfig",
|
||||
"ClfAiGatewayConfig",
|
||||
"DeepgramAudioTranscriptionConfig",
|
||||
"TopazImageVariationConfig",
|
||||
"OpenAITextCompletionConfig",
|
||||
|
|
@ -908,6 +909,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
|
|||
"OpenAIImageVariationConfig",
|
||||
),
|
||||
"DeepInfraConfig": (".llms.deepinfra.chat.transformation", "DeepInfraConfig"),
|
||||
"ClfAiGatewayConfig": (
|
||||
".llms.clf_ai_gateway.chat.transformation",
|
||||
"ClfAiGatewayConfig",
|
||||
),
|
||||
"DeepgramAudioTranscriptionConfig": (
|
||||
".llms.deepgram.audio_transcription.transformation",
|
||||
"DeepgramAudioTranscriptionConfig",
|
||||
|
|
|
|||
|
|
@ -708,6 +708,7 @@ LITELLM_CHAT_PROVIDERS: Final = [
|
|||
"ollama",
|
||||
"ollama_chat",
|
||||
"deepinfra",
|
||||
"clf_ai_gateway",
|
||||
"perplexity",
|
||||
"mistral",
|
||||
"groq",
|
||||
|
|
@ -903,6 +904,7 @@ openai_compatible_endpoints: Final[list] = [
|
|||
"api.perplexity.ai",
|
||||
"api.endpoints.anyscale.com/v1",
|
||||
"api.deepinfra.com/v1/openai",
|
||||
"api.clfaigateway.dev/v1",
|
||||
"api.mistral.ai/v1",
|
||||
"codestral.mistral.ai/v1/chat/completions",
|
||||
"codestral.mistral.ai/v1/fim/completions",
|
||||
|
|
@ -968,6 +970,7 @@ openai_compatible_providers: Final[list] = [
|
|||
"deepseek",
|
||||
"tencent",
|
||||
"deepinfra",
|
||||
"clf_ai_gateway",
|
||||
"perplexity",
|
||||
"xinference",
|
||||
"xai",
|
||||
|
|
|
|||
|
|
@ -269,6 +269,10 @@ def get_llm_provider(
|
|||
elif endpoint == "api.deepinfra.com/v1/openai":
|
||||
custom_llm_provider = "deepinfra"
|
||||
dynamic_api_key = get_secret_str("DEEPINFRA_API_KEY")
|
||||
elif endpoint == "api.clfaigateway.dev/v1":
|
||||
custom_llm_provider = "clf_ai_gateway" # rebind-ok: endpoint inference, like every branch
|
||||
clf_env_key = get_secret_str("CLF_AI_GATEWAY_API_KEY")
|
||||
dynamic_api_key = api_key or clf_env_key # rebind-ok: caller key wins, like every branch
|
||||
elif endpoint == "api.mistral.ai/v1":
|
||||
custom_llm_provider = "mistral"
|
||||
dynamic_api_key = get_secret_str("MISTRAL_API_KEY")
|
||||
|
|
@ -622,6 +626,12 @@ def _get_openai_compatible_provider_info(
|
|||
api_base,
|
||||
dynamic_api_key,
|
||||
) = litellm.DeepInfraConfig()._get_openai_compatible_provider_info(api_base, api_key)
|
||||
elif custom_llm_provider == "clf_ai_gateway":
|
||||
cfg = litellm.ClfAiGatewayConfig()
|
||||
(
|
||||
api_base, # rebind-ok: resolved base/key are what this function returns, like every branch
|
||||
dynamic_api_key, # rebind-ok: see above
|
||||
) = cfg._get_openai_compatible_provider_info(api_base, api_key) # pyright: ignore[reportPrivateUsage] # hook
|
||||
elif custom_llm_provider == "empower":
|
||||
api_base = api_base or get_secret("EMPOWER_API_BASE") or "https://app.empower.dev/api/v1"
|
||||
dynamic_api_key = api_key or get_secret_str("EMPOWER_API_KEY")
|
||||
|
|
|
|||
|
|
@ -222,6 +222,8 @@ def get_supported_openai_params(
|
|||
return litellm.PetalsConfig().get_supported_openai_params(model=model)
|
||||
elif custom_llm_provider == "deepinfra":
|
||||
return litellm.DeepInfraConfig().get_supported_openai_params(model=model)
|
||||
elif custom_llm_provider == "clf_ai_gateway":
|
||||
return litellm.ClfAiGatewayConfig().get_supported_openai_params(model=model)
|
||||
elif custom_llm_provider == "perplexity":
|
||||
return litellm.PerplexityChatConfig().get_supported_openai_params(model=model)
|
||||
elif custom_llm_provider == "nscale":
|
||||
|
|
|
|||
63
litellm/llms/clf_ai_gateway/chat/transformation.py
Normal file
63
litellm/llms/clf_ai_gateway/chat/transformation.py
Normal file
|
|
@ -0,0 +1,63 @@
|
|||
from typing import Final
|
||||
|
||||
import litellm
|
||||
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
|
||||
# Narrower than the generic OpenAI surface on purpose: the gateway rejects legacy
|
||||
# top-level fields (``functions``, ``function_call``, ``logit_bias``) with a 400.
|
||||
SUPPORTED_OPENAI_PARAMS: Final = (
|
||||
"stream",
|
||||
"stream_options",
|
||||
"frequency_penalty",
|
||||
"presence_penalty",
|
||||
"max_tokens",
|
||||
"max_completion_tokens",
|
||||
"n",
|
||||
"stop",
|
||||
"temperature",
|
||||
"top_p",
|
||||
"seed",
|
||||
"response_format",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"parallel_tool_calls",
|
||||
"user",
|
||||
)
|
||||
|
||||
|
||||
class ClfAiGatewayConfig(OpenAIGPTConfig):
|
||||
"""
|
||||
Reference: https://clfaigateway.dev/docs
|
||||
|
||||
CLF AI Gateway is an OpenAI-compatible gateway (chat completions, streaming, tool
|
||||
calling, structured output, ``reasoning_effort``) serving open-weight models
|
||||
(GLM, Kimi, DeepSeek, Qwen) on Cloudflare Workers AI upstream. Model ids are the
|
||||
gateway's canonical names, e.g. ``clf_ai_gateway/glm-5.3``.
|
||||
|
||||
The gateway validates request bodies strictly and rejects unknown or legacy
|
||||
top-level fields with a 400 (``functions``, ``function_call``, ``logit_bias``),
|
||||
so the supported-params list below is deliberately narrower than the generic
|
||||
OpenAI surface.
|
||||
"""
|
||||
|
||||
@property
|
||||
def custom_llm_provider(self) -> str | None:
|
||||
return "clf_ai_gateway"
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> list[str]: # mutable-ok: matches OpenAIGPTConfig's list return
|
||||
supports_reasoning: Final = litellm.supports_reasoning(
|
||||
model=model,
|
||||
custom_llm_provider=self.custom_llm_provider,
|
||||
)
|
||||
extra: Final = ("reasoning_effort",) if supports_reasoning else ()
|
||||
return [*SUPPORTED_OPENAI_PARAMS, *extra] # mutable-ok: one-shot build, list per base contract
|
||||
|
||||
def _get_openai_compatible_provider_info(
|
||||
self, api_base: str | None, api_key: str | None
|
||||
) -> tuple[str | None, str | None]:
|
||||
resolved_api_base: Final = (
|
||||
api_base or get_secret_str("CLF_AI_GATEWAY_API_BASE") or "https://api.clfaigateway.dev/v1"
|
||||
)
|
||||
dynamic_api_key: Final = api_key or get_secret_str("CLF_AI_GATEWAY_API_KEY")
|
||||
return resolved_api_base, dynamic_api_key
|
||||
|
|
@ -78773,5 +78773,179 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"clf_ai_gateway/glm-5.3": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 8.4e-07,
|
||||
"output_cost_per_token": 2.64e-06,
|
||||
"cache_read_input_token_cost": 1.56e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/glm-5.3-flash": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 9e-08,
|
||||
"output_cost_per_token": 3e-07,
|
||||
"cache_read_input_token_cost": 1.8e-08,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/glm-5.2": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 8.4e-07,
|
||||
"output_cost_per_token": 2.64e-06,
|
||||
"cache_read_input_token_cost": 1.56e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/glm-4.7-flash": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 3.6e-08,
|
||||
"output_cost_per_token": 2.4e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/deepseek-v4-pro": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 7.92e-07,
|
||||
"output_cost_per_token": 2.376e-06,
|
||||
"cache_read_input_token_cost": 2.6e-08,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/deepseek-v4-flash": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 2.64e-07,
|
||||
"output_cost_per_token": 7.92e-07,
|
||||
"cache_read_input_token_cost": 8e-09,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/kimi-k2.6": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 5.7e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"cache_read_input_token_cost": 9.6e-08,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/kimi-k2.7-code": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 5.7e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"cache_read_input_token_cost": 1.14e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/qwen3.8-27b": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 2.7e-07,
|
||||
"output_cost_per_token": 1.92e-06,
|
||||
"cache_read_input_token_cost": 2.7e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -492,6 +492,24 @@
|
|||
"interactions": true
|
||||
}
|
||||
},
|
||||
"clf_ai_gateway": {
|
||||
"display_name": "CLF AI Gateway (`clf_ai_gateway`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/clf_ai_gateway",
|
||||
"endpoints": {
|
||||
"chat_completions": true,
|
||||
"messages": true,
|
||||
"responses": true,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
"audio_transcriptions": false,
|
||||
"audio_speech": false,
|
||||
"moderations": false,
|
||||
"batches": false,
|
||||
"rerank": false,
|
||||
"a2a": true,
|
||||
"interactions": true
|
||||
}
|
||||
},
|
||||
"cloudflare": {
|
||||
"display_name": "Cloudflare AI Workers (`cloudflare`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/cloudflare_workers",
|
||||
|
|
|
|||
|
|
@ -4033,6 +4033,7 @@ class LlmProviders(str, Enum):
|
|||
OLLAMA = "ollama"
|
||||
OLLAMA_CHAT = "ollama_chat"
|
||||
DEEPINFRA = "deepinfra"
|
||||
CLF_AI_GATEWAY = "clf_ai_gateway"
|
||||
PERPLEXITY = "perplexity"
|
||||
MISTRAL = "mistral"
|
||||
MILVUS = "milvus"
|
||||
|
|
|
|||
|
|
@ -6787,6 +6787,11 @@ def validate_environment(
|
|||
keys_in_environment = True
|
||||
else:
|
||||
missing_keys.append("DEEPINFRA_API_KEY")
|
||||
elif custom_llm_provider == "clf_ai_gateway":
|
||||
if "CLF_AI_GATEWAY_API_KEY" in os.environ:
|
||||
keys_in_environment = True
|
||||
else:
|
||||
missing_keys.append("CLF_AI_GATEWAY_API_KEY")
|
||||
elif custom_llm_provider == "featherless_ai":
|
||||
if "FEATHERLESS_AI_API_KEY" in os.environ:
|
||||
keys_in_environment = True
|
||||
|
|
@ -8493,6 +8498,7 @@ class ProviderConfigManager:
|
|||
LlmProviders.OOBABOOGA: (lambda: litellm.OobaboogaConfig(), False),
|
||||
LlmProviders.OLLAMA_CHAT: (lambda: litellm.OllamaChatConfig(), False),
|
||||
LlmProviders.DEEPINFRA: (lambda: litellm.DeepInfraConfig(), False),
|
||||
LlmProviders.CLF_AI_GATEWAY: (lambda: litellm.ClfAiGatewayConfig(), False),
|
||||
LlmProviders.PERPLEXITY: (lambda: litellm.PerplexityChatConfig(), False),
|
||||
LlmProviders.MISTRAL: (lambda: litellm.MistralConfig(), False),
|
||||
LlmProviders.CODESTRAL: (lambda: litellm.MistralConfig(), False),
|
||||
|
|
|
|||
|
|
@ -78773,5 +78773,179 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"clf_ai_gateway/glm-5.3": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 8.4e-07,
|
||||
"output_cost_per_token": 2.64e-06,
|
||||
"cache_read_input_token_cost": 1.56e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/glm-5.3-flash": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 9e-08,
|
||||
"output_cost_per_token": 3e-07,
|
||||
"cache_read_input_token_cost": 1.8e-08,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/glm-5.2": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 8.4e-07,
|
||||
"output_cost_per_token": 2.64e-06,
|
||||
"cache_read_input_token_cost": 1.56e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/glm-4.7-flash": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 3.6e-08,
|
||||
"output_cost_per_token": 2.4e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/deepseek-v4-pro": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 7.92e-07,
|
||||
"output_cost_per_token": 2.376e-06,
|
||||
"cache_read_input_token_cost": 2.6e-08,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/deepseek-v4-flash": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 2.64e-07,
|
||||
"output_cost_per_token": 7.92e-07,
|
||||
"cache_read_input_token_cost": 8e-09,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/kimi-k2.6": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 5.7e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"cache_read_input_token_cost": 9.6e-08,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/kimi-k2.7-code": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 5.7e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"cache_read_input_token_cost": 1.14e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
},
|
||||
"clf_ai_gateway/qwen3.8-27b": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 2.7e-07,
|
||||
"output_cost_per_token": 1.92e-06,
|
||||
"cache_read_input_token_cost": 2.7e-07,
|
||||
"litellm_provider": "clf_ai_gateway",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_low_reasoning_effort": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://clfaigateway.dev/models"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -545,6 +545,24 @@
|
|||
"interactions": true
|
||||
}
|
||||
},
|
||||
"clf_ai_gateway": {
|
||||
"display_name": "CLF AI Gateway (`clf_ai_gateway`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/clf_ai_gateway",
|
||||
"endpoints": {
|
||||
"chat_completions": true,
|
||||
"messages": true,
|
||||
"responses": true,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
"audio_transcriptions": false,
|
||||
"audio_speech": false,
|
||||
"moderations": false,
|
||||
"batches": false,
|
||||
"rerank": false,
|
||||
"a2a": true,
|
||||
"interactions": true
|
||||
}
|
||||
},
|
||||
"cloudflare": {
|
||||
"display_name": "Cloudflare AI Workers (`cloudflare`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/cloudflare_workers",
|
||||
|
|
|
|||
|
|
@ -0,0 +1,128 @@
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.clf_ai_gateway.chat.transformation import ClfAiGatewayConfig
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
MODELS = [
|
||||
"glm-5.3",
|
||||
"glm-5.3-flash",
|
||||
"glm-5.2",
|
||||
"glm-4.7-flash",
|
||||
"deepseek-v4-pro",
|
||||
"deepseek-v4-flash",
|
||||
"kimi-k2.6",
|
||||
"kimi-k2.7-code",
|
||||
"qwen3.8-27b",
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _local_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
|
||||
def test_provider_is_registered() -> None:
|
||||
assert LlmProviders.CLF_AI_GATEWAY.value == "clf_ai_gateway"
|
||||
assert "clf_ai_gateway" in litellm.openai_compatible_providers
|
||||
assert "api.clfaigateway.dev/v1" in litellm.openai_compatible_endpoints
|
||||
assert "clf_ai_gateway" in litellm.provider_list
|
||||
assert isinstance(
|
||||
litellm.ProviderConfigManager.get_provider_chat_config(model="glm-5.3", provider=LlmProviders.CLF_AI_GATEWAY),
|
||||
ClfAiGatewayConfig,
|
||||
)
|
||||
|
||||
|
||||
def test_default_api_base_and_env_key(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test")
|
||||
monkeypatch.delenv("CLF_AI_GATEWAY_API_BASE", raising=False)
|
||||
api_base, api_key = ClfAiGatewayConfig()._get_openai_compatible_provider_info(api_base=None, api_key=None)
|
||||
assert api_base == "https://api.clfaigateway.dev/v1"
|
||||
assert api_key == "sk-gw-test"
|
||||
|
||||
api_base, api_key = ClfAiGatewayConfig()._get_openai_compatible_provider_info(
|
||||
api_base="https://example.test/v1", api_key="explicit"
|
||||
)
|
||||
assert api_base == "https://example.test/v1"
|
||||
assert api_key == "explicit"
|
||||
|
||||
|
||||
def test_get_llm_provider_resolves_prefix(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test")
|
||||
model, provider, key, api_base = litellm.get_llm_provider(model="clf_ai_gateway/glm-5.3")
|
||||
assert model == "glm-5.3"
|
||||
assert provider == "clf_ai_gateway"
|
||||
assert key == "sk-gw-test"
|
||||
assert api_base == "https://api.clfaigateway.dev/v1"
|
||||
|
||||
|
||||
def test_get_llm_provider_infers_from_api_base(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test")
|
||||
_, provider, _, _ = litellm.get_llm_provider(model="glm-5.3", api_base="https://api.clfaigateway.dev/v1")
|
||||
assert provider == "clf_ai_gateway"
|
||||
|
||||
_, provider, key, _ = litellm.get_llm_provider(
|
||||
model="glm-5.3", api_base="https://api.clfaigateway.dev/v1", api_key="explicit"
|
||||
)
|
||||
assert provider == "clf_ai_gateway"
|
||||
assert key == "explicit"
|
||||
|
||||
|
||||
def test_supported_params_match_gateway_surface() -> None:
|
||||
params = ClfAiGatewayConfig().get_supported_openai_params(model="glm-5.3")
|
||||
assert "reasoning_effort" in params
|
||||
assert "tools" in params and "tool_choice" in params
|
||||
# the gateway rejects these legacy/unknown fields with a 400 — never advertise them
|
||||
for legacy in ("functions", "function_call", "logit_bias"):
|
||||
assert legacy not in params
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
def test_cost_map_entries(model: str) -> None:
|
||||
key = f"clf_ai_gateway/{model}"
|
||||
assert key in litellm.model_cost, f"missing cost map entry for {key}"
|
||||
entry = litellm.model_cost[key]
|
||||
assert entry["litellm_provider"] == "clf_ai_gateway"
|
||||
assert entry["mode"] == "chat"
|
||||
assert entry["max_output_tokens"] == 131072
|
||||
assert entry["input_cost_per_token"] > 0
|
||||
assert entry["output_cost_per_token"] > 0
|
||||
assert entry["supports_function_calling"] is True
|
||||
assert entry["supports_reasoning"] is True
|
||||
|
||||
|
||||
def test_vision_flags_match_measured_surface() -> None:
|
||||
vision = {"glm-5.3-flash", "kimi-k2.6", "kimi-k2.7-code", "qwen3.8-27b"}
|
||||
for model in MODELS:
|
||||
entry = litellm.model_cost[f"clf_ai_gateway/{model}"]
|
||||
assert bool(entry.get("supports_vision")) is (model in vision), model
|
||||
|
||||
|
||||
def test_validate_environment_reports_env_key(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.delenv("CLF_AI_GATEWAY_API_KEY", raising=False)
|
||||
missing = litellm.validate_environment(model="clf_ai_gateway/glm-5.3")
|
||||
assert missing["keys_in_environment"] is False
|
||||
assert "CLF_AI_GATEWAY_API_KEY" in missing["missing_keys"]
|
||||
|
||||
monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test")
|
||||
present = litellm.validate_environment(model="clf_ai_gateway/glm-5.3")
|
||||
assert present["keys_in_environment"] is True
|
||||
assert present["missing_keys"] == []
|
||||
|
||||
|
||||
def test_get_supported_openai_params_dispatches_to_provider_config() -> None:
|
||||
params = litellm.get_supported_openai_params(model="glm-5.3", custom_llm_provider="clf_ai_gateway")
|
||||
assert params is not None
|
||||
assert "reasoning_effort" in params
|
||||
assert "logit_bias" not in params
|
||||
|
||||
|
||||
def test_completion_routes_without_network(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("CLF_AI_GATEWAY_API_KEY", "sk-gw-test")
|
||||
response = litellm.completion(
|
||||
model="clf_ai_gateway/glm-5.3",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
mock_response="hello from mock",
|
||||
)
|
||||
assert response.choices[0].message.content == "hello from mock"
|
||||
Loading…
Add table
Reference in a new issue