From df7c7b05fae5c1abfaa29c9aafafd431110f7ba9 Mon Sep 17 00:00:00 2001 From: wassel alazhar Date: Sat, 3 Oct 2026 19:45:14 +0200 Subject: [PATCH 1/4] feat(providers): add Umans AI as an OpenAI-compatible provider --- litellm/constants.py | 2 + litellm/llms/openai_like/providers.json | 6 + .../provider_endpoints_support_backup.json | 18 ++ .../provider_create_fields.json | 28 +++ litellm/types/utils.py | 1 + provider_endpoints_support.json | 18 ++ .../openai_like/test_umans_ai_provider.py | 187 ++++++++++++++++++ 7 files changed, 260 insertions(+) create mode 100644 tests/unit/llms/openai_like/test_umans_ai_provider.py diff --git a/litellm/constants.py b/litellm/constants.py index 49514fc4d0e..e904c6eca11 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -965,6 +965,7 @@ openai_compatible_endpoints: Final[list] = [ "https://api.cortecs.ai/v1", "https://api.scx.ai/v1", "https://api.prisminference.com/v1", + "https://api.code.umans.ai/v1", "https://gigachat.devices.sberbank.ru/api/v1", ] @@ -1041,6 +1042,7 @@ openai_compatible_providers: Final[list] = [ "scx-ai", "prism", "sail", + "umans-ai", ] OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers)) diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index 61ff4be3a46..d8af6810f15 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -218,5 +218,11 @@ "api_key_env": "SAIL_API_KEY", "api_base_env": "SAIL_API_BASE", "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"] + }, + "umans-ai": { + "base_url": "https://api.code.umans.ai/v1", + "api_key_env": "UMANS_AI_API_KEY", + "api_base_env": "UMANS_AI_API_BASE", + "supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"] } } diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index c9635587eeb..28948f860ee 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -2262,6 +2262,24 @@ "interactions": true } }, + "umans-ai": { + "display_name": "Umans AI (`umans-ai`)", + "url": "https://docs.litellm.ai/docs/providers/umans_ai", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false, + "interactions": false + } + }, "v0": { "display_name": "V0 (`v0`)", "url": "https://docs.litellm.ai/docs/providers/v0", diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index 6e96d6ad0ec..856866a42e5 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -3291,6 +3291,34 @@ ], "default_model_placeholder": "gpt-3.5-turbo" }, + { + "provider": "UMANS_AI", + "provider_display_name": "Umans AI", + "litellm_provider": "umans-ai", + "credential_fields": [ + { + "key": "api_base", + "label": "API Base", + "placeholder": "https://api.code.umans.ai/v1", + "tooltip": null, + "required": false, + "field_type": "text", + "options": null, + "default_value": null + }, + { + "key": "api_key", + "label": "API Key", + "placeholder": null, + "tooltip": null, + "required": true, + "field_type": "password", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "umans-ai/umans-deepseek-v4-flash-0731" + }, { "provider": "V0", "provider_display_name": "V0", diff --git a/litellm/types/utils.py b/litellm/types/utils.py index a33eeaccaa3..8a50f9f0457 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4170,6 +4170,7 @@ class LlmProviders(str, Enum): DARKBLOOM = "darkbloom" META = "meta" SAIL = "sail" + UMANS_AI = "umans-ai" LITELLM_AGENT = "litellm_agent" CURSOR = "cursor" BEDROCK_MANTLE = "bedrock_mantle" diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index 7ffaacdb3aa..3388e50b830 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -2637,6 +2637,24 @@ "interactions": true } }, + "umans-ai": { + "display_name": "Umans AI (`umans-ai`)", + "url": "https://docs.litellm.ai/docs/providers/umans_ai", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false, + "interactions": false + } + }, "v0": { "display_name": "V0 (`v0`)", "url": "https://docs.litellm.ai/docs/providers/v0", diff --git a/tests/unit/llms/openai_like/test_umans_ai_provider.py b/tests/unit/llms/openai_like/test_umans_ai_provider.py new file mode 100644 index 00000000000..ce0387a485d --- /dev/null +++ b/tests/unit/llms/openai_like/test_umans_ai_provider.py @@ -0,0 +1,187 @@ +import json +from pathlib import Path +from typing import Final + +import pytest +import respx + +import litellm +from litellm.caching.llm_caching_handler import LLMClientCache + + +def test_umans_ai_provider_resolution(monkeypatch: pytest.MonkeyPatch): + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("UMANS_AI_API_KEY", "umans-test-key") + + model, provider, api_key, api_base = get_llm_provider( + model="umans-ai/umans-deepseek-v4-flash-0731", + custom_llm_provider=None, + api_base=None, + api_key=None, + ) + + assert model == "umans-deepseek-v4-flash-0731" + assert provider == "umans-ai" + assert api_key == "umans-test-key" + assert api_base == "https://api.code.umans.ai/v1" + + +def test_umans_ai_provider_keeps_explicit_credentials(monkeypatch: pytest.MonkeyPatch): + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + monkeypatch.setenv("UMANS_AI_API_KEY", "umans-env-key") + + _, provider, api_key, api_base = get_llm_provider( + model="umans-ai/umans-deepseek-v4-flash-0731", + custom_llm_provider=None, + api_base="https://umans.internal.example/v1", + api_key="umans-explicit-key", + ) + + assert provider == "umans-ai" + assert api_key == "umans-explicit-key" + assert api_base == "https://umans.internal.example/v1" + + +def test_umans_ai_is_available_in_add_model_form(): + fields_path = Path(litellm.__file__).parent / "proxy" / "public_endpoints" / "provider_create_fields.json" + providers = json.loads(fields_path.read_text()) + umans = next(provider for provider in providers if provider["litellm_provider"] == "umans-ai") + + assert umans["provider"] == "UMANS_AI" + assert umans["provider_display_name"] == "Umans AI" + assert umans["default_model_placeholder"] == "umans-ai/umans-deepseek-v4-flash-0731" + assert {field["key"]: field["required"] for field in umans["credential_fields"]} == { + "api_base": False, + "api_key": True, + } + + +def test_umans_ai_supported_endpoints(): + matrix_path = Path(litellm.__file__).parent / "provider_endpoints_support_backup.json" + providers = json.loads(matrix_path.read_text())["providers"] + + assert providers["umans-ai"]["endpoints"] == { + "chat_completions": True, + "messages": True, + "responses": True, + "embeddings": False, + "image_generations": False, + "audio_transcriptions": False, + "audio_speech": False, + "moderations": False, + "batches": False, + "rerank": False, + "a2a": False, + "interactions": False, + } + + +def test_umans_ai_chat_completion_request(): + with respx.mock() as upstream: + route: Final = upstream.post("https://api.code.umans.ai/v1/chat/completions").respond( + 200, + json={ + "id": "chatcmpl_umans", + "object": "chat.completion", + "created": 1_789_550_000, + "model": "umans-deepseek-v4-flash-0731", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "Hello from Umans AI"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 4, "completion_tokens": 3, "total_tokens": 7}, + }, + ) + response: Final = litellm.completion( + model="umans-ai/umans-deepseek-v4-flash-0731", + messages=[{"role": "user", "content": "Say hello"}], + api_key="umans-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.code.umans.ai/v1/chat/completions" + assert request.headers["authorization"] == "Bearer umans-test-key" + assert body["model"] == "umans-deepseek-v4-flash-0731" + assert body["messages"] == [{"role": "user", "content": "Say hello"}] + assert response.choices[0].message.content == "Hello from Umans AI" + + +def test_umans_ai_responses_request(): + with respx.mock() as upstream: + route: Final = upstream.post("https://api.code.umans.ai/v1/responses").respond( + 200, + json={ + "id": "resp_umans", + "object": "response", + "created_at": 1_789_550_000, + "model": "umans-deepseek-v4-flash-0731", + "status": "completed", + "output": [ + { + "id": "msg_umans", + "type": "message", + "role": "assistant", + "status": "completed", + "content": [{"type": "output_text", "text": "Hello from Umans AI", "annotations": []}], + } + ], + "usage": {"input_tokens": 4, "output_tokens": 3, "total_tokens": 7}, + }, + ) + response: Final = litellm.responses( + model="umans-ai/umans-deepseek-v4-flash-0731", + input="Say hello", + api_key="umans-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.code.umans.ai/v1/responses" + assert request.headers["authorization"] == "Bearer umans-test-key" + assert body["model"] == "umans-deepseek-v4-flash-0731" + assert body["input"] == "Say hello" + assert response.output[0].content[0].text == "Hello from Umans AI" + + +@pytest.mark.asyncio +async def test_umans_ai_anthropic_messages_request(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache()) + with respx.mock() as upstream: + route: Final = upstream.post("https://api.code.umans.ai/v1/messages").respond( + 200, + json={ + "id": "msg_umans", + "type": "message", + "role": "assistant", + "model": "umans-deepseek-v4-flash-0731", + "content": [{"type": "text", "text": "Hello from Umans AI"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 4, "output_tokens": 3}, + }, + ) + response: Final = await litellm.anthropic.messages.acreate( + model="umans-ai/umans-deepseek-v4-flash-0731", + messages=[{"role": "user", "content": "Say hello"}], + max_tokens=32, + api_key="umans-test-key", + ) + + request: Final = route.calls.last.request + body: Final = json.loads(request.content) + assert route.call_count == 1 + assert str(request.url) == "https://api.code.umans.ai/v1/messages" + assert request.headers["authorization"] == "Bearer umans-test-key" + assert request.headers["anthropic-version"] == "2023-06-01" + assert body["model"] == "umans-deepseek-v4-flash-0731" + assert body["messages"] == [{"role": "user", "content": "Say hello"}] + assert response["content"][0]["text"] == "Hello from Umans AI" From 70acd013095bf3da8331caa82e8f05864ebb5555 Mon Sep 17 00:00:00 2001 From: wassel alazhar Date: Sat, 3 Oct 2026 20:29:14 +0200 Subject: [PATCH 2/4] feat(providers): price Umans AI models and cover streaming paths Adds cost map entries (main and bundled backup) for the six public Umans AI models at the rates on app.umans.ai/pricing, so proxy spend and budgets count Umans usage. Adds streaming tests for chat completions, Responses and Anthropic Messages. --- ...odel_prices_and_context_window_backup.json | 143 +++++++++++++ model_prices_and_context_window.json | 143 +++++++++++++ .../openai_like/test_umans_ai_provider.py | 192 ++++++++++++++++++ 3 files changed, 478 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index a50edb0c9e3..d695aaa5594 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -80003,5 +80003,148 @@ "supports_tool_choice": true, "supports_video_input": false, "supports_vision": false + }, + "umans-ai/umans-deepseek-v4-flash-0731": { + "cache_read_input_token_cost": 2.8e-08, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 393215, + "max_tokens": 393215, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "umans-ai/umans-deepseek-v4.1-flash": { + "cache_read_input_token_cost": 2.8e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 393215, + "max_tokens": 393215, + "mode": "chat", + "output_cost_per_token": 6e-07, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "umans-ai/umans-glm-5.3-flash": { + "cache_read_input_token_cost": 3e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131071, + "max_tokens": 131071, + "mode": "chat", + "output_cost_per_token": 5e-07, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "umans-ai/umans-kimi-k3": { + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "umans-ai/umans-flash": { + "cache_read_input_token_cost": 5e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1e-06, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "umans-ai/umans-coder": { + "cache_read_input_token_cost": 3e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131071, + "max_tokens": 131071, + "mode": "chat", + "output_cost_per_token": 5e-07, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true } } diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a50edb0c9e3..d695aaa5594 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -80003,5 +80003,148 @@ "supports_tool_choice": true, "supports_video_input": false, "supports_vision": false + }, + "umans-ai/umans-deepseek-v4-flash-0731": { + "cache_read_input_token_cost": 2.8e-08, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 393215, + "max_tokens": 393215, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": false + }, + "umans-ai/umans-deepseek-v4.1-flash": { + "cache_read_input_token_cost": 2.8e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 393215, + "max_tokens": 393215, + "mode": "chat", + "output_cost_per_token": 6e-07, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "umans-ai/umans-glm-5.3-flash": { + "cache_read_input_token_cost": 3e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131071, + "max_tokens": 131071, + "mode": "chat", + "output_cost_per_token": 5e-07, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "umans-ai/umans-kimi-k3": { + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "umans-ai/umans-flash": { + "cache_read_input_token_cost": 5e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1e-06, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "umans-ai/umans-coder": { + "cache_read_input_token_cost": 3e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131071, + "max_tokens": 131071, + "mode": "chat", + "output_cost_per_token": 5e-07, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true } } diff --git a/tests/unit/llms/openai_like/test_umans_ai_provider.py b/tests/unit/llms/openai_like/test_umans_ai_provider.py index ce0387a485d..c3b0199822c 100644 --- a/tests/unit/llms/openai_like/test_umans_ai_provider.py +++ b/tests/unit/llms/openai_like/test_umans_ai_provider.py @@ -185,3 +185,195 @@ async def test_umans_ai_anthropic_messages_request(monkeypatch: pytest.MonkeyPat assert body["model"] == "umans-deepseek-v4-flash-0731" assert body["messages"] == [{"role": "user", "content": "Say hello"}] assert response["content"][0]["text"] == "Hello from Umans AI" + + +UMANS_PRICING: Final = ( + # model, then USD per token for input, output and cache read (app.umans.ai/pricing) + ("umans-ai/umans-deepseek-v4-flash-0731", 1.4e-07, 2.8e-07, 2.8e-08), + ("umans-ai/umans-deepseek-v4.1-flash", 1.5e-07, 6e-07, 2.8e-08), + ("umans-ai/umans-glm-5.3-flash", 1.5e-07, 5e-07, 3e-08), + ("umans-ai/umans-kimi-k3", 3e-06, 1.5e-05, 3e-07), + ("umans-ai/umans-flash", 1.5e-07, 1e-06, 5e-08), + ("umans-ai/umans-coder", 1.5e-07, 5e-07, 3e-08), +) + + +def _load_cost_map(filename: str = "model_prices_and_context_window.json") -> dict: + with open(Path(__file__).parents[4] / filename) as f: + return json.load(f) + + +@pytest.mark.parametrize(("model", "input_cost", "output_cost", "cache_read_cost"), UMANS_PRICING) +def test_umans_ai_models_are_priced(model: str, input_cost: float, output_cost: float, cache_read_cost: float): + entry: Final = _load_cost_map()[model] + + assert entry["litellm_provider"] == "umans-ai" + assert entry["mode"] == "chat" + assert entry["input_cost_per_token"] == input_cost + assert entry["output_cost_per_token"] == output_cost + assert entry["cache_read_input_token_cost"] == cache_read_cost + assert _load_cost_map("litellm/model_prices_and_context_window_backup.json")[model] == entry + + +def test_umans_ai_usage_is_billed(): + prompt_cost, completion_cost = litellm.cost_per_token( + model="umans-ai/umans-deepseek-v4-flash-0731", + prompt_tokens=1_000_000, + completion_tokens=1_000_000, + ) + + assert prompt_cost == pytest.approx(0.14) + assert completion_cost == pytest.approx(0.28) + + +def _sse(*events: dict) -> bytes: + return b"".join(f"data: {json.dumps(event)}\n\n".encode() for event in events) + b"data: [DONE]\n\n" + + +def _typed_sse(*events: dict) -> bytes: + return b"".join(f"event: {event['type']}\ndata: {json.dumps(event)}\n\n".encode() for event in events) + + +def test_umans_ai_chat_completion_streaming_request(): + chunk: Final = { + "id": "chatcmpl_umans", + "object": "chat.completion.chunk", + "created": 1_789_550_000, + "model": "umans-deepseek-v4-flash-0731", + } + stream_body: Final = _sse( + { + **chunk, + "choices": [{"index": 0, "delta": {"role": "assistant", "content": "Hello from "}, "finish_reason": None}], + }, + {**chunk, "choices": [{"index": 0, "delta": {"content": "Umans AI"}, "finish_reason": None}]}, + {**chunk, "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]}, + ) + with respx.mock() as upstream: + route: Final = upstream.post("https://api.code.umans.ai/v1/chat/completions").respond( + 200, content=stream_body, headers={"content-type": "text/event-stream"} + ) + chunks: Final = list( + litellm.completion( + model="umans-ai/umans-deepseek-v4-flash-0731", + messages=[{"role": "user", "content": "Say hello"}], + api_key="umans-test-key", + stream=True, + ) + ) + + body: Final = json.loads(route.calls.last.request.content) + assert route.call_count == 1 + assert str(route.calls.last.request.url) == "https://api.code.umans.ai/v1/chat/completions" + assert body["stream"] is True + assert "".join(chunk.choices[0].delta.content or "" for chunk in chunks) == "Hello from Umans AI" + + +def test_umans_ai_responses_streaming_request(): + response: Final = { + "id": "resp_umans", + "object": "response", + "created_at": 1_789_550_000, + "model": "umans-deepseek-v4-flash-0731", + "status": "in_progress", + "output": [], + } + message: Final = { + "id": "msg_umans", + "type": "message", + "role": "assistant", + "status": "completed", + "content": [{"type": "output_text", "text": "Hello from Umans AI", "annotations": []}], + } + stream_body: Final = _typed_sse( + {"type": "response.created", "sequence_number": 0, "response": response}, + { + "type": "response.output_text.delta", + "sequence_number": 1, + "item_id": "msg_umans", + "output_index": 0, + "content_index": 0, + "delta": "Hello from Umans AI", + }, + { + "type": "response.completed", + "sequence_number": 2, + "response": { + **response, + "status": "completed", + "output": [message], + "usage": {"input_tokens": 4, "output_tokens": 3, "total_tokens": 7}, + }, + }, + ) + with respx.mock() as upstream: + route: Final = upstream.post("https://api.code.umans.ai/v1/responses").respond( + 200, content=stream_body, headers={"content-type": "text/event-stream"} + ) + events: Final = list( + litellm.responses( + model="umans-ai/umans-deepseek-v4-flash-0731", + input="Say hello", + api_key="umans-test-key", + stream=True, + ) + ) + + body: Final = json.loads(route.calls.last.request.content) + assert route.call_count == 1 + assert str(route.calls.last.request.url) == "https://api.code.umans.ai/v1/responses" + assert body["stream"] is True + assert "".join(event.delta for event in events if event.type == "response.output_text.delta") == ( + "Hello from Umans AI" + ) + assert events[-1].type == "response.completed" + + +@pytest.mark.asyncio +async def test_umans_ai_anthropic_messages_streaming_request(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache()) + stream_body: Final = _typed_sse( + { + "type": "message_start", + "message": { + "id": "msg_umans", + "type": "message", + "role": "assistant", + "model": "umans-deepseek-v4-flash-0731", + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 4, "output_tokens": 0}, + }, + }, + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "Hello from Umans AI"}}, + {"type": "content_block_stop", "index": 0}, + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 3}, + }, + {"type": "message_stop"}, + ) + with respx.mock() as upstream: + route: Final = upstream.post("https://api.code.umans.ai/v1/messages").respond( + 200, content=stream_body, headers={"content-type": "text/event-stream"} + ) + stream = await litellm.anthropic.messages.acreate( + model="umans-ai/umans-deepseek-v4-flash-0731", + messages=[{"role": "user", "content": "Say hello"}], + max_tokens=32, + api_key="umans-test-key", + stream=True, + ) + streamed: Final = b"".join([chunk async for chunk in stream]).decode() + + body: Final = json.loads(route.calls.last.request.content) + assert route.call_count == 1 + assert str(route.calls.last.request.url) == "https://api.code.umans.ai/v1/messages" + assert body["stream"] is True + assert "event: message_start" in streamed + assert "Hello from Umans AI" in streamed + assert "event: message_stop" in streamed From b1a812178704caa534b48f40b1b456923c37aa47 Mon Sep 17 00:00:00 2001 From: wassel alazhar Date: Sat, 3 Oct 2026 22:13:35 +0200 Subject: [PATCH 3/4] feat(providers): price Umans AI GLM 5.3 Adds the umans-ai/umans-glm-5.3 cost map entry (main and bundled backup) at the rates on app.umans.ai/pricing and covers it in the pricing test. --- ...odel_prices_and_context_window_backup.json | 23 +++++++++++++++++++ model_prices_and_context_window.json | 23 +++++++++++++++++++ .../openai_like/test_umans_ai_provider.py | 1 + 3 files changed, 47 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d695aaa5594..882d6c08fa8 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -80051,6 +80051,29 @@ "supports_tool_choice": true, "supports_vision": true }, + "umans-ai/umans-glm-5.3": { + "cache_read_input_token_cost": 2.6e-07, + "input_cost_per_token": 1.4e-06, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131071, + "max_tokens": 131071, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": false + }, "umans-ai/umans-glm-5.3-flash": { "cache_read_input_token_cost": 3e-08, "input_cost_per_token": 1.5e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d695aaa5594..882d6c08fa8 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -80051,6 +80051,29 @@ "supports_tool_choice": true, "supports_vision": true }, + "umans-ai/umans-glm-5.3": { + "cache_read_input_token_cost": 2.6e-07, + "input_cost_per_token": 1.4e-06, + "litellm_provider": "umans-ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131071, + "max_tokens": 131071, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + "source": "https://app.umans.ai/pricing", + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": false + }, "umans-ai/umans-glm-5.3-flash": { "cache_read_input_token_cost": 3e-08, "input_cost_per_token": 1.5e-07, diff --git a/tests/unit/llms/openai_like/test_umans_ai_provider.py b/tests/unit/llms/openai_like/test_umans_ai_provider.py index c3b0199822c..c31b9cb55e8 100644 --- a/tests/unit/llms/openai_like/test_umans_ai_provider.py +++ b/tests/unit/llms/openai_like/test_umans_ai_provider.py @@ -191,6 +191,7 @@ UMANS_PRICING: Final = ( # model, then USD per token for input, output and cache read (app.umans.ai/pricing) ("umans-ai/umans-deepseek-v4-flash-0731", 1.4e-07, 2.8e-07, 2.8e-08), ("umans-ai/umans-deepseek-v4.1-flash", 1.5e-07, 6e-07, 2.8e-08), + ("umans-ai/umans-glm-5.3", 1.4e-06, 4.4e-06, 2.6e-07), ("umans-ai/umans-glm-5.3-flash", 1.5e-07, 5e-07, 3e-08), ("umans-ai/umans-kimi-k3", 3e-06, 1.5e-05, 3e-07), ("umans-ai/umans-flash", 1.5e-07, 1e-06, 5e-08), From 9bef613eb9cf7c6a6636143f6b04022907dbf3af Mon Sep 17 00:00:00 2001 From: wassel alazhar Date: Sat, 3 Oct 2026 22:39:00 +0200 Subject: [PATCH 4/4] test(providers): assert Umans AI cost map invariants instead of pinned prices Replaces literal vendor prices with checks LiteLLM owns: main and backup agree, every entry is a priced chat model, the Add Model default model is priced, and billing (cache reads included) uses the cost map rates. Marks test locals Final and types the SSE helpers as read-only mappings. --- .../openai_like/test_umans_ai_provider.py | 83 ++++++++++--------- 1 file changed, 46 insertions(+), 37 deletions(-) diff --git a/tests/unit/llms/openai_like/test_umans_ai_provider.py b/tests/unit/llms/openai_like/test_umans_ai_provider.py index c31b9cb55e8..8a233e778a0 100644 --- a/tests/unit/llms/openai_like/test_umans_ai_provider.py +++ b/tests/unit/llms/openai_like/test_umans_ai_provider.py @@ -1,5 +1,7 @@ import json +from collections.abc import Mapping from pathlib import Path +from types import MappingProxyType from typing import Final import pytest @@ -45,9 +47,9 @@ def test_umans_ai_provider_keeps_explicit_credentials(monkeypatch: pytest.Monkey def test_umans_ai_is_available_in_add_model_form(): - fields_path = Path(litellm.__file__).parent / "proxy" / "public_endpoints" / "provider_create_fields.json" - providers = json.loads(fields_path.read_text()) - umans = next(provider for provider in providers if provider["litellm_provider"] == "umans-ai") + fields_path: Final = Path(litellm.__file__).parent / "proxy" / "public_endpoints" / "provider_create_fields.json" + providers: Final = json.loads(fields_path.read_text()) + umans: Final = next(provider for provider in providers if provider["litellm_provider"] == "umans-ai") assert umans["provider"] == "UMANS_AI" assert umans["provider_display_name"] == "Umans AI" @@ -59,8 +61,8 @@ def test_umans_ai_is_available_in_add_model_form(): def test_umans_ai_supported_endpoints(): - matrix_path = Path(litellm.__file__).parent / "provider_endpoints_support_backup.json" - providers = json.loads(matrix_path.read_text())["providers"] + matrix_path: Final = Path(litellm.__file__).parent / "provider_endpoints_support_backup.json" + providers: Final = json.loads(matrix_path.read_text())["providers"] assert providers["umans-ai"]["endpoints"] == { "chat_completions": True, @@ -187,51 +189,58 @@ async def test_umans_ai_anthropic_messages_request(monkeypatch: pytest.MonkeyPat assert response["content"][0]["text"] == "Hello from Umans AI" -UMANS_PRICING: Final = ( - # model, then USD per token for input, output and cache read (app.umans.ai/pricing) - ("umans-ai/umans-deepseek-v4-flash-0731", 1.4e-07, 2.8e-07, 2.8e-08), - ("umans-ai/umans-deepseek-v4.1-flash", 1.5e-07, 6e-07, 2.8e-08), - ("umans-ai/umans-glm-5.3", 1.4e-06, 4.4e-06, 2.6e-07), - ("umans-ai/umans-glm-5.3-flash", 1.5e-07, 5e-07, 3e-08), - ("umans-ai/umans-kimi-k3", 3e-06, 1.5e-05, 3e-07), - ("umans-ai/umans-flash", 1.5e-07, 1e-06, 5e-08), - ("umans-ai/umans-coder", 1.5e-07, 5e-07, 3e-08), -) +COST_FIELDS: Final = ("input_cost_per_token", "output_cost_per_token", "cache_read_input_token_cost") -def _load_cost_map(filename: str = "model_prices_and_context_window.json") -> dict: - with open(Path(__file__).parents[4] / filename) as f: - return json.load(f) +def _umans_cost_entries(filename: str = "model_prices_and_context_window.json") -> Mapping[str, Mapping[str, object]]: + cost_map: Final = json.loads((Path(__file__).parents[4] / filename).read_text()) + return MappingProxyType({model: entry for model, entry in cost_map.items() if model.startswith("umans-ai/")}) -@pytest.mark.parametrize(("model", "input_cost", "output_cost", "cache_read_cost"), UMANS_PRICING) -def test_umans_ai_models_are_priced(model: str, input_cost: float, output_cost: float, cache_read_cost: float): - entry: Final = _load_cost_map()[model] +def test_umans_ai_cost_map_entries_match_backup_and_are_priced(): + entries: Final = _umans_cost_entries() - assert entry["litellm_provider"] == "umans-ai" - assert entry["mode"] == "chat" - assert entry["input_cost_per_token"] == input_cost - assert entry["output_cost_per_token"] == output_cost - assert entry["cache_read_input_token_cost"] == cache_read_cost - assert _load_cost_map("litellm/model_prices_and_context_window_backup.json")[model] == entry + assert entries + assert entries == _umans_cost_entries("litellm/model_prices_and_context_window_backup.json") + for entry in entries.values(): + assert entry["litellm_provider"] == "umans-ai" + assert entry["mode"] == "chat" + assert entry["max_tokens"] == entry["max_output_tokens"] + assert all(isinstance(cost, float) and cost > 0 for cost in (entry[field] for field in COST_FIELDS)) -def test_umans_ai_usage_is_billed(): - prompt_cost, completion_cost = litellm.cost_per_token( - model="umans-ai/umans-deepseek-v4-flash-0731", - prompt_tokens=1_000_000, - completion_tokens=1_000_000, +def test_umans_ai_default_model_is_priced(): + fields_path: Final = Path(litellm.__file__).parent / "proxy" / "public_endpoints" / "provider_create_fields.json" + umans: Final = next( + provider for provider in json.loads(fields_path.read_text()) if provider["litellm_provider"] == "umans-ai" ) - assert prompt_cost == pytest.approx(0.14) - assert completion_cost == pytest.approx(0.28) + assert umans["default_model_placeholder"] in _umans_cost_entries() -def _sse(*events: dict) -> bytes: +def test_umans_ai_usage_is_billed_at_cost_map_rates(): + model: Final = "umans-ai/umans-deepseek-v4-flash-0731" + entry: Final = _umans_cost_entries()[model] + input_cost: Final = entry["input_cost_per_token"] + output_cost: Final = entry["output_cost_per_token"] + cache_read_cost: Final = entry["cache_read_input_token_cost"] + assert isinstance(input_cost, float) + assert isinstance(output_cost, float) + assert isinstance(cache_read_cost, float) + + prompt_cost, completion_cost = litellm.cost_per_token( + model=model, prompt_tokens=1_000, completion_tokens=2_000, cache_read_input_tokens=800 + ) + + assert prompt_cost == pytest.approx(200 * input_cost + 800 * cache_read_cost) + assert completion_cost == pytest.approx(2_000 * output_cost) + + +def _sse(*events: Mapping[str, object]) -> bytes: return b"".join(f"data: {json.dumps(event)}\n\n".encode() for event in events) + b"data: [DONE]\n\n" -def _typed_sse(*events: dict) -> bytes: +def _typed_sse(*events: Mapping[str, object]) -> bytes: return b"".join(f"event: {event['type']}\ndata: {json.dumps(event)}\n\n".encode() for event in events) @@ -362,7 +371,7 @@ async def test_umans_ai_anthropic_messages_streaming_request(monkeypatch: pytest route: Final = upstream.post("https://api.code.umans.ai/v1/messages").respond( 200, content=stream_body, headers={"content-type": "text/event-stream"} ) - stream = await litellm.anthropic.messages.acreate( + stream: Final = await litellm.anthropic.messages.acreate( model="umans-ai/umans-deepseek-v4-flash-0731", messages=[{"role": "user", "content": "Say hello"}], max_tokens=32,