feat(openinfer): add OpenInfer as OpenAI-compatible provider

Register OpenInfer in the JSON provider registry with catalog models,
proxy create fields, and unit coverage so requests route via openinfer/

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
AJ-ing 2026-10-01 10:05:44 +10:00
parent ed4caebb65
commit 64be726bee
9 changed files with 337 additions and 0 deletions

View file

@ -964,6 +964,7 @@ openai_compatible_endpoints: Final[list] = [
"https://api.scx.ai/v1",
"https://api.prisminference.com/v1",
"https://gigachat.devices.sberbank.ru/api/v1",
"https://api.openinfer.ai/v1",
]
@ -1032,6 +1033,7 @@ openai_compatible_providers: Final[list] = [
"docker_model_runner",
"ragflow",
"pinstripes", # Pinstripes - JSON-configured provider
"openinfer",
"darkbloom",
"meta", # Meta Model API (Muse Spark) - JSON-configured provider
"cognition",

View file

@ -212,5 +212,16 @@
"api_key_env": "SAIL_API_KEY",
"api_base_env": "SAIL_API_BASE",
"supported_endpoints": ["/v1/chat/completions", "/v1/responses", "/v1/messages"]
},
"openinfer": {
"base_url": "https://api.openinfer.ai/v1",
"api_key_env": "OPENINFER_API_KEY",
"api_base_env": "OPENINFER_API_BASE",
"base_class": "openai_gpt",
"param_mappings": {
"max_completion_tokens": "max_tokens"
},
"supported_endpoints": ["/v1/chat/completions"]
}
}

View file

@ -79420,5 +79420,49 @@
"supported_endpoints": [
"/v1/audio/speech"
]
},
"openinfer/@oi/Llama-3.2-1B-Instruct": {
"max_tokens": 8192,
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"input_cost_per_token": 2e-08,
"output_cost_per_token": 2e-08,
"litellm_provider": "openinfer",
"mode": "chat",
"supports_function_calling": true,
"source": "https://api.openinfer.ai/v1/models"
},
"openinfer/@oi/Qwen3.5-9B": {
"max_tokens": 8192,
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 1.8e-07,
"litellm_provider": "openinfer",
"mode": "chat",
"supports_function_calling": true,
"source": "https://api.openinfer.ai/v1/models"
},
"openinfer/@oi/Qwen3.5-27B": {
"max_tokens": 8192,
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"input_cost_per_token": 7.2e-07,
"output_cost_per_token": 7.2e-07,
"litellm_provider": "openinfer",
"mode": "chat",
"supports_function_calling": true,
"source": "https://api.openinfer.ai/v1/models"
},
"openinfer/@oi/Gemma4-31B-It": {
"max_tokens": 8192,
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"input_cost_per_token": 5.2e-07,
"output_cost_per_token": 7.5e-07,
"litellm_provider": "openinfer",
"mode": "chat",
"supports_function_calling": true,
"source": "https://api.openinfer.ai/v1/models"
}
}

View file

@ -2641,6 +2641,23 @@
"messages": true,
"responses": true
}
},
"openinfer": {
"display_name": "OpenInfer (`openinfer`)",
"url": "https://openinfer.io",
"endpoints": {
"chat_completions": true,
"messages": false,
"responses": false,
"embeddings": false,
"image_generations": false,
"audio_transcriptions": false,
"audio_speech": false,
"moderations": false,
"batches": false,
"rerank": false,
"a2a": false
}
}
},
"endpoints": {

View file

@ -2499,6 +2499,34 @@
],
"default_model_placeholder": "gpt-3.5-turbo"
},
{
"provider": "OPENINFER",
"provider_display_name": "OpenInfer",
"litellm_provider": "openinfer",
"credential_fields": [
{
"key": "api_base",
"label": "API Base",
"placeholder": "https://api.openinfer.ai/v1",
"tooltip": null,
"required": false,
"field_type": "text",
"options": null,
"default_value": null
},
{
"key": "api_key",
"label": "API Key",
"placeholder": null,
"tooltip": null,
"required": true,
"field_type": "password",
"options": null,
"default_value": null
}
],
"default_model_placeholder": "openinfer/@oi/Llama-3.2-1B-Instruct"
},
{
"provider": "OPENAI_LIKE",
"provider_display_name": "Openai Like",

View file

@ -4155,6 +4155,7 @@ class LlmProviders(str, Enum):
COGNITION = "cognition"
SCX_AI = "scx-ai"
PRISM = "prism"
OPENINFER = "openinfer"
DARKBLOOM = "darkbloom"
META = "meta"
SAIL = "sail"

View file

@ -79420,5 +79420,49 @@
"supported_endpoints": [
"/v1/audio/speech"
]
},
"openinfer/@oi/Llama-3.2-1B-Instruct": {
"max_tokens": 8192,
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"input_cost_per_token": 2e-08,
"output_cost_per_token": 2e-08,
"litellm_provider": "openinfer",
"mode": "chat",
"supports_function_calling": true,
"source": "https://api.openinfer.ai/v1/models"
},
"openinfer/@oi/Qwen3.5-9B": {
"max_tokens": 8192,
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 1.8e-07,
"litellm_provider": "openinfer",
"mode": "chat",
"supports_function_calling": true,
"source": "https://api.openinfer.ai/v1/models"
},
"openinfer/@oi/Qwen3.5-27B": {
"max_tokens": 8192,
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"input_cost_per_token": 7.2e-07,
"output_cost_per_token": 7.2e-07,
"litellm_provider": "openinfer",
"mode": "chat",
"supports_function_calling": true,
"source": "https://api.openinfer.ai/v1/models"
},
"openinfer/@oi/Gemma4-31B-It": {
"max_tokens": 8192,
"max_input_tokens": 262144,
"max_output_tokens": 8192,
"input_cost_per_token": 5.2e-07,
"output_cost_per_token": 7.5e-07,
"litellm_provider": "openinfer",
"mode": "chat",
"supports_function_calling": true,
"source": "https://api.openinfer.ai/v1/models"
}
}

View file

@ -3104,6 +3104,23 @@
"rerank": false,
"a2a": false
}
},
"openinfer": {
"display_name": "OpenInfer (`openinfer`)",
"url": "https://openinfer.io",
"endpoints": {
"chat_completions": true,
"messages": false,
"responses": false,
"embeddings": false,
"image_generations": false,
"audio_transcriptions": false,
"audio_speech": false,
"moderations": false,
"batches": false,
"rerank": false,
"a2a": false
}
}
},
"endpoints": {

View file

@ -0,0 +1,173 @@
"""
Tests for the OpenInfer LLM provider configuration and integration.
"""
import json
from pathlib import Path
import pytest
import litellm
_REPO_ROOT = Path(__file__).resolve().parents[4]
_PRICE_FILES = (
_REPO_ROOT / "model_prices_and_context_window.json",
_REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json",
)
class TestOpenInferProviderConfig:
def test_openinfer_in_provider_list(self):
from litellm import LlmProviders
assert LlmProviders.OPENINFER.value == "openinfer"
assert "openinfer" in litellm.provider_list
def test_openinfer_json_config(self):
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
provider = JSONProviderRegistry.get("openinfer")
assert provider is not None
assert provider.base_url == "https://api.openinfer.ai/v1"
assert provider.api_key_env == "OPENINFER_API_KEY"
assert provider.api_base_env == "OPENINFER_API_BASE"
assert not JSONProviderRegistry.supports_responses_api("openinfer")
def test_openinfer_in_openai_compatible_providers(self):
from litellm.constants import openai_compatible_providers
assert "openinfer" in openai_compatible_providers
def test_provider_prefixed_model_routes_to_openinfer(self):
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
model, provider, api_key, api_base = get_llm_provider(
model="openinfer/@oi/Llama-3.2-1B-Instruct",
custom_llm_provider=None,
api_base=None,
api_key="sk-test",
)
assert model == "@oi/Llama-3.2-1B-Instruct"
assert provider == "openinfer"
assert api_key == "sk-test"
assert api_base == "https://api.openinfer.ai/v1"
def test_api_key_and_base_resolved_from_env(self, monkeypatch):
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
monkeypatch.setenv("OPENINFER_API_KEY", "sk-env-key")
monkeypatch.setenv("OPENINFER_API_BASE", "https://proxy.internal/v1")
_, provider, api_key, api_base = get_llm_provider(
model="openinfer/@oi/Qwen3.5-9B",
custom_llm_provider=None,
api_base=None,
api_key=None,
)
assert provider == "openinfer"
assert api_key == "sk-env-key"
assert api_base == "https://proxy.internal/v1"
def test_url_autodetection_from_api_base(self, monkeypatch):
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
monkeypatch.setenv("OPENINFER_API_KEY", "sk-env-key")
_, provider, api_key, api_base = get_llm_provider(
model="@oi/Llama-3.2-1B-Instruct",
custom_llm_provider=None,
api_base="https://api.openinfer.ai/v1",
api_key=None,
)
assert provider == "openinfer"
assert api_key == "sk-env-key"
def test_chat_completions_url(self):
config = litellm.ProviderConfigManager.get_provider_chat_config(
model="@oi/Llama-3.2-1B-Instruct", provider=litellm.LlmProviders.OPENINFER
)
assert config is not None
assert (
config.get_complete_url(
api_base=None,
api_key="sk-test",
model="@oi/Llama-3.2-1B-Instruct",
optional_params={},
litellm_params={},
)
== "https://api.openinfer.ai/v1/chat/completions"
)
def test_outgoing_chat_request_contract(self):
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
from litellm.llms.openai_like.dynamic_config import create_config_class
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
model, provider, api_key, api_base = get_llm_provider(
model="openinfer/@oi/Llama-3.2-1B-Instruct",
custom_llm_provider=None,
api_base=None,
api_key="sk-test",
)
assert model == "@oi/Llama-3.2-1B-Instruct"
assert provider == "openinfer"
assert api_key == "sk-test"
assert api_base == "https://api.openinfer.ai/v1"
provider_cfg = JSONProviderRegistry.get("openinfer")
assert provider_cfg is not None
config = create_config_class(provider_cfg)()
assert (
config.get_complete_url(
api_base=None,
api_key=api_key,
model=model,
optional_params={},
litellm_params={},
)
== "https://api.openinfer.ai/v1/chat/completions"
)
headers = config.validate_environment(
headers={},
model=model,
messages=[{"role": "user", "content": "hi"}],
optional_params={},
litellm_params={},
api_key=api_key,
api_base=api_base,
)
assert headers["Authorization"] == "Bearer sk-test"
optional_params = config.map_openai_params(
non_default_params={"max_completion_tokens": 128},
optional_params={},
model=model,
drop_params=False,
)
assert optional_params["max_tokens"] == 128
assert "max_completion_tokens" not in optional_params
@pytest.mark.parametrize(
("model", "input_cost_per_token", "output_cost_per_token"),
(
("openinfer/@oi/Llama-3.2-1B-Instruct", 2e-08, 2e-08),
("openinfer/@oi/Qwen3.5-9B", 1.5e-07, 1.8e-07),
("openinfer/@oi/Qwen3.5-27B", 7.2e-07, 7.2e-07),
("openinfer/@oi/Gemma4-31B-It", 5.2e-07, 7.5e-07),
),
)
def test_catalog_token_rates_match_vendor_per_million_prices(
self, model: str, input_cost_per_token: float, output_cost_per_token: float
):
for path in _PRICE_FILES:
catalog = json.loads(path.read_text())
row = catalog[model]
assert row["input_cost_per_token"] == input_cost_per_token
assert row["output_cost_per_token"] == output_cost_per_token
assert row["input_cost_per_token"] > 0
assert row["output_cost_per_token"] > 0