fix(config): use the Responses API whenever the model's catalog entry lists /v1/responses

This commit is contained in:
Ahmed Allam 2026-10-02 15:25:24 +00:00 • committed by Ahmed Allam
parent 2635fb5307
commit 99c0711687
2 changed files with 23 additions and 54 deletions

View file

@ -660,37 +660,23 @@ def configure_sdk_api_route(model_name: str, settings: Settings) -> None:
set_default_openai_api(api_type)
_OPENAI_HOSTS = frozenset({"api.openai.com"})
_CHAT_COMPLETIONS_ENDPOINT = "/v1/chat/completions"
_RESPONSES_ENDPOINT = "/v1/responses"
def resolve_api_type(model_name: str, settings: Settings) -> ApiType:
"""The SDK-native OpenAI route for ``model_name``: Responses or chat completions.
An explicit ``STRIX_API_TYPE`` wins. Without a base URL the request goes to
OpenAI itself, where every model is served by the Responses API. With a base
URL the model decides, not the URL: a model whose catalog entry lists no
``/v1/chat/completions`` endpoint cannot be reached there at all, and
api.openai.com behind a base URL is still OpenAI. Anything else is an
OpenAI-compatible gateway whose common denominator is chat completions.
An explicit ``STRIX_API_TYPE`` wins. Otherwise the model decides: Responses
when LiteLLM's catalog lists ``/v1/responses`` for it, chat completions for
everything else.
"""
if settings.llm.api_type is not None:
return settings.llm.api_type
api_base = (settings.llm.api_base or "").strip()
if not api_base or _is_openai_host(api_base):
return "responses"
endpoints = _catalog_supported_endpoints(model_name)
if endpoints and _CHAT_COMPLETIONS_ENDPOINT not in endpoints:
if _RESPONSES_ENDPOINT in _catalog_supported_endpoints(model_name):
return "responses"
return "chat_completions"
def _is_openai_host(api_base: str) -> bool:
from urllib.parse import urlsplit
return (urlsplit(api_base).hostname or "").lower() in _OPENAI_HOSTS
def _catalog_supported_endpoints(model_name: str) -> list[str]:
entry = _catalog_entry(model_name)
endpoints = entry.get("supported_endpoints") if entry else None
@ -899,9 +885,7 @@ def uses_chat_completions_tool_schema(model_name: str, settings: Settings) -> bo
model = model_name.strip().lower()
if "/" in model and not model.startswith("openai/"):
return True
if settings.llm.api_type is not None or settings.llm.api_base:
return resolve_api_type(model_name, settings) == "chat_completions"
return not model_supports_reasoning(model_name)
return resolve_api_type(model_name, settings) == "chat_completions"
def supports_strict_tool_schemas(model_name: str) -> bool:

View file

@ -113,7 +113,7 @@ def test_api_type_override_settings(monkeypatch: pytest.MonkeyPatch) -> None:
@pytest.mark.parametrize(
("api_type", "expected"),
[
(None, OpenAIChatCompletionsModel),
(None, OpenAIResponsesModel),
("chat_completions", OpenAIChatCompletionsModel),
("responses", OpenAIResponsesModel),
],
@ -121,7 +121,7 @@ def test_api_type_override_settings(monkeypatch: pytest.MonkeyPatch) -> None:
def test_api_type_overrides_the_api_base_route(
monkeypatch: pytest.MonkeyPatch, api_type: str | None, expected: type
) -> None:
"""``LLM_API_BASE`` defaults to chat completions. ``STRIX_API_TYPE`` must win."""
"""gpt-5 is catalogued on /v1/responses, so a base URL alone changes nothing."""
monkeypatch.setattr(_openai_shared, "_use_responses_by_default", True)
monkeypatch.setattr(_openai_shared, "_default_openai_client", None)
monkeypatch.setattr(_openai_shared, "_default_openai_key", None)
@ -152,35 +152,20 @@ def _settings(monkeypatch: pytest.MonkeyPatch, model: str, api_base: str | None)
return Settings()
def test_resolve_api_type_without_base_url_is_responses(monkeypatch: pytest.MonkeyPatch) -> None:
assert resolve_api_type("gpt-5", _settings(monkeypatch, "gpt-5", None)) == "responses"
assert resolve_api_type("gpt-4o", _settings(monkeypatch, "gpt-4o", None)) == "responses"
def test_resolve_api_type_gateway_defaults_to_chat_completions(
monkeypatch: pytest.MonkeyPatch,
@pytest.mark.parametrize(
"api_base", [None, "https://api.openai.com/v1", "https://gateway.example/v1"]
)
def test_resolve_api_type_follows_the_catalog_not_the_base_url(
monkeypatch: pytest.MonkeyPatch, api_base: str | None
) -> None:
settings = _settings(monkeypatch, "gpt-5.6-sol", "https://gateway.example/v1")
assert resolve_api_type("gpt-5.6-sol", settings) == "chat_completions"
assert resolve_api_type("my-private-model", settings) == "chat_completions"
def test_resolve_api_type_responses_only_model_ignores_base_url(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A model LiteLLM lists on /v1/responses alone cannot be served by chat completions."""
settings = _settings(monkeypatch, "gpt-daybreak-blue-latest", "https://gateway.example/v1")
assert resolve_api_type("gpt-daybreak-blue-latest", settings) == "responses"
assert resolve_api_type("openai/gpt-daybreak-blue-latest", settings) == "responses"
assert uses_chat_completions_tool_schema("gpt-daybreak-blue-latest", settings) is False
def test_resolve_api_type_openai_host_behind_base_url_is_responses(
monkeypatch: pytest.MonkeyPatch,
) -> None:
settings = _settings(monkeypatch, "gpt-5", "https://api.openai.com/v1")
assert resolve_api_type("gpt-5", settings) == "responses"
assert uses_chat_completions_tool_schema("gpt-5", settings) is False
"""Responses when LiteLLM lists /v1/responses for the model, chat completions otherwise."""
settings = _settings(monkeypatch, "gpt-5", api_base)
for model in ("gpt-5", "gpt-5.6-sol", "openai/gpt-5.4", "gpt-daybreak-blue-latest"):
assert resolve_api_type(model, settings) == "responses", model
assert uses_chat_completions_tool_schema(model, settings) is False, model
for model in ("gpt-4o", "my-private-model"):
assert resolve_api_type(model, settings) == "chat_completions", model
assert uses_chat_completions_tool_schema(model, settings) is True, model
def test_resolve_api_type_explicit_override_wins(monkeypatch: pytest.MonkeyPatch) -> None:
@ -214,6 +199,6 @@ def test_configure_sdk_api_route_follows_the_given_model(
settings = _settings(monkeypatch, "gpt-5", "https://gateway.example/v1")
models.configure_sdk_api_route("gpt-5", settings)
models.configure_sdk_api_route("gpt-daybreak-blue-latest", settings)
models.configure_sdk_api_route("my-private-model", settings)
assert routes == ["chat_completions", "responses"]
assert routes == ["responses", "chat_completions"]