mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(apodex): correct model metadata and the cancel-response rebuild
Cross-checked the provider against platform.apodex.ai/docs and a live GET /v1/models call. - apodex-1.1 and apodex-1.1-mini advertised 256K max output; /v1/models reports 65536, and max_tokens is the legacy alias of max_output_tokens - apodex-1-1-deep-discover is Responses-API-only; /v1/chat/completions answers 400 unsupported_api for the Discover tiers - core models do not support response_format, so state it explicitly - transform_cancel_response_api_response carried Content-Encoding over to a response whose body it had already replaced, so httpx tried to decompress plain JSON on read. A non-JSON body (the gateway answers a timed-out cancel with an HTML 504) also escaped as a pydantic ValidationError instead of the provider error - drop the undocumented response.swarm.llm_delta mapping - only the Deep Research tiers default stream to true; the core models follow OpenAI and default it to false
This commit is contained in:
parent
4ace5c8db3
commit
c3f8e5aa67
9 changed files with 94 additions and 77 deletions
|
|
@ -351,7 +351,6 @@ def get_llm_provider(
|
|||
dynamic_api_key = get_secret_str("META_API_KEY")
|
||||
elif endpoint == litellm.ApodexChatConfig.API_BASE_URL:
|
||||
custom_llm_provider = "apodex" # rebind-ok: dispatch chain resolves in place
|
||||
# rebind-ok: dispatch chain resolves in place
|
||||
dynamic_api_key = litellm.ApodexChatConfig.get_api_key()
|
||||
|
||||
if api_base is not None and not isinstance(api_base, str):
|
||||
|
|
|
|||
|
|
@ -529,7 +529,8 @@ def anthropic_messages_handler(
|
|||
|
||||
anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig()
|
||||
if anthropic_messages_provider_config is None:
|
||||
# Route to Responses API for OpenAI / Azure, chat/completions for everything else.
|
||||
# Route to a Responses API for the providers that serve one, chat/completions
|
||||
# for everything else.
|
||||
_shared_kwargs: Final = dict(
|
||||
max_tokens=max_tokens,
|
||||
messages=messages,
|
||||
|
|
|
|||
|
|
@ -1,12 +1,14 @@
|
|||
"""
|
||||
Apodex chat completions — OpenAI-compatible, with two provider quirks:
|
||||
|
||||
- `stream` defaults to true upstream, so a non-streaming call has to say so
|
||||
explicitly or Apodex answers with SSE that a plain call cannot parse
|
||||
- the Deep Research tiers default `stream` to true, so a non-streaming call has
|
||||
to say so explicitly or Apodex answers with SSE that a plain call cannot
|
||||
parse. The core models follow OpenAI and default it to false
|
||||
- the Deep Research tiers ignore sampling parameters and reject OpenAI-style
|
||||
tools; only the core models take them
|
||||
|
||||
Ref: https://platform.apodex.ai/docs/chat-completions
|
||||
https://platform.apodex.ai/docs/models
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping
|
||||
|
|
@ -104,9 +106,10 @@ class ApodexChatConfig(OpenAIGPTConfig):
|
|||
if renamed.get("stream"):
|
||||
return renamed
|
||||
|
||||
# The OpenAI SDK drops `stream` from the body when it is false, which would
|
||||
# leave Apodex on its streaming default. extra_body is merged into the
|
||||
# request body by the SDK, so it survives that drop.
|
||||
# The OpenAI chat handler pops `stream` out of the params it forwards
|
||||
# (litellm/llms/openai/openai.py), which would leave the Deep Research tiers
|
||||
# on their streaming default. extra_body is merged into the request body
|
||||
# further down, so it survives that pop.
|
||||
requested_extra_body: Final = renamed.get("extra_body")
|
||||
extra_body: Final = (
|
||||
requested_extra_body if isinstance(requested_extra_body, Mapping) else {} # mutable-ok: JSON request body
|
||||
|
|
|
|||
|
|
@ -8,7 +8,8 @@ provider-wide:
|
|||
- core models are a stateless subset: `store` is forced to false, and
|
||||
`previous_response_id` or `background` come back as HTTP 400
|
||||
- the Deep Research tiers keep server-side state, so they take all three
|
||||
- both default `stream` to true, so a non-streaming call has to say so
|
||||
- the Deep Research tiers default `stream` to true, so a non-streaming call has
|
||||
to say so; pinning it for the core models too keeps one code path
|
||||
|
||||
Ref: https://platform.apodex.ai/docs/responses-api
|
||||
https://platform.apodex.ai/docs/models
|
||||
|
|
@ -21,14 +22,13 @@ from time import time
|
|||
from typing import TYPE_CHECKING, Final
|
||||
|
||||
import httpx
|
||||
from pydantic import TypeAdapter
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
|
||||
import litellm
|
||||
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
|
||||
from litellm.types.llms.openai import (
|
||||
ResponsesAPIOptionalRequestParams,
|
||||
ResponsesAPIResponse,
|
||||
ResponsesAPIStreamingResponse,
|
||||
)
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
|
@ -42,6 +42,7 @@ if TYPE_CHECKING:
|
|||
# to resume and requests are always executed inline.
|
||||
_STATEFUL_PARAMS: Final = ("previous_response_id", "background")
|
||||
_CANCEL_RESPONSE_ADAPTER: Final = TypeAdapter(dict[str, object])
|
||||
_BODY_FRAMING_HEADERS: Final = frozenset({"content-encoding", "content-length"})
|
||||
|
||||
|
||||
class ApodexResponsesConfig(OpenAIResponsesAPIConfig):
|
||||
|
|
@ -80,10 +81,22 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig):
|
|||
raw_response: httpx.Response,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
) -> ResponsesAPIResponse:
|
||||
payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content)
|
||||
"""Backfill the fields Apodex omits from a cancel payload but ResponsesAPIResponse requires."""
|
||||
try:
|
||||
payload: Final = _CANCEL_RESPONSE_ADAPTER.validate_json(raw_response.content)
|
||||
except ValidationError:
|
||||
return super().transform_cancel_response_api_response(
|
||||
raw_response=raw_response,
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
normalized_response: Final = httpx.Response(
|
||||
status_code=raw_response.status_code,
|
||||
headers=raw_response.headers,
|
||||
# Content-Encoding and Content-Length describe the body being replaced here;
|
||||
# carrying them over makes httpx try to decompress plain JSON on read.
|
||||
headers={
|
||||
name: value for name, value in raw_response.headers.items() if name.lower() not in _BODY_FRAMING_HEADERS
|
||||
},
|
||||
json={
|
||||
**payload,
|
||||
"created_at": payload.get("created_at", int(time())),
|
||||
|
|
@ -95,40 +108,6 @@ class ApodexResponsesConfig(OpenAIResponsesAPIConfig):
|
|||
logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
def transform_streaming_response(
|
||||
self,
|
||||
model: str,
|
||||
parsed_chunk: dict, # mutable-ok: matches the base-class signature
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
) -> ResponsesAPIStreamingResponse:
|
||||
swarm: Final = parsed_chunk.get("swarm")
|
||||
swarm_data: Final = swarm.get("data") if isinstance(swarm, dict) else None
|
||||
if (
|
||||
parsed_chunk.get("type") != "response.swarm.llm_delta"
|
||||
or not isinstance(swarm_data, dict)
|
||||
or swarm_data.get("channel") != "output_text"
|
||||
or not isinstance(swarm_data.get("delta"), str)
|
||||
):
|
||||
return super().transform_streaming_response(
|
||||
model=model,
|
||||
parsed_chunk=parsed_chunk,
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
response_id: Final = str(parsed_chunk.get("response_id", ""))
|
||||
return super().transform_streaming_response(
|
||||
model=model,
|
||||
parsed_chunk={
|
||||
"type": "response.output_text.delta",
|
||||
"item_id": f"msg_{response_id}",
|
||||
"output_index": 0,
|
||||
"content_index": 0,
|
||||
"delta": swarm_data["delta"],
|
||||
"sequence_number": parsed_chunk.get("sequence_number", 0),
|
||||
},
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: matches the base-class signature
|
||||
inherited: Final = super().get_supported_openai_params(model)
|
||||
if is_deep_research_model(model):
|
||||
|
|
|
|||
|
|
@ -48297,9 +48297,9 @@
|
|||
"supports_audio_output": true
|
||||
},
|
||||
"apodex/apodex-1.1": {
|
||||
"max_tokens": 262144,
|
||||
"max_tokens": 65536,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_output_tokens": 65536,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"output_cost_per_token": 3e-06,
|
||||
|
|
@ -48318,14 +48318,15 @@
|
|||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": false,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"apodex/apodex-1.1-mini": {
|
||||
"max_tokens": 262144,
|
||||
"max_tokens": 65536,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_output_tokens": 65536,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"output_cost_per_token": 1e-06,
|
||||
|
|
@ -48344,6 +48345,7 @@
|
|||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": false,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
|
|
@ -48404,7 +48406,6 @@
|
|||
"mode": "chat",
|
||||
"source": "https://platform.apodex.ai/docs/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supports_function_calling": false,
|
||||
|
|
|
|||
|
|
@ -48297,9 +48297,9 @@
|
|||
"supports_audio_output": true
|
||||
},
|
||||
"apodex/apodex-1.1": {
|
||||
"max_tokens": 262144,
|
||||
"max_tokens": 65536,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_output_tokens": 65536,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"output_cost_per_token": 3e-06,
|
||||
|
|
@ -48318,14 +48318,15 @@
|
|||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": false,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"apodex/apodex-1.1-mini": {
|
||||
"max_tokens": 262144,
|
||||
"max_tokens": 65536,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"max_output_tokens": 65536,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"output_cost_per_token": 1e-06,
|
||||
|
|
@ -48344,6 +48345,7 @@
|
|||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": false,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
|
|
@ -48404,7 +48406,6 @@
|
|||
"mode": "chat",
|
||||
"source": "https://platform.apodex.ai/docs/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supports_function_calling": false,
|
||||
|
|
|
|||
|
|
@ -99,10 +99,11 @@ class TestProviderResolution:
|
|||
|
||||
|
||||
class TestStreamDefault:
|
||||
"""Apodex defaults `stream` to true, so a non-streaming call has to pin it to false.
|
||||
"""The Deep Research tiers default `stream` to true, so a non-streaming call must pin it false.
|
||||
|
||||
Regression guard: the OpenAI SDK drops `stream` from the body when it is false,
|
||||
which would leave Apodex streaming SSE at a call that cannot parse it.
|
||||
Regression guard: the OpenAI chat handler pops `stream` out of the params it
|
||||
forwards, which would leave those tiers streaming SSE at a call that cannot
|
||||
parse it. The core models default to false but are pinned the same way.
|
||||
"""
|
||||
|
||||
def test_non_streaming_call_pins_stream_false(self):
|
||||
|
|
|
|||
|
|
@ -99,6 +99,9 @@ class TestModelMetadata:
|
|||
assert info["litellm_provider"] == "apodex"
|
||||
assert info["mode"] == "chat"
|
||||
assert info["max_input_tokens"] == 262144
|
||||
# GET /v1/models reports max_completion_tokens 65536, well under the context window
|
||||
assert info["max_output_tokens"] == 65536
|
||||
assert info["max_tokens"] == info["max_output_tokens"]
|
||||
assert info["input_cost_per_token"] == 3e-07
|
||||
assert info["cache_read_input_token_cost"] == 3e-08
|
||||
assert info["output_cost_per_token"] == 3e-06
|
||||
|
|
@ -126,6 +129,17 @@ class TestModelMetadata:
|
|||
"""Apodex serves /v1/messages for the core models only."""
|
||||
assert "/v1/messages" not in model_cost[f"apodex/{model}"]["supported_endpoints"]
|
||||
|
||||
def test_discover_is_responses_only(self, model_cost: dict):
|
||||
"""The Discover tiers answer 400 unsupported_api on /v1/chat/completions."""
|
||||
assert model_cost["apodex/apodex-1-1-deep-discover"]["supported_endpoints"] == ["/v1/responses"]
|
||||
|
||||
@pytest.mark.parametrize("model", ("apodex-1-1-deep-research", "apodex-1-1-deep-solve"))
|
||||
def test_the_other_deep_tiers_keep_chat_completions(self, model_cost: dict, model: str):
|
||||
assert model_cost[f"apodex/{model}"]["supported_endpoints"] == [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses",
|
||||
]
|
||||
|
||||
def test_backup_cost_map_in_sync(self, model_cost: dict):
|
||||
with open(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json") as f:
|
||||
backup = json.load(f)
|
||||
|
|
|
|||
|
|
@ -6,6 +6,8 @@ Research tiers keep server-side state, so the parameter contract is keyed off
|
|||
the model rather than applied provider-wide.
|
||||
"""
|
||||
|
||||
import gzip
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
|
|
@ -113,9 +115,7 @@ class TestStreamDefault:
|
|||
raise RuntimeError("captured")
|
||||
|
||||
with pytest.raises(Exception, match="captured"):
|
||||
await litellm.aresponses(
|
||||
model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler()
|
||||
)
|
||||
await litellm.aresponses(model=DEEP_RESEARCH_MODEL, input="hi", stream=True, client=CapturingHandler())
|
||||
|
||||
assert captured["body"]["stream"] is True
|
||||
|
||||
|
|
@ -203,21 +203,39 @@ class TestDeepResearchKeepsState:
|
|||
assert response.output == []
|
||||
assert response.created_at > 0
|
||||
|
||||
def test_deep_research_output_delta_is_normalized(self):
|
||||
config = _responses_config("apodex-1-1-deep-research")
|
||||
event = config.transform_streaming_response(
|
||||
model="apodex-1-1-deep-research",
|
||||
parsed_chunk={
|
||||
"type": "response.swarm.llm_delta",
|
||||
"response_id": "w_123",
|
||||
"sequence_number": 12,
|
||||
"swarm": {
|
||||
"agent_id": "reporter",
|
||||
"data": {"channel": "output_text", "delta": "final answer"},
|
||||
},
|
||||
def test_cancel_response_survives_a_compressed_upstream_response(self):
|
||||
"""The body is rebuilt, so the original framing headers must not follow it.
|
||||
|
||||
httpx decodes on read, so carrying Content-Encoding over from the compressed
|
||||
upstream response makes it try to gunzip the plain JSON replacement.
|
||||
"""
|
||||
body = b'{"id": "resp_1", "object": "response", "status": "cancelled"}'
|
||||
# As it arrives off the wire: httpx decodes the body but leaves the header in place
|
||||
upstream = httpx.Response(
|
||||
200,
|
||||
headers={
|
||||
"content-encoding": "gzip",
|
||||
"x-ratelimit-remaining-requests": "42",
|
||||
},
|
||||
content=gzip.compress(body),
|
||||
)
|
||||
assert upstream.content == body
|
||||
|
||||
config = _responses_config("apodex-1-1-deep-research")
|
||||
response = config.transform_cancel_response_api_response(
|
||||
raw_response=upstream,
|
||||
logging_obj=None,
|
||||
)
|
||||
assert event.type == "response.output_text.delta"
|
||||
assert event.item_id == "msg_w_123"
|
||||
assert event.delta == "final answer"
|
||||
|
||||
assert response.id == "resp_1"
|
||||
assert response.status == "cancelled"
|
||||
assert response._hidden_params["headers"]["x-ratelimit-remaining-requests"] == "42"
|
||||
|
||||
def test_non_json_cancel_body_raises_the_provider_error(self):
|
||||
"""The gateway answers a timed-out cancel with an HTML 504, not the JSON envelope."""
|
||||
config = _responses_config("apodex-1-1-deep-research")
|
||||
with pytest.raises(Exception, match="gateway timeout"):
|
||||
config.transform_cancel_response_api_response(
|
||||
raw_response=httpx.Response(504, content=b"<html>gateway timeout</html>"),
|
||||
logging_obj=None,
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue