From 39d14bd8557737dbdc963748f46a909c005a0c24 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 19:08:16 -0700 Subject: [PATCH] feat(fireworks_ai): route and list the auto, auto-instant and firerouter routers (#43641) * feat(fireworks_ai): route and list the auto, auto-instant and firerouter routers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(fireworks_ai): drive the router request test through an httpx MockTransport Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(fireworks_ai): let custom firerouter/ IDs inherit the firerouter row's capabilities Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(fireworks_ai): integration coverage for router short names forwarding tool_choice and reasoning_effort Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(fireworks_ai): assert tool definitions reach the router upstream Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../llms/fireworks_ai/chat/transformation.py | 9 +++ litellm/llms/fireworks_ai/common_utils.py | 3 +- ...odel_prices_and_context_window_backup.json | 27 +++++++ model_prices_and_context_window.json | 27 +++++++ .../test_fireworks_ai_router_slug_wire.py | 74 +++++++++++++++++ .../test_fireworks_ai_chat_transformation.py | 81 +++++++++++++++++++ .../test_fireworks_ai_common_utils.py | 7 ++ 7 files changed, 227 insertions(+), 1 deletion(-) diff --git a/litellm/llms/fireworks_ai/chat/transformation.py b/litellm/llms/fireworks_ai/chat/transformation.py index 196022b3558..15049e71bbc 100644 --- a/litellm/llms/fireworks_ai/chat/transformation.py +++ b/litellm/llms/fireworks_ai/chat/transformation.py @@ -40,6 +40,7 @@ from ...openai.chat.gpt_transformation import ( OpenAIGPTConfig, ) from ..common_utils import ( + FIREROUTER, FireworksAIException, FireworksAIMixin, resolve_fireworks_resource_name, @@ -574,12 +575,20 @@ class FireworksAIConfig(FireworksAIMixin, OpenAIGPTConfig): short_name = short_name.removeprefix("accounts/fireworks/models/") return short_name + @staticmethod + def _firerouter_family_cost_keys(model: str) -> tuple[str, ...]: + firerouter_resource: Final = f"accounts/fireworks/routers/{FIREROUTER}" + if not resolve_fireworks_resource_name(model).startswith(f"{firerouter_resource}/"): + return () + return (f"fireworks_ai/{firerouter_resource}",) + def _get_model_cost_capability_exact(self, model: str, capability: str) -> bool | None: short_name: Final = self._short_model_name(model) candidate_keys: Final = ( model, f"fireworks_ai/{short_name}", f"fireworks_ai/accounts/fireworks/models/{short_name}", + *self._firerouter_family_cost_keys(model), ) for candidate_key in candidate_keys: model_info = litellm.model_cost.get(candidate_key) diff --git a/litellm/llms/fireworks_ai/common_utils.py b/litellm/llms/fireworks_ai/common_utils.py index ae52b89aa58..8352690235d 100644 --- a/litellm/llms/fireworks_ai/common_utils.py +++ b/litellm/llms/fireworks_ai/common_utils.py @@ -60,6 +60,7 @@ def resolve_fireworks_api_key(api_key: str | None) -> str | None: AZURE_FOUNDRY_FIREWORKS_MODEL_ID_PREFIX: Final = "FW-" FIREROUTER: Final = "firerouter" +ROUTER_SHORT_NAMES: Final = frozenset({FIREROUTER, "auto", "auto-instant"}) def resolve_fireworks_resource_name(model: str) -> str: @@ -68,7 +69,7 @@ def resolve_fireworks_resource_name(model: str) -> str: return stripped if stripped.startswith(("routers/", "models/")): return f"accounts/fireworks/{stripped}" - if stripped.endswith("-fast") or stripped == FIREROUTER or stripped.startswith(f"{FIREROUTER}/"): + if stripped.endswith("-fast") or stripped in ROUTER_SHORT_NAMES or stripped.startswith(f"{FIREROUTER}/"): return f"accounts/fireworks/routers/{stripped}" return f"accounts/fireworks/models/{stripped}" diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 58b4e9fb65a..7bb83a95bd4 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -64178,6 +64178,33 @@ "supports_tool_choice": true, "supports_vision": false }, + "fireworks_ai/accounts/fireworks/routers/auto": { + "litellm_provider": "fireworks_ai", + "mode": "chat", + "source": "https://docs.fireworks.ai/nexus/firerouter#example-router-ids", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "fireworks_ai/accounts/fireworks/routers/auto-instant": { + "litellm_provider": "fireworks_ai", + "mode": "chat", + "source": "https://docs.fireworks.ai/nexus/firerouter#example-router-ids", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "fireworks_ai/accounts/fireworks/routers/firerouter": { + "litellm_provider": "fireworks_ai", + "mode": "chat", + "source": "https://docs.fireworks.ai/nexus/firerouter#example-router-ids", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, "fireworks_ai/glm-5p3-fast": { "cache_read_input_token_cost": 3.9e-07, "input_cost_per_token": 2.1e-06, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 58b4e9fb65a..7bb83a95bd4 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -64178,6 +64178,33 @@ "supports_tool_choice": true, "supports_vision": false }, + "fireworks_ai/accounts/fireworks/routers/auto": { + "litellm_provider": "fireworks_ai", + "mode": "chat", + "source": "https://docs.fireworks.ai/nexus/firerouter#example-router-ids", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "fireworks_ai/accounts/fireworks/routers/auto-instant": { + "litellm_provider": "fireworks_ai", + "mode": "chat", + "source": "https://docs.fireworks.ai/nexus/firerouter#example-router-ids", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "fireworks_ai/accounts/fireworks/routers/firerouter": { + "litellm_provider": "fireworks_ai", + "mode": "chat", + "source": "https://docs.fireworks.ai/nexus/firerouter#example-router-ids", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, "fireworks_ai/glm-5p3-fast": { "cache_read_input_token_cost": 3.9e-07, "input_cost_per_token": 2.1e-06, diff --git a/tests/integration/providers/test_fireworks_ai_router_slug_wire.py b/tests/integration/providers/test_fireworks_ai_router_slug_wire.py index 55c83945ec7..f6262cc21f1 100644 --- a/tests/integration/providers/test_fireworks_ai_router_slug_wire.py +++ b/tests/integration/providers/test_fireworks_ai_router_slug_wire.py @@ -44,6 +44,35 @@ def _catalog_cost(model: str, field: str) -> float: _ROUTED_MODEL: Final = _pick_routed_model() +_FIREWORKS_MODEL_PREFIX: Final = "fireworks_ai/accounts/fireworks/models/" + + +def _pick_open_model_key() -> str: + catalog: Final = _COST_MAP.validate_json(_COST_MAP_PATH.read_bytes()) + return next( + key + for key, entry in catalog.items() + if key.startswith(_FIREWORKS_MODEL_PREFIX) + and _positive_rate(entry, "input_cost_per_token") + and _positive_rate(entry, "output_cost_per_token") + ) + + +_SERVED_OPEN_MODEL_KEY: Final = _pick_open_model_key() +_ROUTERS_ACCEPTING_TOOL_CHOICE_AND_REASONING: Final = ( + "auto", + "auto-instant", + "firerouter", + "firerouter/opus", + "firerouter/auto", +) +_WEATHER_TOOL: Final = { + "type": "function", + "function": { + "name": "get_weather", + "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]}, + }, +} def _approx(value: float) -> object: @@ -197,3 +226,48 @@ def test_fireworks_firerouter_claude_leg_is_charged_at_the_routed_models_own_rat spend: Final = rows[0]["spend"] assert isinstance(spend, (int, float, str)) assert float(spend) == _approx(expected_cost) + + +@pytest.mark.parametrize("router", _ROUTERS_ACCEPTING_TOOL_CHOICE_AND_REASONING) +def test_fireworks_router_forwards_tool_choice_and_reasoning_and_bills_the_served_open_model( + gateway: Gateway, router: str +) -> None: + identity: Final = f"fw-{router.replace('/', '-')}-{uuid.uuid4().hex}" + served_resource: Final = _SERVED_OPEN_MODEL_KEY.removeprefix("fireworks_ai/") + + def respond(request: Request) -> Reply: + body: Final = _provider_body(request, "/chat/completions") + assert body["model"] == f"accounts/fireworks/routers/{router}", body + assert body["tools"] == [_WEATHER_TOOL], body + assert body["tool_choice"] == "any", body + assert body["reasoning_effort"] == "low", body + return Reply(body=_chat_completion(identity, served_resource, 23, 41)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"fireworks_ai/{router}", api_base=wire.url, api_key=_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": _PROMPT}], + "tools": [_WEATHER_TOOL], + "tool_choice": "required", + "reasoning_effort": "low", + }, + ) + assert response.status_code == 200, response.text + expected_cost: Final = 23 * _catalog_cost(_SERVED_OPEN_MODEL_KEY, "input_cost_per_token") + 41 * _catalog_cost( + _SERVED_OPEN_MODEL_KEY, "output_cost_per_token" + ) + assert expected_cost > 0 + assert float(response.headers["x-litellm-response-cost"]) == _approx(expected_cost) + assert [(request.method, request.target) for request in wire.drain()] == [("POST", "/chat/completions")] + rows: Final = eventually( + lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)), + lambda values: len(values) == 1, + seconds=70, + ) + spend: Final = rows[0]["spend"] + assert isinstance(spend, (int, float, str)) + assert float(spend) == _approx(expected_cost) diff --git a/tests/unit/llms/fireworks_ai/chat/test_fireworks_ai_chat_transformation.py b/tests/unit/llms/fireworks_ai/chat/test_fireworks_ai_chat_transformation.py index 3f740edf834..2298bed2b76 100644 --- a/tests/unit/llms/fireworks_ai/chat/test_fireworks_ai_chat_transformation.py +++ b/tests/unit/llms/fireworks_ai/chat/test_fireworks_ai_chat_transformation.py @@ -1,10 +1,13 @@ import json +from typing import Final from unittest.mock import MagicMock, patch +import httpx import pytest import litellm from litellm.constants import SESSION_ID_GENERATED_METADATA_KEY +from litellm.llms.custom_httpx.http_handler import HTTPHandler from litellm.llms.fireworks_ai.chat.transformation import FireworksAIConfig from litellm.llms.fireworks_ai.common_utils import get_fireworks_session_id from litellm.types.utils import ( @@ -1781,8 +1784,86 @@ def test_streaming_preserves_selected_model_for_private_accounting(): [ ("deepseek-r1", "fireworks_ai/accounts/fireworks/models/deepseek-r1"), ("glm-5p3-fast", "fireworks_ai/accounts/fireworks/routers/glm-5p3-fast"), + ("auto", "fireworks_ai/accounts/fireworks/routers/auto"), ("accounts/fireworks/models/deepseek-r1", "fireworks_ai/accounts/fireworks/models/deepseek-r1"), ], ) def test_get_model_cost_key_resolves_short_names_to_long_keys(model: str, expected: str) -> None: assert FireworksAIConfig().get_model_cost_key(model) == expected + + +_LISTED_ROUTERS = ("auto", "auto-instant", "firerouter") + + +@pytest.mark.parametrize("router", _LISTED_ROUTERS) +def test_listed_router_short_name_resolves_to_its_catalog_row_and_accepts_tool_choice_and_reasoning( + router: str, +) -> None: + info = litellm.get_model_info(model=f"fireworks_ai/{router}") + params = FireworksAIConfig().get_supported_openai_params(router) + + assert info["key"] == f"fireworks_ai/accounts/fireworks/routers/{router}" + assert {"tools", "tool_choice", "reasoning_effort"} <= set(params), params + + +@pytest.mark.parametrize( + "router", + [ + "firerouter/opus", + "firerouter/auto", + "firerouter/auto-instant", + "firerouter/kimi-k3/glm-5p3", + "fireworks_ai/firerouter/opus", + "accounts/fireworks/routers/firerouter/opus", + ], +) +def test_custom_firerouter_id_accepts_the_same_tool_choice_and_reasoning_params_as_firerouter(router: str) -> None: + params: Final = FireworksAIConfig().get_supported_openai_params(router) + + assert {"tools", "tool_choice", "reasoning_effort"} <= set(params), params + + +@pytest.mark.parametrize("model", ["firerouter-v2", "models/firerouter-opus", "routers/firerouter-opus"]) +def test_names_that_only_start_with_firerouter_do_not_inherit_the_firerouter_row(model: str) -> None: + params: Final = FireworksAIConfig().get_supported_openai_params(model) + + assert "tool_choice" not in params, params + + +class _RecordingChatHandler: + def __init__(self, reply: dict[str, object]) -> None: + self.reply: Final = reply + self.request_body: dict[str, object] | None = None + + def __call__(self, request: httpx.Request) -> httpx.Response: + self.request_body = json.loads(request.content) + return httpx.Response(200, json=self.reply, request=request) + + +@pytest.mark.parametrize("router", _LISTED_ROUTERS) +def test_listed_router_request_is_sent_to_the_router_resource_and_billed_at_the_served_models_rate(router: str) -> None: + served_model: Final = "glm-5p3-flash" + handler: Final = _RecordingChatHandler( + { + "id": f"chat-{router}", + "object": "chat.completion", + "created": 1, + "model": served_model, + "choices": [{"index": 0, "message": {"role": "assistant", "content": "pong"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 23, "completion_tokens": 41, "total_tokens": 64}, + } + ) + + response: Final = litellm.completion( + model=f"fireworks_ai/{router}", + messages=[{"role": "user", "content": "ping"}], + api_key="fw-test-key", + client=HTTPHandler(client=httpx.Client(transport=httpx.MockTransport(handler))), + ) + + served_info: Final = litellm.model_cost[f"fireworks_ai/{served_model}"] + expected_cost: Final = 23 * served_info["input_cost_per_token"] + 41 * served_info["output_cost_per_token"] + assert handler.request_body is not None + assert handler.request_body["model"] == f"accounts/fireworks/routers/{router}" + assert expected_cost > 0 + assert response._hidden_params["response_cost"] == pytest.approx(expected_cost) diff --git a/tests/unit/llms/fireworks_ai/test_fireworks_ai_common_utils.py b/tests/unit/llms/fireworks_ai/test_fireworks_ai_common_utils.py index e505f2ae8a6..7ebbd0c6a8a 100644 --- a/tests/unit/llms/fireworks_ai/test_fireworks_ai_common_utils.py +++ b/tests/unit/llms/fireworks_ai/test_fireworks_ai_common_utils.py @@ -18,6 +18,13 @@ from litellm.llms.fireworks_ai.common_utils import resolve_fireworks_resource_na ("fireworks_ai/firerouter", "accounts/fireworks/routers/firerouter"), ("firerouter/kimi-k3/deepseek-v4", "accounts/fireworks/routers/firerouter/kimi-k3/deepseek-v4"), ("firerouter-v2", "accounts/fireworks/models/firerouter-v2"), + ("auto", "accounts/fireworks/routers/auto"), + ("fireworks_ai/auto", "accounts/fireworks/routers/auto"), + ("auto-instant", "accounts/fireworks/routers/auto-instant"), + ("fireworks_ai/auto-instant", "accounts/fireworks/routers/auto-instant"), + ("firerouter/auto", "accounts/fireworks/routers/firerouter/auto"), + ("autoglm-9b", "accounts/fireworks/models/autoglm-9b"), + ("auto-v2", "accounts/fireworks/models/auto-v2"), ( "accounts/fireworks/routers/glm-latest", "accounts/fireworks/routers/glm-latest",