mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(boldrouter): add model metadata so tools, vision and cost tracking work; keep it off the transcription path
This commit is contained in:
parent
1ab3dffd0a
commit
1e53db3641
4 changed files with 2824 additions and 3 deletions
|
|
@ -950,7 +950,6 @@ openai_compatible_endpoints: Final[list] = [
|
|||
"https://api.cognition.ai/v1",
|
||||
"https://api.scx.ai/v1",
|
||||
"https://api.prisminference.com/v1",
|
||||
"https://boldrouter.com/v1",
|
||||
"https://gigachat.devices.sberbank.ru/api/v1",
|
||||
]
|
||||
|
||||
|
|
@ -1026,7 +1025,6 @@ openai_compatible_providers: Final[list] = [
|
|||
"scx-ai",
|
||||
"prism",
|
||||
"sail",
|
||||
"boldrouter", # BoldRouter - JSON-configured provider
|
||||
]
|
||||
|
||||
OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers))
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
File diff suppressed because it is too large
Load diff
|
|
@ -40,7 +40,6 @@ def test_boldrouter_provider_resolution(monkeypatch: pytest.MonkeyPatch):
|
|||
api_key=None,
|
||||
)
|
||||
|
||||
# BoldRouter model IDs are themselves provider-prefixed; only the litellm prefix is stripped.
|
||||
assert model == "anthropic/claude-sonnet-4.6"
|
||||
assert provider == "boldrouter"
|
||||
assert api_key == "sk-bold-test"
|
||||
|
|
@ -148,3 +147,151 @@ async def test_boldrouter_async_chat_request(monkeypatch: pytest.MonkeyPatch):
|
|||
assert route.call_count == 1
|
||||
assert body["model"] == "deepseek/deepseek-v4-flash"
|
||||
assert response.choices[0].message.content == "Hi"
|
||||
|
||||
|
||||
BOLDROUTER_MODELS: Final = tuple(sorted(name for name in litellm.model_cost if name.startswith("boldrouter/")))
|
||||
BOLDROUTER_ROUTERS: Final = ("boldrouter/bold/auto", "boldrouter/bold/fast", "boldrouter/bold/frontier")
|
||||
|
||||
|
||||
def test_boldrouter_backup_registry_mirrors_cost_map():
|
||||
package_root = Path(litellm.__file__).parent
|
||||
cost_map = json.loads((package_root.parent / "model_prices_and_context_window.json").read_text())
|
||||
backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text())
|
||||
entries = {name: entry for name, entry in cost_map.items() if name.startswith("boldrouter/")}
|
||||
|
||||
assert tuple(sorted(entries)) == BOLDROUTER_MODELS
|
||||
assert set(BOLDROUTER_ROUTERS) <= set(entries)
|
||||
assert entries == {name: backup[name] for name in entries}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", [m for m in BOLDROUTER_MODELS if m not in BOLDROUTER_ROUTERS])
|
||||
def test_boldrouter_model_cost(model: str):
|
||||
from litellm.cost_calculator import cost_per_token
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model=model,
|
||||
prompt_tokens=1_000_000,
|
||||
completion_tokens=1_000_000,
|
||||
custom_llm_provider="boldrouter",
|
||||
)
|
||||
model_info = litellm.get_model_info(model)
|
||||
|
||||
assert model_info["litellm_provider"] == "boldrouter"
|
||||
assert model_info["mode"] == "chat"
|
||||
assert model_info["input_cost_per_token"] > 0
|
||||
assert prompt_cost == pytest.approx(model_info["input_cost_per_token"] * 1_000_000)
|
||||
assert completion_cost == pytest.approx(model_info["output_cost_per_token"] * 1_000_000)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", BOLDROUTER_ROUTERS)
|
||||
def test_boldrouter_routers_support_tools_and_vision(model: str):
|
||||
assert litellm.supports_function_calling(model) is True
|
||||
assert litellm.supports_vision(model) is True
|
||||
|
||||
|
||||
def test_boldrouter_chat_request_sends_tools(monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache())
|
||||
tool: Final = {
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"parameters": {"type": "object", "properties": {"city": {"type": "string"}}},
|
||||
},
|
||||
}
|
||||
tool_call_response: Final = _chat_completion("")
|
||||
tool_call_response["choices"][0]["message"] = {
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"type": "function",
|
||||
"function": {"name": "get_weather", "arguments": '{"city": "Zurich"}'},
|
||||
}
|
||||
],
|
||||
}
|
||||
tool_call_response["choices"][0]["finish_reason"] = "tool_calls"
|
||||
with respx.mock() as upstream:
|
||||
route: Final = upstream.post(BOLDROUTER_CHAT_URL).respond(200, json=tool_call_response)
|
||||
response: Final = litellm.completion(
|
||||
model="boldrouter/bold/auto",
|
||||
messages=[{"role": "user", "content": "Weather in Zurich?"}],
|
||||
api_key="sk-bold-test",
|
||||
tools=[tool],
|
||||
tool_choice="auto",
|
||||
)
|
||||
|
||||
body: Final = json.loads(route.calls.last.request.content)
|
||||
assert body["tools"] == [tool]
|
||||
assert body["tool_choice"] == "auto"
|
||||
assert response.choices[0].message.tool_calls[0].function.name == "get_weather"
|
||||
|
||||
|
||||
def test_boldrouter_streaming_reports_usage(monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache())
|
||||
chunks: Final = [
|
||||
{"choices": [{"index": 0, "delta": {"role": "assistant", "content": "Hel"}, "finish_reason": None}]},
|
||||
{"choices": [{"index": 0, "delta": {"content": "lo"}, "finish_reason": None}]},
|
||||
{"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]},
|
||||
{
|
||||
"choices": [],
|
||||
"usage": {
|
||||
"prompt_tokens": 5,
|
||||
"completion_tokens": 12,
|
||||
"total_tokens": 17,
|
||||
"completion_tokens_details": {"reasoning_tokens": 10},
|
||||
},
|
||||
},
|
||||
]
|
||||
sse: Final = (
|
||||
"".join(
|
||||
"data: "
|
||||
+ json.dumps(
|
||||
{
|
||||
"id": "chatcmpl-boldrouter",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "bold/fast",
|
||||
**c,
|
||||
}
|
||||
)
|
||||
+ "\n\n"
|
||||
for c in chunks
|
||||
)
|
||||
+ "data: [DONE]\n\n"
|
||||
)
|
||||
with respx.mock() as upstream:
|
||||
route: Final = upstream.post(BOLDROUTER_CHAT_URL).respond(
|
||||
200, content=sse.encode(), headers={"content-type": "text/event-stream"}
|
||||
)
|
||||
stream: Final = litellm.completion(
|
||||
model="boldrouter/bold/fast",
|
||||
messages=[{"role": "user", "content": "Say hello"}],
|
||||
api_key="sk-bold-test",
|
||||
stream=True,
|
||||
stream_options={"include_usage": True},
|
||||
)
|
||||
text = ""
|
||||
usage = None
|
||||
for chunk in stream:
|
||||
if chunk.choices:
|
||||
text += chunk.choices[0].delta.content or ""
|
||||
usage = getattr(chunk, "usage", None) or usage
|
||||
|
||||
body: Final = json.loads(route.calls.last.request.content)
|
||||
assert body["stream"] is True
|
||||
assert text == "Hello"
|
||||
assert usage is not None
|
||||
assert usage.completion_tokens == 12
|
||||
assert usage.completion_tokens_details.reasoning_tokens == 10
|
||||
|
||||
|
||||
def test_boldrouter_transcription_is_rejected_before_any_request():
|
||||
with respx.mock() as upstream:
|
||||
with pytest.raises(Exception, match="Unmapped provider"):
|
||||
litellm.transcription(
|
||||
model="boldrouter/whisper-1",
|
||||
file=("audio.wav", b"RIFF0000WAVE", "audio/wav"),
|
||||
api_key="sk-bold-test",
|
||||
)
|
||||
assert upstream.calls.call_count == 0
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue