mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-24 00:52:24 +00:00
* ci: benchmark and gate an installed release wheel Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * ci: simplify installed-wheel benchmark check Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(rust): add native tokenizer codec Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(tokenizer): route Python tokenization through the Rust extension Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * style(lint): format tokenizer call Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(packaging): restore runtime dependencies and native images Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(tokenizer): preserve Python SDK behavior with Rust tokenizers * fix(tokenizer): restore compatibility paths Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(tokenizer): count custom tokenizers directly Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(tokenizer): preserve caller-supplied Python tokenizer counts * fix(tokenizer): reuse packaged vocabularies in the native wheel * refactor(rust_bridge): route token counting through the catalog as RUST_OPT_IN Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(spend_tracking): compare tokenizer groups by value Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(deps): re-resolve filelock under the <4.0 pin Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(llms): align transformation override signatures with base configs Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * build(rust): use fat LTO to keep the native wheel under the 35 MB limit Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(tokenizer): preserve Python defaults with opt-in Rust dispatch * test(proxy): tolerate missing litellm.utils.Tokenizer when patching it Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(proxy): patch the tokenizer dispatch function instead of the removed alias Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(tokenizer): give the Rust wrappers the tiktoken and tokenizers surface Callers of litellm.encoding and litellm.create_tokenizer must see the same read-only API whichever backend the catalog selects. - OpenAIEncoding mirrors tiktoken.Encoding: n_vocab, max_token_value, token_byte_values, encode_single_token, encode_with_unstable, encode_to_numpy, decode_with_offsets, is_special_token, repr; the Rust tiktoken crate keeps a Vocabulary beside each CoreBPE and reports the requested encoding name (gpt2 stays gpt2). - HuggingFaceTokenizer mirrors the read-only tokenizers.Tokenizer surface (token_to_id, id_to_token, get_vocab, get_vocab_size, get_added_tokens_decoder, num_special_tokens_to_add, padding, truncation, encode_special_tokens, from_buffer); HuggingFaceEncoding gains the char/word/token lookups, pad, truncate, set_sequence_id and merge. Mutators stay on the Python tokenizer. - from_json/from_pretrained claim the fork gate only when the huggingface feature is compiled in; the surrogate fallback matches on the Codec. - Tokenizer caching is keyed on the same catalog Context the dispatch runs on; rust_tokenizer reads the encoding name without loading an encoding; LITELLM_RUST parsing is cached. - Drop the unused tiktoken_encoding_for_model export and Error::Download. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> * fix(tokenizer): close the exhaustive matches with assert_never CodeQL reads a `match` over a Literal with no default arm as an implicit `None` return. `assert_never` makes the exhaustiveness explicit for both the HuggingFace tokenizer loader and the Rust token-counter factory. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> * feat(tokenizer): derive the fast counter from the shared tokenizer The count-only counter (`fast` feature) and the codec each parsed the same artifact: TokenCounter took the Anthropic JSON and the tiktoken rank files from Python while Tokenizer loaded them again. One parse now serves both. - FastTokenizer builds from a model another loader holds: `from_shared` takes the Arc<tokenizers::Tokenizer> the HF codec keeps, and `from_*_pairs` take the ranks the tiktoken vocabulary already parsed. - `FastCounter::fast_counter` in the core crate derives it from either codec; encodings the fast scanner does not reproduce are refused. - Native `Tokenizer.count(text, fast=False)` opts into that counter, built once per tokenizer on first use; `TokenCounter.from_tokenizer(tokenizer, fast=False)` replaces the JSON and rank-file constructors. - The Python route counts over the native tokenizers the codec path shares (`native_encoding`, `native_anthropic`) and no longer reads rank files; the packaged Anthropic tokenizer has one loader, `tokenizer_dispatch.anthropic`. - Public wrappers gain `count(text, fast=False)`. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> --------- Co-authored-by: Yujong Lee <yujong@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: Claude Fable 5.1 <noreply@anthropic.com>
104 lines
3.6 KiB
Python
104 lines
3.6 KiB
Python
"""
|
|
Regression tests for the proxy token_counter custom_tokenizer bug.
|
|
|
|
Bug: model_info was never populated from the matched deployment, so
|
|
custom_tokenizer was always None and token counting silently fell back to the
|
|
OpenAI tokenizer instead of the configured HuggingFace tokenizer.
|
|
|
|
The HuggingFace download boundary (Tokenizer.from_pretrained) is mocked so these
|
|
stay hermetic unit tests; the proxy's extraction-and-selection path runs for real.
|
|
"""
|
|
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
import pytest
|
|
|
|
import litellm
|
|
import litellm.proxy.proxy_server
|
|
import litellm.utils
|
|
from litellm import Router
|
|
from litellm.proxy._types import TokenCountRequest
|
|
from litellm.proxy.proxy_server import token_counter
|
|
|
|
|
|
def _fake_hf_tokenizer(num_tokens: int) -> MagicMock:
|
|
tokenizer = MagicMock()
|
|
tokenizer.encode_batch_fast.return_value = [[0] * num_tokens]
|
|
return tokenizer
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_custom_tokenizer_from_model_info_is_used(monkeypatch):
|
|
"""
|
|
A deployment carrying model_info.custom_tokenizer must load and use that
|
|
tokenizer. The model name deliberately matches no built-in HuggingFace
|
|
tokenizer, so without the fix the response would fall back to
|
|
"openai_tokenizer" and from_pretrained would never see the configured id.
|
|
"""
|
|
llm_router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "my-embedding-model",
|
|
"litellm_params": {
|
|
"model": "openai/self-hosted-embedder",
|
|
"api_base": "http://localhost:8080/v1",
|
|
},
|
|
"model_info": {
|
|
"mode": "embedding",
|
|
"custom_tokenizer": {
|
|
"identifier": "my-org/custom-tokenizer",
|
|
"revision": "v2",
|
|
"auth_token": None,
|
|
},
|
|
},
|
|
}
|
|
]
|
|
)
|
|
monkeypatch.setattr(litellm.proxy.proxy_server, "llm_router", llm_router)
|
|
|
|
with patch.object(litellm.utils, "tokenizer_dispatch") as mock_tokenizer_cls:
|
|
mock_tokenizer_cls.from_pretrained.return_value = _fake_hf_tokenizer(7)
|
|
|
|
response = await token_counter(
|
|
request=TokenCountRequest(
|
|
model="my-embedding-model",
|
|
messages=[{"role": "user", "content": "Bonjour le monde"}],
|
|
)
|
|
)
|
|
|
|
mock_tokenizer_cls.from_pretrained.assert_called_once_with("my-org/custom-tokenizer", revision="v2", token=None)
|
|
assert response.tokenizer_type == "huggingface_tokenizer"
|
|
assert response.request_model == "my-embedding-model"
|
|
assert response.model_used == "self-hosted-embedder"
|
|
assert response.total_tokens >= 7
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_model_without_custom_tokenizer_uses_default(monkeypatch):
|
|
"""
|
|
Control: a deployment with no custom_tokenizer must not touch HuggingFace and
|
|
must report the default OpenAI tokenizer.
|
|
"""
|
|
llm_router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "gpt-4",
|
|
"litellm_params": {"model": "gpt-4"},
|
|
"model_info": {},
|
|
}
|
|
]
|
|
)
|
|
monkeypatch.setattr(litellm.proxy.proxy_server, "llm_router", llm_router)
|
|
|
|
with patch.object(litellm.utils, "tokenizer_dispatch") as mock_tokenizer_cls:
|
|
response = await token_counter(
|
|
request=TokenCountRequest(
|
|
model="gpt-4",
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
)
|
|
)
|
|
|
|
mock_tokenizer_cls.from_pretrained.assert_not_called()
|
|
assert response.tokenizer_type == "openai_tokenizer"
|
|
assert response.model_used == "gpt-4"
|
|
assert response.total_tokens > 0
|