mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
perf(proxy): count admission tokens with the Rust fast counter when LITELLM_RUST is on
Route.TOKEN_COUNTER rolls out as RUST_OPT_IN, so a process with LITELLM_RUST=1 reaches the native body counter instead of the Python tokenizers, and the counter is built over the byte-level fast path (exact count, no encode). Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
086d76ab54
commit
121c09e928
4 changed files with 5 additions and 5 deletions
|
|
@ -88,7 +88,7 @@ RULES: Final[Rules] = (
|
|||
RouteRule(Route.MESSAGES, Rollout.RUST_OPT_IN, providers=frozenset({"anthropic"})),
|
||||
RouteRule(Route.MESSAGES, Rollout.PYTHON_ONLY),
|
||||
RouteRule(Route.RESPONSES, Rollout.PYTHON_ONLY),
|
||||
RouteRule(Route.TOKEN_COUNTER, Rollout.PYTHON_ONLY),
|
||||
RouteRule(Route.TOKEN_COUNTER, Rollout.RUST_OPT_IN),
|
||||
RouteRule(Route.TOKENIZER, Rollout.PYTHON_ONLY),
|
||||
RouteRule(Route.TRANSCRIPTION, Rollout.RUST_REQUIRED, providers=frozenset({"bedrock"})),
|
||||
SecretManagerRule(Rollout.PYTHON_ONLY, systems=frozenset({KeyManagementSystem.GOOGLE_KMS.value})),
|
||||
|
|
|
|||
|
|
@ -79,7 +79,7 @@ def rust_tokenizer(model: str) -> RustTokenizer | None:
|
|||
|
||||
@lru_cache(maxsize=4)
|
||||
def _counter(factory: RustTokenCounterFactory, tokenizer: RustTokenizer) -> RustTokenCounter:
|
||||
return factory.from_tokenizer(_native_tokenizer(tokenizer))
|
||||
return factory.from_tokenizer(_native_tokenizer(tokenizer), fast=True)
|
||||
|
||||
|
||||
def _native_tokenizer(tokenizer: RustTokenizer) -> NativeTokenizer:
|
||||
|
|
|
|||
|
|
@ -47,7 +47,7 @@ def test_shipped_decisions(
|
|||
if route is Route.OCR or (route is Route.TRANSCRIPTION and provider == "bedrock"):
|
||||
assert catalog.rollout(context) is Rollout.RUST_REQUIRED
|
||||
assert catalog.decision(context) is Decision.RUST_REQUIRED
|
||||
elif route is Route.MESSAGES and provider == "anthropic":
|
||||
elif (route is Route.MESSAGES and provider == "anthropic") or route is Route.TOKEN_COUNTER:
|
||||
assert catalog.rollout(context) is Rollout.RUST_OPT_IN
|
||||
opted_in: Final = environment == "1" or (environment is None and process is True)
|
||||
assert catalog.decision(context) is (Decision.RUST_WITH_FALLBACK if opted_in else Decision.PYTHON)
|
||||
|
|
|
|||
|
|
@ -94,7 +94,7 @@ async def test_native_count_returns_typed_count_and_reuses_one_counter(fake_toke
|
|||
assert second == first
|
||||
assert len(factory.counters) == 1
|
||||
assert factory.counters[0].bodies == [BODY, BODY]
|
||||
assert factory.counters[0].fast is False
|
||||
assert factory.counters[0].fast is True
|
||||
assert factory.counters[0].tokenizer is tokenizer_dispatch.native_anthropic()
|
||||
assert json.loads(factory.counters[0].tokenizer.json or "")["model"]["type"] == "BPE"
|
||||
|
||||
|
|
@ -113,7 +113,7 @@ async def test_tiktoken_counter_is_built_over_the_shared_encoding_once(
|
|||
assert len(factory.counters) == 1
|
||||
assert factory.counters[0].tokenizer.name == tokenizer
|
||||
assert factory.counters[0].tokenizer is tokenizer_dispatch.native_encoding(tokenizer)
|
||||
assert factory.counters[0].fast is False
|
||||
assert factory.counters[0].fast is True
|
||||
assert factory.counters[0].bodies == [BODY, BODY]
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue