mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
test(e2e): request reasoning explicitly on the reasoning-cost assertions
The two tests that assert on reasoning cost read reasoning_tokens off the response and required it to be nonzero, without ever asking the model to reason. Both now send reasoning_effort, so the assertion rests on a parameter the test sets rather than on the model's default behavior. The cache-breakdown test sends it on its prime call too: OpenAI's prefix cache keys on the reasoning setting as well as the tokens, so priming at a different effort never produces a read.
This commit is contained in:
parent
69278ae37b
commit
975a6806c3
2 changed files with 33 additions and 4 deletions
|
|
@ -30,6 +30,12 @@ OpenAI caching is best-effort, so each test retries with a fresh prefix (new
|
|||
marker = brand-new cache identity) up to three times before failing; the prime and
|
||||
measured calls share the prefix but differ in the trailing question, which defeats
|
||||
the proxy's own response cache without touching the provider's prefix cache.
|
||||
|
||||
The test that asserts on reasoning cost requests reasoning explicitly with
|
||||
`reasoning_effort`, so that assertion rests on a parameter the test sets rather
|
||||
than on whatever the model happens to do by default. Its prime call carries the
|
||||
same value: OpenAI's prefix cache keys on the reasoning setting as well as the
|
||||
tokens, so a prime at a different effort never produces a read.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
|
@ -67,6 +73,7 @@ CACHE_WRITE_RATE = 5e-05
|
|||
|
||||
PRIME_QUESTION = "Reply with the single word ready."
|
||||
REASONING_QUESTION = "Compute 47*83 - 19*7 step by step, then reply with just the final number."
|
||||
REASONING_EFFORT = "high"
|
||||
|
||||
|
||||
class _StreamChunk(BaseModel):
|
||||
|
|
@ -84,12 +91,15 @@ def _cache_priced_params(backend: str) -> LiteLLMParamsBody:
|
|||
)
|
||||
|
||||
|
||||
def _chat_body(model: str, content: str, *, stream: bool = False) -> ChatBody:
|
||||
def _chat_body(
|
||||
model: str, content: str, *, stream: bool = False, reasoning_effort: str | None = None
|
||||
) -> ChatBody:
|
||||
return ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content=content)],
|
||||
stream=stream,
|
||||
max_completion_tokens=4000,
|
||||
reasoning_effort=reasoning_effort,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -151,9 +161,23 @@ class TestCacheCostAccounting:
|
|||
|
||||
for _ in range(CACHE_ATTEMPTS):
|
||||
prefix = cacheable_prefix(unique_marker())
|
||||
unwrap(client.proxy.chat(scoped_key, _chat_body(model, f"{prefix}\n{PRIME_QUESTION}")))
|
||||
unwrap(
|
||||
client.proxy.chat(
|
||||
scoped_key,
|
||||
_chat_body(
|
||||
model, f"{prefix}\n{PRIME_QUESTION}", reasoning_effort=REASONING_EFFORT
|
||||
),
|
||||
)
|
||||
)
|
||||
chat = unwrap(
|
||||
client.proxy.chat(scoped_key, _chat_body(model, f"{prefix}\n{REASONING_QUESTION}"))
|
||||
client.proxy.chat(
|
||||
scoped_key,
|
||||
_chat_body(
|
||||
model,
|
||||
f"{prefix}\n{REASONING_QUESTION}",
|
||||
reasoning_effort=REASONING_EFFORT,
|
||||
),
|
||||
)
|
||||
)
|
||||
assert chat.id, f"chat response carried no id: {chat}"
|
||||
row = _require_row(client, chat.id)
|
||||
|
|
|
|||
|
|
@ -10,7 +10,9 @@ computed from the wrong tier (or a mix) cannot match the expected numbers. The
|
|||
prompt is a fresh unique marker per run, keeping cached tokens out of the math.
|
||||
The response's own `service_tier` echo is asserted first: if OpenAI ever declined
|
||||
priority processing and served the default tier, the test fails there instead of
|
||||
producing a vacuous rate comparison.
|
||||
producing a vacuous rate comparison. Reasoning is requested explicitly with
|
||||
`reasoning_effort`, so the reasoning-rate assertion rests on a parameter the test
|
||||
sets rather than on whatever the model happens to do by default.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
|
@ -38,6 +40,8 @@ OUTPUT_RATE = 8e-05
|
|||
PRIORITY_INPUT_RATE = 6e-05
|
||||
PRIORITY_OUTPUT_RATE = 1.6e-04
|
||||
|
||||
REASONING_EFFORT = "high"
|
||||
|
||||
|
||||
class TestServiceTierPricing:
|
||||
@pytest.mark.covers("quota_management.spend_tracking.service_tier.bills_tier_rates")
|
||||
|
|
@ -74,6 +78,7 @@ class TestServiceTierPricing:
|
|||
],
|
||||
max_completion_tokens=4000,
|
||||
service_tier="priority",
|
||||
reasoning_effort=REASONING_EFFORT,
|
||||
),
|
||||
)
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue