From 975a6806c318976efd1d9adc6ee44aed4d8ff8d4 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 20 Aug 2026 04:11:08 -0700 Subject: [PATCH] test(e2e): request reasoning explicitly on the reasoning-cost assertions The two tests that assert on reasoning cost read reasoning_tokens off the response and required it to be nonzero, without ever asking the model to reason. Both now send reasoning_effort, so the assertion rests on a parameter the test sets rather than on the model's default behavior. The cache-breakdown test sends it on its prime call too: OpenAI's prefix cache keys on the reasoning setting as well as the tokens, so priming at a different effort never produces a read. --- .../test_cache_cost_accounting_e2e.py | 30 +++++++++++++++++-- .../test_service_tier_pricing_e2e.py | 7 ++++- 2 files changed, 33 insertions(+), 4 deletions(-) diff --git a/tests/e2e/quota_management/spend_tracking/test_cache_cost_accounting_e2e.py b/tests/e2e/quota_management/spend_tracking/test_cache_cost_accounting_e2e.py index ca5985fcda6..c50ec3d902f 100644 --- a/tests/e2e/quota_management/spend_tracking/test_cache_cost_accounting_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_cache_cost_accounting_e2e.py @@ -30,6 +30,12 @@ OpenAI caching is best-effort, so each test retries with a fresh prefix (new marker = brand-new cache identity) up to three times before failing; the prime and measured calls share the prefix but differ in the trailing question, which defeats the proxy's own response cache without touching the provider's prefix cache. + +The test that asserts on reasoning cost requests reasoning explicitly with +`reasoning_effort`, so that assertion rests on a parameter the test sets rather +than on whatever the model happens to do by default. Its prime call carries the +same value: OpenAI's prefix cache keys on the reasoning setting as well as the +tokens, so a prime at a different effort never produces a read. """ import pytest @@ -67,6 +73,7 @@ CACHE_WRITE_RATE = 5e-05 PRIME_QUESTION = "Reply with the single word ready." REASONING_QUESTION = "Compute 47*83 - 19*7 step by step, then reply with just the final number." +REASONING_EFFORT = "high" class _StreamChunk(BaseModel): @@ -84,12 +91,15 @@ def _cache_priced_params(backend: str) -> LiteLLMParamsBody: ) -def _chat_body(model: str, content: str, *, stream: bool = False) -> ChatBody: +def _chat_body( + model: str, content: str, *, stream: bool = False, reasoning_effort: str | None = None +) -> ChatBody: return ChatBody( model=model, messages=[ChatMessage(role="user", content=content)], stream=stream, max_completion_tokens=4000, + reasoning_effort=reasoning_effort, ) @@ -151,9 +161,23 @@ class TestCacheCostAccounting: for _ in range(CACHE_ATTEMPTS): prefix = cacheable_prefix(unique_marker()) - unwrap(client.proxy.chat(scoped_key, _chat_body(model, f"{prefix}\n{PRIME_QUESTION}"))) + unwrap( + client.proxy.chat( + scoped_key, + _chat_body( + model, f"{prefix}\n{PRIME_QUESTION}", reasoning_effort=REASONING_EFFORT + ), + ) + ) chat = unwrap( - client.proxy.chat(scoped_key, _chat_body(model, f"{prefix}\n{REASONING_QUESTION}")) + client.proxy.chat( + scoped_key, + _chat_body( + model, + f"{prefix}\n{REASONING_QUESTION}", + reasoning_effort=REASONING_EFFORT, + ), + ) ) assert chat.id, f"chat response carried no id: {chat}" row = _require_row(client, chat.id) diff --git a/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py b/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py index 171c849fb4c..770c5699b4e 100644 --- a/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py @@ -10,7 +10,9 @@ computed from the wrong tier (or a mix) cannot match the expected numbers. The prompt is a fresh unique marker per run, keeping cached tokens out of the math. The response's own `service_tier` echo is asserted first: if OpenAI ever declined priority processing and served the default tier, the test fails there instead of -producing a vacuous rate comparison. +producing a vacuous rate comparison. Reasoning is requested explicitly with +`reasoning_effort`, so the reasoning-rate assertion rests on a parameter the test +sets rather than on whatever the model happens to do by default. """ import pytest @@ -38,6 +40,8 @@ OUTPUT_RATE = 8e-05 PRIORITY_INPUT_RATE = 6e-05 PRIORITY_OUTPUT_RATE = 1.6e-04 +REASONING_EFFORT = "high" + class TestServiceTierPricing: @pytest.mark.covers("quota_management.spend_tracking.service_tier.bills_tier_rates") @@ -74,6 +78,7 @@ class TestServiceTierPricing: ], max_completion_tokens=4000, service_tier="priority", + reasoning_effort=REASONING_EFFORT, ), ) )