From 7d999a15864a2821dfd85f023c5064fe497eb897 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 20 Aug 2026 19:41:12 -0700 Subject: [PATCH] fix(cognition): price swe-1.7 at the standard tier, add swe-1.7-lightning The cost map shipped cognition/swe-1.7 at $2.50 in / $12.50 out per million with $1.00 cache reads. Those are the Lightning numbers. Cognition's own model list at https://docs.devin.ai/desktop/models has uid swe-1-7 at $0.50 / $2.50 with $0.20 cache reads, and uid swe-1-7-lightning at $2.50 / $12.50 with $1.00 cache reads, so every swe-1.7 call has been costed at 5x since the entry landed. swe-1.7 now carries the standard rates and the Lightning tier gets its own entry, in both cost map copies. The source field on both moves to the desktop models page, which is the one that lists both tiers. --- ...odel_prices_and_context_window_backup.json | 12 +++- model_prices_and_context_window.json | 12 +++- .../openai_like/test_cognition_provider.py | 58 ++++++++++++++++--- 3 files changed, 72 insertions(+), 10 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index efbe0d2ebb7..c5579f03a3e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -48949,6 +48949,16 @@ "source": "https://docs.devin.ai/windsurf/plugins/cascade/models" }, "cognition/swe-1.7": { + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2.5e-06, + "cache_read_input_token_cost": 2e-07, + "litellm_provider": "cognition", + "mode": "chat", + "supports_function_calling": true, + "supports_prompt_caching": true, + "source": "https://docs.devin.ai/desktop/models" + }, + "cognition/swe-1.7-lightning": { "input_cost_per_token": 2.5e-06, "output_cost_per_token": 1.25e-05, "cache_read_input_token_cost": 1e-06, @@ -48956,7 +48966,7 @@ "mode": "chat", "supports_function_calling": true, "supports_prompt_caching": true, - "source": "https://docs.devin.ai/windsurf/plugins/cascade/models" + "source": "https://docs.devin.ai/desktop/models" }, "pinstripes/ps/glm-4.5-air": { "max_tokens": 128000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index efbe0d2ebb7..c5579f03a3e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -48949,6 +48949,16 @@ "source": "https://docs.devin.ai/windsurf/plugins/cascade/models" }, "cognition/swe-1.7": { + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2.5e-06, + "cache_read_input_token_cost": 2e-07, + "litellm_provider": "cognition", + "mode": "chat", + "supports_function_calling": true, + "supports_prompt_caching": true, + "source": "https://docs.devin.ai/desktop/models" + }, + "cognition/swe-1.7-lightning": { "input_cost_per_token": 2.5e-06, "output_cost_per_token": 1.25e-05, "cache_read_input_token_cost": 1e-06, @@ -48956,7 +48966,7 @@ "mode": "chat", "supports_function_calling": true, "supports_prompt_caching": true, - "source": "https://docs.devin.ai/windsurf/plugins/cascade/models" + "source": "https://docs.devin.ai/desktop/models" }, "pinstripes/ps/glm-4.5-air": { "max_tokens": 128000, diff --git a/tests/test_litellm/llms/openai_like/test_cognition_provider.py b/tests/test_litellm/llms/openai_like/test_cognition_provider.py index c358d178f60..5c71b60e08a 100644 --- a/tests/test_litellm/llms/openai_like/test_cognition_provider.py +++ b/tests/test_litellm/llms/openai_like/test_cognition_provider.py @@ -111,33 +111,51 @@ class TestCognitionProviderIdentity: class TestCognitionCostTracking: @pytest.mark.parametrize( - "model, input_cost, output_cost", + "model, input_cost, output_cost, cache_read_cost", [ - ("cognition/swe-1.6", 5e-07, 2.5e-06), - ("cognition/swe-1.7", 2.5e-06, 1.25e-05), + ("cognition/swe-1.6", 5e-07, 2.5e-06, 2e-07), + ("cognition/swe-1.7", 5e-07, 2.5e-06, 2e-07), + ("cognition/swe-1.7-lightning", 2.5e-06, 1.25e-05, 1e-06), ], ) - def test_cost_map_entries(self, model: str, input_cost: float, output_cost: float): + def test_cost_map_entries(self, model: str, input_cost: float, output_cost: float, cache_read_cost: float): info = litellm.get_model_info(model=model) assert info["litellm_provider"] == "cognition" assert info["mode"] == "chat" assert info["input_cost_per_token"] == input_cost assert info["output_cost_per_token"] == output_cost + assert info["cache_read_input_token_cost"] == cache_read_cost - def test_cost_differs_from_openai_pricing(self): + @pytest.mark.parametrize( + "model, expected_prompt_cost, expected_completion_cost", + [ + ("cognition/swe-1.7", 0.5, 2.5), + ("cognition/swe-1.7-lightning", 2.5, 12.5), + ], + ) + def test_cost_differs_from_openai_pricing( + self, model: str, expected_prompt_cost: float, expected_completion_cost: float + ): """A cognition-prefixed model must never be priced off an OpenAI cost entry.""" from litellm.cost_calculator import cost_per_token prompt_cost, completion_cost = cost_per_token( - model="cognition/swe-1.7", + model=model, prompt_tokens=1_000_000, completion_tokens=1_000_000, custom_llm_provider="cognition", ) - assert prompt_cost == pytest.approx(2.5) - assert completion_cost == pytest.approx(12.5) + assert prompt_cost == pytest.approx(expected_prompt_cost) + assert completion_cost == pytest.approx(expected_completion_cost) + + def test_lightning_is_five_times_the_standard_tier(self): + standard = litellm.get_model_info(model="cognition/swe-1.7") + lightning = litellm.get_model_info(model="cognition/swe-1.7-lightning") + + assert lightning["input_cost_per_token"] == pytest.approx(standard["input_cost_per_token"] * 5) + assert lightning["output_cost_per_token"] == pytest.approx(standard["output_cost_per_token"] * 5) def test_supported_endpoints_matrix(self): matrix = json.loads((Path(litellm.__file__).parent / "provider_endpoints_support_backup.json").read_text()) @@ -170,6 +188,30 @@ class TestCognitionRouting: mock_response="hello from swe", ) + usage = response.usage + expected = usage.prompt_tokens * 5e-07 + usage.completion_tokens * 2.5e-06 + assert response._hidden_params["response_cost"] == pytest.approx(expected) + + @pytest.mark.asyncio + async def test_router_spend_uses_the_lightning_entry_for_lightning(self): + """The Lightning tier is its own model, costed off its own entry.""" + from litellm import Router + + router = Router( + model_list=[ + { + "model_name": "swe-lightning", + "litellm_params": {"model": "cognition/swe-1.7-lightning", "api_key": "sk-test"}, + } + ] + ) + + response = await router.acompletion( + model="swe-lightning", + messages=[{"role": "user", "content": "hi"}], + mock_response="hello from swe lightning", + ) + usage = response.usage expected = usage.prompt_tokens * 2.5e-06 + usage.completion_tokens * 1.25e-05 assert response._hidden_params["response_cost"] == pytest.approx(expected)