diff --git a/tests/test_litellm/llms/databricks/test_databricks_cost_calculator.py b/tests/test_litellm/llms/databricks/test_databricks_cost_calculator.py index d4f96abfe37..29ad8ee4b6e 100644 --- a/tests/test_litellm/llms/databricks/test_databricks_cost_calculator.py +++ b/tests/test_litellm/llms/databricks/test_databricks_cost_calculator.py @@ -64,10 +64,6 @@ PUBLISHED_DBU_PER_MILLION: Final = { "databricks/databricks-kimi-k3": ("42.857", "214.286", "42.857", "4.286"), "databricks/databricks-glm-5-2": ("20.000", "62.857", "20.000", "3.714"), } -MILLION_TOKEN_CONTEXT_MODELS: Final = ( - "databricks/databricks-kimi-k3", - "databricks/databricks-glm-5-2", -) PROMOTIONAL_DISCOUNT: Final = 0.80 PROMOTION_EXPIRES: Final = "2027-01-31" ENTRIES_STORING_PROMOTIONAL_RATE: Final = ( @@ -217,33 +213,7 @@ def test_every_model_without_published_cache_dbu_bills_cache_at_its_own_input_ra assert info[field] == pytest.approx(info["input_cost_per_token"]), (model, field) -@pytest.mark.parametrize("model", MILLION_TOKEN_CONTEXT_MODELS) -def test_million_token_context_models_price_and_size_at_published_values( - local_model_cost_map: None, - model: str, -) -> None: - info: Final = _model_info(model) - input_dbu, output_dbu, _, cache_read_dbu = PUBLISHED_DBU_PER_MILLION[model] - - assert info["input_cost_per_token"] == _dollars_per_token(input_dbu) - assert info["output_cost_per_token"] == _dollars_per_token(output_dbu) - assert info["cache_read_input_token_cost"] == _dollars_per_token(cache_read_dbu) - assert info["max_input_tokens"] == 1000000 - assert info["mode"] == "chat" - assert info["supports_prompt_caching"] is True - - -def test_output_ceilings_match_what_each_vendor_publishes(local_model_cost_map: None) -> None: - assert _model_info("databricks/databricks-kimi-k3")["max_output_tokens"] == 1048576 - assert _model_info("databricks/databricks-glm-5-2")["max_output_tokens"] == 131072 - - -def test_kimi_k3_accepts_images_while_glm_5_2_is_text_only(local_model_cost_map: None) -> None: - assert _model_info("databricks/databricks-kimi-k3")["supports_vision"] is True - assert _model_info("databricks/databricks-glm-5-2")["supports_vision"] is False - - -@pytest.mark.parametrize("model", NEW_MODELS + MILLION_TOKEN_CONTEXT_MODELS) +@pytest.mark.parametrize("model", NEW_MODELS) def test_backup_price_map_matches_main(model: str) -> None: main_cost: Final = json.loads(MAIN_PRICES.read_text()) backup_cost: Final = json.loads(BACKUP_PRICES.read_text()) diff --git a/tests/test_litellm/llms/zai/test_zai_provider.py b/tests/test_litellm/llms/zai/test_zai_provider.py index ad87a65a740..38ddac8d510 100644 --- a/tests/test_litellm/llms/zai/test_zai_provider.py +++ b/tests/test_litellm/llms/zai/test_zai_provider.py @@ -128,36 +128,6 @@ def test_glm47_cost_calculation(local_model_cost_map): assert math.isclose(completion_cost, 2.2, rel_tol=1e-6) -def test_glm53_context_and_capabilities(local_model_cost_map): - """GLM-5.3 ships a 1M context with 128K max output, text in, reasoning always on""" - - info = litellm.model_cost["zai/glm-5.3"] - - assert info["max_input_tokens"] == 1000000 - assert info["max_output_tokens"] == 128000 - assert info["supports_reasoning"] is True - assert info["supports_prompt_caching"] is True - assert info.get("supports_vision") is not True - - -def test_glm53_cost_calculation(local_model_cost_map): - """GLM-5.3 bills $1.4/M input, $4.4/M output, $0.26/M on a cache hit""" - - prompt_cost, completion_cost = cost_per_token( - model="zai/glm-5.3", - prompt_tokens=1000000, - completion_tokens=1000000, - ) - - assert math.isclose(prompt_cost, 1.4, rel_tol=1e-6) - assert math.isclose(completion_cost, 4.4, rel_tol=1e-6) - assert math.isclose( - litellm.model_cost["zai/glm-5.3"]["cache_read_input_token_cost"] * 1000000, - 0.26, - rel_tol=1e-6, - ) - - @pytest.mark.asyncio async def test_zai_completion_call(respx_mock, zai_response, monkeypatch): """Test completion call with zai provider using mocked response"""