From 62ab9e5a5d6b8ed544f03c58ec0b47f1fb926f02 Mon Sep 17 00:00:00 2001 From: halfaipg Date: Fri, 28 Aug 2026 23:48:16 -0400 Subject: [PATCH] fix(provider): correct AIPG output limits --- litellm/model_prices_and_context_window_backup.json | 9 ++++++--- model_prices_and_context_window.json | 9 ++++++--- tests/llm_translation/test_aipg.py | 10 ++++++---- 3 files changed, 18 insertions(+), 10 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 1c57107350b..9f41cc85f90 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -54601,7 +54601,8 @@ "input_cost_per_token": 7.5e-08, "litellm_provider": "aipg", "max_input_tokens": 60000, - "max_tokens": 60000, + "max_output_tokens": 32768, + "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 3e-07, "source": "https://docs.aipowergrid.io/streaming-api", @@ -54614,7 +54615,8 @@ "input_cost_per_token": 7e-08, "litellm_provider": "aipg", "max_input_tokens": 262144, - "max_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1.4e-07, "source": "https://docs.aipowergrid.io/streaming-api", @@ -54627,7 +54629,8 @@ "input_cost_per_token": 5e-09, "litellm_provider": "aipg", "max_input_tokens": 2048, - "max_tokens": 2048, + "max_output_tokens": 1024, + "max_tokens": 1024, "mode": "chat", "output_cost_per_token": 1e-08, "source": "https://docs.aipowergrid.io/streaming-api", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 1c57107350b..9f41cc85f90 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -54601,7 +54601,8 @@ "input_cost_per_token": 7.5e-08, "litellm_provider": "aipg", "max_input_tokens": 60000, - "max_tokens": 60000, + "max_output_tokens": 32768, + "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 3e-07, "source": "https://docs.aipowergrid.io/streaming-api", @@ -54614,7 +54615,8 @@ "input_cost_per_token": 7e-08, "litellm_provider": "aipg", "max_input_tokens": 262144, - "max_tokens": 262144, + "max_output_tokens": 32768, + "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1.4e-07, "source": "https://docs.aipowergrid.io/streaming-api", @@ -54627,7 +54629,8 @@ "input_cost_per_token": 5e-09, "litellm_provider": "aipg", "max_input_tokens": 2048, - "max_tokens": 2048, + "max_output_tokens": 1024, + "max_tokens": 1024, "mode": "chat", "output_cost_per_token": 1e-08, "source": "https://docs.aipowergrid.io/streaming-api", diff --git a/tests/llm_translation/test_aipg.py b/tests/llm_translation/test_aipg.py index 075284aebfd..d229ee0a0a2 100644 --- a/tests/llm_translation/test_aipg.py +++ b/tests/llm_translation/test_aipg.py @@ -70,14 +70,16 @@ def test_get_llm_provider_aipg(): def test_aipg_model_metadata(): model_cost = litellm.get_model_cost_map(url="") expected = { - "aipg/gpt-oss-120b": (60000, 7.5e-08, 3e-07), - "aipg/deepseek-v4-flash-nvfp4": (262144, 7e-08, 1.4e-07), - "aipg/Smollm-135m": (2048, 5e-09, 1e-08), + "aipg/gpt-oss-120b": (60000, 32768, 7.5e-08, 3e-07), + "aipg/deepseek-v4-flash-nvfp4": (262144, 32768, 7e-08, 1.4e-07), + "aipg/Smollm-135m": (2048, 1024, 5e-09, 1e-08), } - for model, (context, input_cost, output_cost) in expected.items(): + for model, (context, output_limit, input_cost, output_cost) in expected.items(): info = model_cost[model] assert info["litellm_provider"] == "aipg" assert info["mode"] == "chat" assert info["max_input_tokens"] == context + assert info["max_output_tokens"] == output_limit + assert info["max_tokens"] == output_limit assert info["input_cost_per_token"] == input_cost assert info["output_cost_per_token"] == output_cost