mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
refactor(pricing): bill reference pixels at the shared input_cost_per_pixel rate
This commit is contained in:
parent
384c7558b2
commit
4db328319d
16 changed files with 16 additions and 83 deletions
|
|
@ -345,8 +345,6 @@ pub struct ModelInfo {
|
|||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub input_cost_per_pixel: Option<f64>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub input_cost_per_reference_pixel: Option<f64>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub input_cost_per_query: Option<f64>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub input_cost_per_request: Option<f64>,
|
||||
|
|
|
|||
|
|
@ -57,8 +57,8 @@ def cost_calculator(
|
|||
|
||||
num_images: Final = n if n is not None else len(image_response.data or ())
|
||||
generated_cost: Final = _generated_cost(resolved, num_images, _generated_pixels(optional_params, size))
|
||||
reference_rate: Final = _rate(resolved, "input_cost_per_reference_pixel") or 0.0
|
||||
reference_cost: Final = reference_rate * (image_response.reference_pixels or 0)
|
||||
per_pixel: Final = _rate(resolved, "input_cost_per_pixel") or 0.0
|
||||
reference_cost: Final = per_pixel * (image_response.reference_pixels or 0)
|
||||
return generated_cost + reference_cost
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -10964,7 +10964,6 @@
|
|||
},
|
||||
"azure_ai/FLUX.2-flex": {
|
||||
"input_cost_per_pixel": 4.76837158203125e-08,
|
||||
"input_cost_per_reference_pixel": 4.76837158203125e-08,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 32000,
|
||||
"max_tokens": 32000,
|
||||
|
|
|
|||
|
|
@ -791,7 +791,6 @@ class ModelGroupInfo(BaseModel):
|
|||
input_cost_per_token: float | None = None
|
||||
output_cost_per_token: float | None = None
|
||||
input_cost_per_pixel: float | None = None
|
||||
input_cost_per_reference_pixel: float | None = None
|
||||
mode: (
|
||||
str
|
||||
| Literal["chat", "embedding", "completion", "image_generation", "audio_transcription", "rerank", "moderations"]
|
||||
|
|
|
|||
|
|
@ -319,7 +319,6 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
input_cost_per_image_token_batches: ReadOnly[float | None]
|
||||
input_cost_per_second: float | None # for OpenAI Speech models
|
||||
input_cost_per_pixel: ReadOnly[float | None]
|
||||
input_cost_per_reference_pixel: ReadOnly[float | None]
|
||||
input_cost_per_token_batches: float | None
|
||||
input_cost_per_video_token_batches: ReadOnly[float | None]
|
||||
input_cost_per_token_above_272k_tokens_batches: ReadOnly[float | None]
|
||||
|
|
@ -3712,7 +3711,6 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
output_cost_per_image_1024: float | None = None
|
||||
output_cost_per_image_1536: float | None = None
|
||||
input_cost_per_pixel: float | None = None
|
||||
input_cost_per_reference_pixel: float | None = None
|
||||
output_cost_per_pixel: float | None = None
|
||||
|
||||
# Include all ModelInfoBase fields as optional
|
||||
|
|
|
|||
|
|
@ -6202,7 +6202,6 @@ def _get_model_info_helper(
|
|||
output_cost_per_video_per_second=_model_info.get("output_cost_per_video_per_second", None),
|
||||
output_cost_per_image=_model_info.get("output_cost_per_image", None),
|
||||
input_cost_per_pixel=_model_info.get("input_cost_per_pixel", None),
|
||||
input_cost_per_reference_pixel=_model_info.get("input_cost_per_reference_pixel", None),
|
||||
output_cost_per_pixel=_model_info.get("output_cost_per_pixel", None),
|
||||
output_cost_per_image_token=_model_info.get("output_cost_per_image_token", None),
|
||||
output_cost_per_video_token=_model_info.get("output_cost_per_video_token", None),
|
||||
|
|
|
|||
|
|
@ -10964,7 +10964,6 @@
|
|||
},
|
||||
"azure_ai/FLUX.2-flex": {
|
||||
"input_cost_per_pixel": 4.76837158203125e-08,
|
||||
"input_cost_per_reference_pixel": 4.76837158203125e-08,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 32000,
|
||||
"max_tokens": 32000,
|
||||
|
|
|
|||
|
|
@ -318,10 +318,6 @@
|
|||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"input_cost_per_reference_pixel": {
|
||||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"input_cost_per_request": {
|
||||
"type": "number",
|
||||
"minimum": 0
|
||||
|
|
|
|||
|
|
@ -157,8 +157,6 @@ The following arguments are supported:
|
|||
|
||||
* `input_cost_per_pixel` - (Optional) float. Cost applied per input pixel for models that charge by image size.
|
||||
|
||||
* `input_cost_per_reference_pixel` - (Optional) float. Cost applied per reference image pixel for image edit models that meter reference inputs.
|
||||
|
||||
* `output_cost_per_pixel` - (Optional) float. Cost applied per output pixel for image-generation models.
|
||||
|
||||
* `input_cost_per_second` - (Optional) float. Cost applied per input second for audio/transcription models.
|
||||
|
|
|
|||
|
|
@ -119,10 +119,6 @@ func resourceLiteLLMModel() *schema.Resource {
|
|||
Type: schema.TypeFloat,
|
||||
Optional: true,
|
||||
},
|
||||
"input_cost_per_reference_pixel": {
|
||||
Type: schema.TypeFloat,
|
||||
Optional: true,
|
||||
},
|
||||
"output_cost_per_pixel": {
|
||||
Type: schema.TypeFloat,
|
||||
Optional: true,
|
||||
|
|
|
|||
|
|
@ -9,7 +9,6 @@ import (
|
|||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
"github.com/hashicorp/go-cty/cty"
|
||||
"github.com/hashicorp/terraform-plugin-sdk/v2/helper/schema"
|
||||
)
|
||||
|
||||
|
|
@ -125,9 +124,6 @@ func createOrUpdateModel(d *schema.ResourceData, m interface{}, isUpdate bool) e
|
|||
if inputCostPerPixel := d.Get("input_cost_per_pixel").(float64); inputCostPerPixel > 0 {
|
||||
litellmParams["input_cost_per_pixel"] = inputCostPerPixel
|
||||
}
|
||||
if raw, err := d.GetRawConfigAt(cty.GetAttrPath("input_cost_per_reference_pixel")); err == nil && !raw.IsNull() && raw.Type() == cty.Number {
|
||||
litellmParams["input_cost_per_reference_pixel"] = d.Get("input_cost_per_reference_pixel").(float64)
|
||||
}
|
||||
if outputCostPerPixel := d.Get("output_cost_per_pixel").(float64); outputCostPerPixel > 0 {
|
||||
litellmParams["output_cost_per_pixel"] = outputCostPerPixel
|
||||
}
|
||||
|
|
|
|||
|
|
@ -93,7 +93,6 @@ type LiteLLMParams struct {
|
|||
InputCostPerToken float64 `json:"input_cost_per_token,omitempty"`
|
||||
OutputCostPerToken float64 `json:"output_cost_per_token,omitempty"`
|
||||
InputCostPerPixel float64 `json:"input_cost_per_pixel,omitempty"`
|
||||
InputCostPerReferencePixel float64 `json:"input_cost_per_reference_pixel,omitempty"`
|
||||
OutputCostPerPixel float64 `json:"output_cost_per_pixel,omitempty"`
|
||||
InputCostPerSecond float64 `json:"input_cost_per_second,omitempty"`
|
||||
OutputCostPerSecond float64 `json:"output_cost_per_second,omitempty"`
|
||||
|
|
|
|||
|
|
@ -127,7 +127,6 @@ def test_flux2_flex_model_info():
|
|||
assert model_info["max_input_tokens"] == 32000
|
||||
assert model_info["max_tokens"] == 32000
|
||||
assert model_info["supported_endpoints"] == ["/v1/images/generations", "/v1/images/edits"]
|
||||
assert catalog_info["input_cost_per_reference_pixel"] == catalog_info["input_cost_per_pixel"]
|
||||
assert catalog_info["input_cost_per_pixel"] * 1024 * 1024 == pytest.approx(0.05), (
|
||||
"Azure Retail Prices API, product 'Azure BFL Flux Models', meters 'Flex Megapixel' and "
|
||||
"'Flex Ref Megapixel' are $0.05 per MP where 1 MP = 1024x1024 pixels; confirmed against "
|
||||
|
|
|
|||
|
|
@ -40,18 +40,16 @@ def _edit_cost(response: ImageResponse, size: str = "1024x1024", **kwargs: objec
|
|||
)
|
||||
|
||||
|
||||
def test_get_model_info_surfaces_flux2_flex_pixel_rates() -> None:
|
||||
def test_get_model_info_surfaces_flux2_flex_pixel_rate() -> None:
|
||||
model_info = litellm.get_model_info(model="FLUX.2-flex", custom_llm_provider="azure_ai")
|
||||
catalog_info = litellm.model_cost["azure_ai/FLUX.2-flex"]
|
||||
|
||||
assert model_info["input_cost_per_pixel"] == catalog_info["input_cost_per_pixel"]
|
||||
assert model_info["input_cost_per_reference_pixel"] == catalog_info["input_cost_per_reference_pixel"]
|
||||
|
||||
|
||||
def test_flux2_flex_catalog_pixel_rates_match_reference_rates() -> None:
|
||||
def test_flux2_flex_catalog_pixel_rate_is_azure_megapixel_price() -> None:
|
||||
catalog_info = litellm.model_cost["azure_ai/FLUX.2-flex"]
|
||||
|
||||
assert catalog_info["input_cost_per_reference_pixel"] == catalog_info["input_cost_per_pixel"]
|
||||
assert catalog_info["input_cost_per_pixel"] * 1024 * 1024 == pytest.approx(0.05), (
|
||||
"Azure Retail Prices API, product 'Azure BFL Flux Models', meters 'Flex Megapixel' and "
|
||||
"'Flex Ref Megapixel' are $0.05 per MP where 1 MP = 1024x1024 pixels; confirmed against "
|
||||
|
|
@ -68,40 +66,24 @@ def test_edit_cost_adds_reference_pixels_to_generated_pixels() -> None:
|
|||
|
||||
|
||||
def test_edit_cost_prefers_deployment_rates_over_catalog() -> None:
|
||||
cost: Final = _edit_cost(
|
||||
_edit_response(), model_info={"input_cost_per_pixel": 2e-07, "input_cost_per_reference_pixel": 3e-07}
|
||||
)
|
||||
cost: Final = _edit_cost(_edit_response(), model_info={"input_cost_per_pixel": 2e-07})
|
||||
|
||||
assert cost == pytest.approx(2e-07 * 1024 * 1024 + 3e-07 * REFERENCE_PIXELS)
|
||||
assert cost == pytest.approx(2e-07 * (1024 * 1024 + REFERENCE_PIXELS))
|
||||
|
||||
|
||||
def test_edit_cost_honors_explicit_zero_reference_rate() -> None:
|
||||
cost: Final = _edit_cost(
|
||||
_edit_response(), model_info={"input_cost_per_pixel": 2e-07, "input_cost_per_reference_pixel": 0.0}
|
||||
)
|
||||
def test_edit_cost_honors_explicit_zero_pixel_rate() -> None:
|
||||
cost: Final = _edit_cost(_edit_response(), model_info={"input_cost_per_pixel": 0.0})
|
||||
|
||||
assert cost == pytest.approx(2e-07 * 1024 * 1024)
|
||||
|
||||
|
||||
def test_edit_cost_honors_explicit_zero_generated_rate() -> None:
|
||||
cost: Final = _edit_cost(
|
||||
_edit_response(), model_info={"input_cost_per_pixel": 0.0, "input_cost_per_reference_pixel": 1e-07}
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(1e-07 * REFERENCE_PIXELS)
|
||||
assert cost == pytest.approx(0.0)
|
||||
|
||||
|
||||
def test_per_image_rate_beats_per_pixel_rate(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
cost: Final = _edit_cost(
|
||||
_edit_response(),
|
||||
model_info={
|
||||
"output_cost_per_image": 0.04,
|
||||
"input_cost_per_pixel": 1e-07,
|
||||
"input_cost_per_reference_pixel": 1.5e-08,
|
||||
},
|
||||
model_info={"output_cost_per_image": 0.04, "input_cost_per_pixel": 1e-07},
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(0.04 + 1.5e-08 * REFERENCE_PIXELS)
|
||||
assert cost == pytest.approx(0.04 + 1e-07 * REFERENCE_PIXELS)
|
||||
|
||||
|
||||
def test_usage_short_circuits_pixel_billing() -> None:
|
||||
|
|
@ -118,7 +100,6 @@ def test_usage_short_circuits_pixel_billing() -> None:
|
|||
response,
|
||||
model_info={
|
||||
"input_cost_per_pixel": 5e-08,
|
||||
"input_cost_per_reference_pixel": 5e-08,
|
||||
"input_cost_per_token": 1e-05,
|
||||
"input_cost_per_image_token": 2e-05,
|
||||
"output_cost_per_image_token": 4e-05,
|
||||
|
|
@ -173,10 +154,10 @@ def test_unlisted_model_bills_deployment_rates() -> None:
|
|||
custom_llm_provider="azure_ai",
|
||||
size="1024x1024",
|
||||
call_type="image_edit",
|
||||
model_info={"input_cost_per_pixel": 1e-07, "input_cost_per_reference_pixel": 1e-07},
|
||||
model_info={"input_cost_per_pixel": 1e-07},
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(1e-07 * 1024 * 1024 + 1e-07 * REFERENCE_PIXELS)
|
||||
assert cost == pytest.approx(1e-07 * (1024 * 1024 + REFERENCE_PIXELS))
|
||||
|
||||
|
||||
def test_non_image_response_raises() -> None:
|
||||
|
|
|
|||
|
|
@ -229,25 +229,7 @@ def test_model_info_rejects_offset_aware_access_window_times():
|
|||
with pytest.raises(ValidationError):
|
||||
ModelInfo(
|
||||
id="x",
|
||||
access_windows=[{"start": "22:00+05:00", "end": "06:00", "timezone": "UTC", "team_ids": ["t"]}],
|
||||
access_windows=[
|
||||
{"start": "22:00+05:00", "end": "06:00", "timezone": "UTC", "team_ids": ["t"]}
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def test_model_group_info_surfaces_input_cost_per_reference_pixel():
|
||||
from typing import Final
|
||||
|
||||
from litellm import Router
|
||||
|
||||
configured_rate: Final = 3e-7
|
||||
router: Final = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "flux-edit",
|
||||
"litellm_params": {"model": "azure_ai/FLUX.2-flex"},
|
||||
"model_info": {"id": "dep-flux-1", "input_cost_per_reference_pixel": configured_rate},
|
||||
}
|
||||
]
|
||||
)
|
||||
group: Final = router.get_model_group_info("flux-edit")
|
||||
assert group is not None
|
||||
assert group.input_cost_per_reference_pixel == configured_rate
|
||||
|
|
|
|||
6
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
6
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -31409,8 +31409,6 @@ export interface components {
|
|||
input_cost_per_pixel?: number | null;
|
||||
/** Input Cost Per Query */
|
||||
input_cost_per_query?: number | null;
|
||||
/** Input Cost Per Reference Pixel */
|
||||
input_cost_per_reference_pixel?: number | null;
|
||||
/** Input Cost Per Second */
|
||||
input_cost_per_second?: number | null;
|
||||
/** Input Cost Per Token */
|
||||
|
|
@ -33950,8 +33948,6 @@ export interface components {
|
|||
health_status?: string | null;
|
||||
/** Input Cost Per Pixel */
|
||||
input_cost_per_pixel?: number | null;
|
||||
/** Input Cost Per Reference Pixel */
|
||||
input_cost_per_reference_pixel?: number | null;
|
||||
/** Input Cost Per Token */
|
||||
input_cost_per_token?: number | null;
|
||||
/**
|
||||
|
|
@ -42308,8 +42304,6 @@ export interface components {
|
|||
input_cost_per_pixel?: number | null;
|
||||
/** Input Cost Per Query */
|
||||
input_cost_per_query?: number | null;
|
||||
/** Input Cost Per Reference Pixel */
|
||||
input_cost_per_reference_pixel?: number | null;
|
||||
/** Input Cost Per Second */
|
||||
input_cost_per_second?: number | null;
|
||||
/** Input Cost Per Token */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue