mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge pull request #19323 from BerriAI/litellm_fix_stability_issues12
[Fix] Bedrock stability model usage issues
This commit is contained in:
commit
9e405ce6cc
14 changed files with 180 additions and 97 deletions
|
|
@ -173,6 +173,14 @@ Stability AI returns images in base64 format. The response is OpenAI-compatible:
|
|||
|
||||
Stability AI supports various image editing operations including inpainting, upscaling, outpainting, background removal, and more.
|
||||
|
||||
:::info Optional Parameters
|
||||
**Important:** Different Stability models have different parameter requirements:
|
||||
- Some models don't require a `prompt` (e.g., upscaling, background removal)
|
||||
- The `style-transfer` model uses `init_image` and `style_image` instead of `image`
|
||||
- The `outpaint` model requires numeric parameters (`left`, `right`, `up`, `down`)
|
||||
LiteLLM automatically handles these differences for you.
|
||||
:::
|
||||
|
||||
### Usage - LiteLLM Python SDK
|
||||
|
||||
#### Inpainting (Edit with Mask)
|
||||
|
|
@ -217,11 +225,11 @@ response = image_edit(
|
|||
creativity=0.3, # 0-0.35, higher = more creative
|
||||
)
|
||||
|
||||
# Fast upscaling - quick upscaling
|
||||
# Fast upscaling - quick upscaling (no prompt needed)
|
||||
response = image_edit(
|
||||
model="stability/stable-fast-upscale-v1:0",
|
||||
image=open("low_res_image.png", "rb"),
|
||||
prompt="Quickly upscale this image",
|
||||
# No prompt required for fast upscale
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
|
@ -259,7 +267,7 @@ os.environ['STABILITY_API_KEY'] = "your-api-key"
|
|||
response = image_edit(
|
||||
model="stability/stable-image-remove-background-v1:0",
|
||||
image=open("portrait.png", "rb"),
|
||||
prompt="Remove the background",
|
||||
# No prompt required for fast upscale
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
|
@ -329,10 +337,29 @@ response = image_edit(
|
|||
model="stability/stable-image-erase-object-v1:0",
|
||||
image=open("scene.png", "rb"),
|
||||
mask=open("object_mask.png", "rb"), # Mask the object to erase
|
||||
prompt="Remove the object",
|
||||
# No prompt needed
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
#### Style Transfer
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import image_edit
|
||||
import os
|
||||
|
||||
os.environ['STABILITY_API_KEY'] = "your-api-key"
|
||||
|
||||
# Transfer style from one image to another
|
||||
# Note: Uses init_image (via image param) and style_image
|
||||
response = image_edit(
|
||||
model="stability/stable-style-transfer-v1:0",
|
||||
image=open("content_image.png", "rb"), # Maps to init_image
|
||||
style_image=open("style_reference.png", "rb"), # Style to apply
|
||||
fidelity=0.5, # 0-1, balance between content and style
|
||||
# No prompt needed
|
||||
)
|
||||
|
||||
print(response)
|
||||
|
||||
### Supported Image Edit Models
|
||||
|
||||
|
|
@ -419,6 +446,23 @@ response = image_edit(
|
|||
)
|
||||
print(response)
|
||||
```
|
||||
# Fast upscale without prompt
|
||||
response = image_edit(
|
||||
model="bedrock/stability.stable-fast-upscale-v1:0",
|
||||
image=open("low_res_image.png", "rb"),
|
||||
)
|
||||
|
||||
# Outpaint with numeric parameters
|
||||
response = image_edit(
|
||||
model="bedrock/stability.stable-outpaint-v1:0",
|
||||
image=open("original_image.png", "rb"),
|
||||
left=100, # Automatically converted to int
|
||||
right=100,
|
||||
up=50,
|
||||
down=50,
|
||||
)
|
||||
|
||||
print(response)
|
||||
|
||||
### Supported Bedrock Stability Models
|
||||
|
||||
|
|
|
|||
|
|
@ -714,8 +714,8 @@ def image_variation(
|
|||
|
||||
@client
|
||||
def image_edit( # noqa: PLR0915
|
||||
image: Union[FileTypes, List[FileTypes]],
|
||||
prompt: str,
|
||||
image: Optional[Union[FileTypes, List[FileTypes]]] = None,
|
||||
prompt: Optional[str]= None,
|
||||
model: Optional[str] = None,
|
||||
mask: Optional[str] = None,
|
||||
n: Optional[int] = None,
|
||||
|
|
@ -766,7 +766,7 @@ def image_edit( # noqa: PLR0915
|
|||
_is_async = kwargs.pop("async_call", False) is True
|
||||
|
||||
# add images / or return a single image
|
||||
images = image if isinstance(image, list) else [image]
|
||||
images = image if isinstance(image, list) else ([image] if image is not None else [])
|
||||
|
||||
headers_from_kwargs = kwargs.get("headers")
|
||||
merged_extra_headers: Dict[str, Any] = {}
|
||||
|
|
|
|||
|
|
@ -88,7 +88,7 @@ class AzureFoundryFlux2ImageEditConfig(OpenAIImageEditConfig):
|
|||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict,
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
|
|
@ -102,6 +102,9 @@ class AzureFoundryFlux2ImageEditConfig(OpenAIImageEditConfig):
|
|||
if prompt is None:
|
||||
raise ValueError("FLUX 2 image edit requires a prompt.")
|
||||
|
||||
if image is None:
|
||||
raise ValueError("FLUX 2 image edit requires an image.")
|
||||
|
||||
image_b64 = self._convert_image_to_base64(image)
|
||||
|
||||
# Build request body with required params
|
||||
|
|
|
|||
|
|
@ -93,7 +93,7 @@ class BaseImageEditConfig(ABC):
|
|||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict,
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
|
|
|
|||
|
|
@ -62,7 +62,7 @@ class BedrockImageEdit(BaseAWSLLM):
|
|||
self,
|
||||
model: str,
|
||||
image: list,
|
||||
prompt: str,
|
||||
prompt: Optional[str],
|
||||
model_response: ImageResponse,
|
||||
optional_params: dict,
|
||||
logging_obj: LitellmLogging,
|
||||
|
|
@ -127,7 +127,7 @@ class BedrockImageEdit(BaseAWSLLM):
|
|||
timeout: Optional[Union[float, httpx.Timeout]],
|
||||
model: str,
|
||||
logging_obj: LitellmLogging,
|
||||
prompt: str,
|
||||
prompt: Optional[str],
|
||||
model_response: ImageResponse,
|
||||
client: Optional[AsyncHTTPHandler] = None,
|
||||
) -> ImageResponse:
|
||||
|
|
@ -163,7 +163,7 @@ class BedrockImageEdit(BaseAWSLLM):
|
|||
self,
|
||||
model: str,
|
||||
image: list,
|
||||
prompt: str,
|
||||
prompt: Optional[str],
|
||||
optional_params: dict,
|
||||
api_base: Optional[str],
|
||||
extra_headers: Optional[dict],
|
||||
|
|
@ -176,7 +176,7 @@ class BedrockImageEdit(BaseAWSLLM):
|
|||
Args:
|
||||
model (str): The model to use for the image edit
|
||||
image (list): The images to edit
|
||||
prompt (str): The prompt for the edit
|
||||
prompt (Optional[str]): The prompt for the edit
|
||||
optional_params (dict): The optional parameters for the image edit
|
||||
api_base (Optional[str]): The base URL for the Bedrock API
|
||||
extra_headers (Optional[dict]): The extra headers to include in the request
|
||||
|
|
@ -248,7 +248,7 @@ class BedrockImageEdit(BaseAWSLLM):
|
|||
self,
|
||||
model: str,
|
||||
image: list,
|
||||
prompt: str,
|
||||
prompt: Optional[str],
|
||||
optional_params: dict,
|
||||
) -> dict:
|
||||
"""
|
||||
|
|
@ -276,7 +276,7 @@ class BedrockImageEdit(BaseAWSLLM):
|
|||
model_response: ImageResponse,
|
||||
model: str,
|
||||
logging_obj: LitellmLogging,
|
||||
prompt: str,
|
||||
prompt: Optional[str],
|
||||
response: httpx.Response,
|
||||
data: dict,
|
||||
) -> ImageResponse:
|
||||
|
|
|
|||
|
|
@ -150,11 +150,11 @@ class BedrockStabilityImageEditConfig(BaseImageEditConfig):
|
|||
|
||||
return mapped_params
|
||||
|
||||
def transform_image_edit_request(
|
||||
def transform_image_edit_request( #noqa: PLR0915
|
||||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict,
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
|
|
@ -164,32 +164,38 @@ class BedrockStabilityImageEditConfig(BaseImageEditConfig):
|
|||
|
||||
Returns the request body dict that will be JSON-encoded by the handler.
|
||||
"""
|
||||
if prompt is None:
|
||||
raise ValueError("Bedrock Stability image edit requires a prompt.")
|
||||
|
||||
# Build Bedrock Stability request
|
||||
data: Dict[str, Any] = {
|
||||
"prompt": prompt,
|
||||
"output_format": "png", # Default to PNG
|
||||
}
|
||||
|
||||
# Convert image to base64
|
||||
image_b64: str
|
||||
if hasattr(image, 'read') and callable(getattr(image, 'read', None)):
|
||||
# File-like object (e.g., BufferedReader from open())
|
||||
image_bytes = image.read() # type: ignore
|
||||
image_b64 = base64.b64encode(image_bytes).decode('utf-8') # type: ignore
|
||||
elif isinstance(image, bytes):
|
||||
# Raw bytes
|
||||
image_b64 = base64.b64encode(image).decode('utf-8')
|
||||
elif isinstance(image, str):
|
||||
# Already a base64 string
|
||||
image_b64 = image
|
||||
else:
|
||||
# Try to handle as bytes
|
||||
image_b64 = base64.b64encode(bytes(image)).decode('utf-8') # type: ignore
|
||||
# Add prompt only if provided (some models don't require it)
|
||||
if prompt is not None and prompt != "":
|
||||
data["prompt"] = prompt
|
||||
|
||||
# Convert image to base64 if provided
|
||||
if image is not None:
|
||||
image_b64: str
|
||||
if hasattr(image, 'read') and callable(getattr(image, 'read', None)):
|
||||
# File-like object (e.g., BufferedReader from open())
|
||||
image_bytes = image.read() # type: ignore
|
||||
image_b64 = base64.b64encode(image_bytes).decode('utf-8') # type: ignore
|
||||
elif isinstance(image, bytes):
|
||||
# Raw bytes
|
||||
image_b64 = base64.b64encode(image).decode('utf-8')
|
||||
elif isinstance(image, str):
|
||||
# Already a base64 string
|
||||
image_b64 = image
|
||||
else:
|
||||
# Try to handle as bytes
|
||||
image_b64 = base64.b64encode(bytes(image)).decode('utf-8') # type: ignore
|
||||
|
||||
data["image"] = image_b64
|
||||
# For style-transfer models, map image to init_image
|
||||
model_lower = model.lower()
|
||||
if "style-transfer" in model_lower:
|
||||
data["init_image"] = image_b64
|
||||
else:
|
||||
data["image"] = image_b64
|
||||
|
||||
# Add optional params (already mapped in map_openai_params)
|
||||
for key, value in image_edit_optional_request_params.items(): # type: ignore
|
||||
|
|
@ -221,30 +227,43 @@ class BedrockStabilityImageEditConfig(BaseImageEditConfig):
|
|||
file_b64 = str(file_bytes)
|
||||
data[key] = file_b64
|
||||
continue
|
||||
|
||||
# Supported text fields
|
||||
if key in [
|
||||
"negative_prompt",
|
||||
"aspect_ratio",
|
||||
"seed",
|
||||
"output_format",
|
||||
"model",
|
||||
"mode",
|
||||
|
||||
# Numeric fields that need to be converted to int/float
|
||||
numeric_int_fields = ["left", "right", "up", "down", "seed"]
|
||||
numeric_float_fields = [
|
||||
"strength",
|
||||
"style_preset",
|
||||
"creativity",
|
||||
"control_strength",
|
||||
"grow_mask",
|
||||
"left",
|
||||
"right",
|
||||
"up",
|
||||
"down",
|
||||
"select_prompt",
|
||||
"search_prompt",
|
||||
"fidelity",
|
||||
"composition_fidelity",
|
||||
"style_strength",
|
||||
"change_strength",
|
||||
]
|
||||
|
||||
if key in numeric_int_fields:
|
||||
# Convert to int (these are pixel values for outpaint)
|
||||
try:
|
||||
data[key] = int(value) # type: ignore
|
||||
except (ValueError, TypeError):
|
||||
data[key] = value # type: ignore
|
||||
elif key in numeric_float_fields:
|
||||
# Convert to float
|
||||
try:
|
||||
data[key] = float(value) # type: ignore
|
||||
except (ValueError, TypeError):
|
||||
data[key] = value # type: ignore
|
||||
|
||||
# Supported text fields
|
||||
elif key in [
|
||||
"negative_prompt",
|
||||
"aspect_ratio",
|
||||
"output_format",
|
||||
"model",
|
||||
"mode",
|
||||
"style_preset",
|
||||
"select_prompt",
|
||||
"search_prompt",
|
||||
]:
|
||||
data[key] = value # type: ignore
|
||||
|
||||
|
|
|
|||
|
|
@ -81,21 +81,23 @@ class GeminiImageEditConfig(BaseImageEditConfig):
|
|||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict[str, Any],
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
) -> Tuple[Dict[str, Any], Optional[RequestFiles]]:
|
||||
inline_parts = self._prepare_inline_image_parts(image)
|
||||
inline_parts = self._prepare_inline_image_parts(image) if image else []
|
||||
if not inline_parts:
|
||||
raise ValueError("Gemini image edit requires at least one image.")
|
||||
|
||||
if prompt is None:
|
||||
raise ValueError("Gemini image edit requires a prompt.")
|
||||
# Build parts list with image and prompt (if provided)
|
||||
parts = inline_parts.copy()
|
||||
if prompt is not None and prompt != "":
|
||||
parts.append({"text": prompt})
|
||||
|
||||
contents = [
|
||||
{
|
||||
"parts": inline_parts + [{"text": prompt}],
|
||||
"parts": parts,
|
||||
}
|
||||
]
|
||||
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@ class DallE2ImageEditConfig(OpenAIImageEditConfig):
|
|||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict,
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
|
|
@ -40,18 +40,20 @@ class DallE2ImageEditConfig(OpenAIImageEditConfig):
|
|||
Transform image edit request for DALL-E-2.
|
||||
|
||||
DALL-E-2 only accepts a single image with field name "image" (not "image[]").
|
||||
"""
|
||||
if prompt is None:
|
||||
raise ValueError("DALL-E-2 image edit requires a prompt.")
|
||||
|
||||
request = ImageEditRequestParams(
|
||||
model=model,
|
||||
image=image,
|
||||
prompt=prompt,
|
||||
"""
|
||||
request_params = {
|
||||
"model": model,
|
||||
**image_edit_optional_request_params,
|
||||
)
|
||||
}
|
||||
if image is not None:
|
||||
request_params["image"] = image
|
||||
if prompt is not None:
|
||||
request_params["prompt"] = prompt
|
||||
|
||||
request = ImageEditRequestParams(**request_params)
|
||||
request_dict = cast(Dict, request)
|
||||
|
||||
|
||||
#########################################################
|
||||
# Separate images and masks as `files` and send other parameters as `data`
|
||||
#########################################################
|
||||
|
|
|
|||
|
|
@ -80,7 +80,7 @@ class OpenAIImageEditConfig(BaseImageEditConfig):
|
|||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict,
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
|
|
@ -91,15 +91,17 @@ class OpenAIImageEditConfig(BaseImageEditConfig):
|
|||
Handles multipart/form-data for images. Uses "image[]" field name
|
||||
to support multiple images (e.g., for gpt-image-1).
|
||||
"""
|
||||
if prompt is None:
|
||||
raise ValueError("OpenAI image edit requires a prompt.")
|
||||
|
||||
request = ImageEditRequestParams(
|
||||
model=model,
|
||||
image=image,
|
||||
prompt=prompt,
|
||||
# Build request params, only including non-None values
|
||||
request_params = {
|
||||
"model": model,
|
||||
**image_edit_optional_request_params,
|
||||
)
|
||||
}
|
||||
if image is not None:
|
||||
request_params["image"] = image
|
||||
if prompt is not None:
|
||||
request_params["prompt"] = prompt
|
||||
|
||||
request = ImageEditRequestParams(**request_params)
|
||||
request_dict = cast(Dict, request)
|
||||
|
||||
#########################################################
|
||||
|
|
|
|||
|
|
@ -102,7 +102,7 @@ class RecraftImageEditConfig(BaseImageEditConfig):
|
|||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict,
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
|
|
@ -114,15 +114,15 @@ class RecraftImageEditConfig(BaseImageEditConfig):
|
|||
https://www.recraft.ai/docs#image-to-image
|
||||
"""
|
||||
|
||||
if prompt is None:
|
||||
raise ValueError("Recraft image edit requires a prompt.")
|
||||
|
||||
request_body: RecraftImageEditRequestParams = RecraftImageEditRequestParams(
|
||||
model=model,
|
||||
prompt=prompt,
|
||||
strength=image_edit_optional_request_params.pop("strength", self.DEFAULT_STRENGTH),
|
||||
request_params = {
|
||||
"model": model,
|
||||
"strength": image_edit_optional_request_params.pop("strength", self.DEFAULT_STRENGTH),
|
||||
**image_edit_optional_request_params,
|
||||
)
|
||||
}
|
||||
if prompt is not None:
|
||||
request_params["prompt"] = prompt
|
||||
|
||||
request_body = RecraftImageEditRequestParams(**request_params)
|
||||
request_dict = cast(Dict, request_body)
|
||||
#########################################################
|
||||
# Reuse OpenAI logic: Separate images as `files` and send other parameters as `data`
|
||||
|
|
|
|||
|
|
@ -171,7 +171,7 @@ class StabilityImageEditConfig(BaseImageEditConfig):
|
|||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict,
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
|
|
@ -190,11 +190,14 @@ class StabilityImageEditConfig(BaseImageEditConfig):
|
|||
}
|
||||
|
||||
# Add prompt only if provided (some Stability endpoints don't require it)
|
||||
if prompt is not None:
|
||||
if prompt is not None and prompt != "":
|
||||
data["prompt"] = prompt
|
||||
# Handle image parameter - could be a single file or list
|
||||
image_file = image[0] if isinstance(image, list) else image # type: ignore
|
||||
files: Dict[str, Any] = {"image": image_file}
|
||||
files: Dict[str, Any] = {}
|
||||
if image is not None:
|
||||
image_file = image[0] if isinstance(image, list) else image # type: ignore
|
||||
files["image"] = image_file
|
||||
|
||||
# Add optional params (already mapped in map_openai_params)
|
||||
for key, value in image_edit_optional_request_params.items(): # type: ignore
|
||||
|
|
|
|||
|
|
@ -152,22 +152,24 @@ class VertexAIGeminiImageEditConfig(BaseImageEditConfig, VertexLLM):
|
|||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict[str, Any],
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
) -> Tuple[Dict[str, Any], Optional[RequestFiles]]:
|
||||
inline_parts = self._prepare_inline_image_parts(image)
|
||||
inline_parts = self._prepare_inline_image_parts(image) if image else []
|
||||
if not inline_parts:
|
||||
raise ValueError("Vertex AI Gemini image edit requires at least one image.")
|
||||
|
||||
if prompt is None:
|
||||
raise ValueError("Vertex AI Gemini image edit requires a prompt.")
|
||||
# Build parts list with image and prompt (if provided)
|
||||
parts = inline_parts.copy()
|
||||
if prompt is not None and prompt != "":
|
||||
parts.append({"text": prompt})
|
||||
|
||||
# Correct format for Vertex AI Gemini image editing
|
||||
contents = {
|
||||
"role": "USER",
|
||||
"parts": inline_parts + [{"text": prompt}]
|
||||
"parts": parts
|
||||
}
|
||||
|
||||
request_body: Dict[str, Any] = {"contents": contents}
|
||||
|
|
|
|||
|
|
@ -144,7 +144,7 @@ class VertexAIImagenImageEditConfig(BaseImageEditConfig, VertexLLM):
|
|||
self,
|
||||
model: str,
|
||||
prompt: Optional[str],
|
||||
image: FileTypes,
|
||||
image: Optional[FileTypes],
|
||||
image_edit_optional_request_params: Dict[str, Any],
|
||||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
|
|
|
|||
|
|
@ -244,8 +244,10 @@ async def image_edit_api(
|
|||
if mask is None and mask_array is not None:
|
||||
mask = mask_array
|
||||
|
||||
if image is None:
|
||||
raise HTTPException(status_code=422, detail="Field required: image")
|
||||
# if image is None:
|
||||
# raise HTTPException(status_code=422, detail="Field required: image")
|
||||
# Note: Image is optional for some models (e.g., Bedrock Stability style-transfer)
|
||||
# The validation will be done at the model level if image is truly required
|
||||
|
||||
from litellm.proxy.proxy_server import (
|
||||
_read_request_body,
|
||||
|
|
@ -272,6 +274,10 @@ async def image_edit_api(
|
|||
data["image"] = image_files
|
||||
if mask_files:
|
||||
data["mask"] = mask_files
|
||||
|
||||
# Ensure prompt exists in data (default to None for models that don't require it)
|
||||
if "prompt" not in data:
|
||||
data["prompt"] = None
|
||||
|
||||
data["model"] = (
|
||||
model
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue