diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md index d959498ce76..fa18525ee19 100644 --- a/docs/my-website/docs/providers/vertex.md +++ b/docs/my-website/docs/providers/vertex.md @@ -152,8 +152,14 @@ LiteLLM Supports the following image types passed in `url` - Images with Cloud Storage URIs - gs://cloud-samples-data/generative-ai/image/boats.jpeg - Images with direct links - https://storage.googleapis.com/github-repo/img/gemini/intro/landmark3.jpg - Videos with Cloud Storage URIs - https://storage.googleapis.com/github-repo/img/gemini/multimodality_usecases_overview/pixel8.mp4 +- Base64 Encoded Local Images + +**Example Request - image url** + + + + -**Example Request** ```python import litellm @@ -179,6 +185,43 @@ response = litellm.completion( ) print(response) ``` + + + + +```python +import litellm + +def encode_image(image_path): + import base64 + + with open(image_path, "rb") as image_file: + return base64.b64encode(image_file.read()).decode("utf-8") + +image_path = "cached_logo.jpg" +# Getting the base64 string +base64_image = encode_image(image_path) +response = litellm.completion( + model="vertex_ai/gemini-pro-vision", + messages=[ + { + "role": "user", + "content": [ + {"type": "text", "text": "Whats in this image?"}, + { + "type": "image_url", + "image_url": { + "url": "data:image/jpeg;base64," + base64_image + }, + }, + ], + } + ], +) +print(response) +``` + + ## Chat Models diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index d69abcbfbde..f1c69f53a62 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -122,8 +122,6 @@ const sidebars = { "providers/openai_compatible", "providers/azure", "providers/azure_ai", - "providers/huggingface", - "providers/ollama", "providers/vertex", "providers/palm", "providers/gemini", @@ -132,6 +130,8 @@ const sidebars = { "providers/aws_sagemaker", "providers/bedrock", "providers/anyscale", + "providers/huggingface", + "providers/ollama", "providers/perplexity", "providers/groq", "providers/vllm", diff --git a/litellm/llms/vertex_ai.py b/litellm/llms/vertex_ai.py index 18c06d4d677..1cda0ccdba4 100644 --- a/litellm/llms/vertex_ai.py +++ b/litellm/llms/vertex_ai.py @@ -225,6 +225,24 @@ def _gemini_vision_convert_messages(messages: list): part_mime = "video/mp4" google_clooud_part = Part.from_uri(img, mime_type=part_mime) processed_images.append(google_clooud_part) + elif "base64" in img: + # Case 4: Images with base64 encoding + import base64, re + + # base 64 is passed as data:image/jpeg;base64, + image_metadata, img_without_base_64 = img.split(",") + + # read mime_type from img_without_base_64=data:image/jpeg;base64 + # Extract MIME type using regular expression + mime_type_match = re.match(r"data:(.*?);base64", image_metadata) + + if mime_type_match: + mime_type = mime_type_match.group(1) + else: + mime_type = "image/jpeg" + decoded_img = base64.b64decode(img_without_base_64) + processed_image = Part.from_data(data=decoded_img, mime_type=mime_type) + processed_images.append(processed_image) return prompt, processed_images except Exception as e: raise e diff --git a/litellm/tests/test_amazing_vertex_completion.py b/litellm/tests/test_amazing_vertex_completion.py index f5df00c8d9b..35d66907c34 100644 --- a/litellm/tests/test_amazing_vertex_completion.py +++ b/litellm/tests/test_amazing_vertex_completion.py @@ -336,6 +336,52 @@ def test_gemini_pro_vision(): # test_gemini_pro_vision() +def encode_image(image_path): + import base64 + + with open(image_path, "rb") as image_file: + return base64.b64encode(image_file.read()).decode("utf-8") + + +@pytest.mark.skip( + reason="we already test gemini-pro-vision, this is just another way to pass images" +) +def test_gemini_pro_vision_base64(): + try: + load_vertex_ai_credentials() + litellm.set_verbose = True + litellm.num_retries = 3 + image_path = "cached_logo.jpg" + # Getting the base64 string + base64_image = encode_image(image_path) + resp = litellm.completion( + model="vertex_ai/gemini-pro-vision", + messages=[ + { + "role": "user", + "content": [ + {"type": "text", "text": "Whats in this image?"}, + { + "type": "image_url", + "image_url": { + "url": "data:image/jpeg;base64," + base64_image + }, + }, + ], + } + ], + ) + print(resp) + + prompt_tokens = resp.usage.prompt_tokens + + except Exception as e: + if "500 Internal error encountered.'" in str(e): + pass + else: + pytest.fail(f"An exception occurred - {str(e)}") + + def test_gemini_pro_function_calling(): load_vertex_ai_credentials() tools = [