diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md index fcdd1193c09..95dac9fc09e 100644 --- a/docs/my-website/docs/providers/vertex.md +++ b/docs/my-website/docs/providers/vertex.md @@ -1450,7 +1450,7 @@ curl http://0.0.0.0:4000/v1/chat/completions \ | code-gecko@latest| `completion('code-gecko@latest', messages)` | -## Embedding Models +## **Embedding Models** #### Usage - Embedding ```python @@ -1504,7 +1504,158 @@ response = litellm.embedding( ) ``` -## Image Generation Models +## **Multi-Modal Embeddings** + +Usage + + + + +```python +response = await litellm.aembedding( + model="vertex_ai/multimodalembedding@001", + input=[ + { + "image": { + "gcsUri": "gs://cloud-samples-data/vertex-ai/llm/prompts/landmark1.png" + }, + "text": "this is a unicorn", + }, + ], +) +``` + + + + +1. Add model to config.yaml +```yaml +model_list: + - model_name: multimodalembedding@001 + litellm_params: + model: vertex_ai/multimodalembedding@001 + vertex_project: "adroit-crow-413218" + vertex_location: "us-central1" + vertex_credentials: adroit-crow-413218-a956eef1a2a8.json + +litellm_settings: + drop_params: True +``` + +2. Start Proxy + +``` +$ litellm --config /path/to/config.yaml +``` + +3. Make Request use OpenAI Python SDK + +```python +import openai + +client = openai.OpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000") + +# # request sent to model set on litellm proxy, `litellm --model` +response = client.embeddings.create( + model="multimodalembedding@001", + input = None, + extra_body = { + "instances": [ + { + "image": { + "gcsUri": "gs://cloud-samples-data/vertex-ai/llm/prompts/landmark1.png" + }, + "text": "this is a unicorn", + }, + ], + } +) + +print(response) +``` + + + + +1. Add model to config.yaml +```yaml +default_vertex_config: + vertex_project: "adroit-crow-413218" + vertex_location: "us-central1" + vertex_credentials: adroit-crow-413218-a956eef1a2a8.json +``` + +2. Start Proxy + +``` +$ litellm --config /path/to/config.yaml +``` + +3. Make Request use OpenAI Python SDK + +```python +import vertexai + +from vertexai.vision_models import Image, MultiModalEmbeddingModel, Video +from vertexai.vision_models import VideoSegmentConfig +from google.auth.credentials import Credentials + + +LITELLM_PROXY_API_KEY = "sk-1234" +LITELLM_PROXY_BASE = "http://0.0.0.0:4000/vertex-ai" + +import datetime + +class CredentialsWrapper(Credentials): + def __init__(self, token=None): + super().__init__() + self.token = token + self.expiry = None # or set to a future date if needed + + def refresh(self, request): + pass + + def apply(self, headers, token=None): + headers['Authorization'] = f'Bearer {self.token}' + + @property + def expired(self): + return False # Always consider the token as non-expired + + @property + def valid(self): + return True # Always consider the credentials as valid + +credentials = CredentialsWrapper(token=LITELLM_PROXY_API_KEY) + +vertexai.init( + project="adroit-crow-413218", + location="us-central1", + api_endpoint=LITELLM_PROXY_BASE, + credentials = credentials, + api_transport="rest", + request_metadata=[("Authorization", f"Bearer {LITELLM_PROXY_API_KEY}")], +) + +model = MultiModalEmbeddingModel.from_pretrained("multimodalembedding") +image = Image.load_from_file( + "gs://cloud-samples-data/vertex-ai/llm/prompts/landmark1.png" +) + +embeddings = model.get_embeddings( + image=image, + contextual_text="Colosseum", + dimension=1408, +) +print(f"Image Embedding: {embeddings.image_embedding}") +print(f"Text Embedding: {embeddings.text_embedding}") +``` + + + + + +## **Image Generation Models** Usage diff --git a/litellm/proxy/hooks/parallel_request_limiter.py b/litellm/proxy/hooks/parallel_request_limiter.py index 38b57c19eab..08baf78d4be 100644 --- a/litellm/proxy/hooks/parallel_request_limiter.py +++ b/litellm/proxy/hooks/parallel_request_limiter.py @@ -120,6 +120,8 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger): max_parallel_requests = user_api_key_dict.max_parallel_requests if max_parallel_requests is None: max_parallel_requests = sys.maxsize + if data is None: + data = {} global_max_parallel_requests = data.get("metadata", {}).get( "global_max_parallel_requests", None ) diff --git a/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py b/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py index b50fbb0c5cc..b9ab7526a3c 100644 --- a/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py +++ b/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py @@ -301,16 +301,19 @@ async def pass_through_request( request=request, headers=headers, forward_headers=forward_headers ) + _parsed_body = None if custom_body: _parsed_body = custom_body else: request_body = await request.body() - body_str = request_body.decode() - try: - _parsed_body = ast.literal_eval(body_str) - except Exception: - _parsed_body = json.loads(body_str) - + if request_body == b"" or request_body is None: + _parsed_body = None + else: + body_str = request_body.decode() + try: + _parsed_body = ast.literal_eval(body_str) + except Exception: + _parsed_body = json.loads(body_str) verbose_proxy_logger.debug( "Pass through endpoint sending request to \nURL {}\nheaders: {}\nbody: {}\n".format( url, headers, _parsed_body @@ -320,7 +323,7 @@ async def pass_through_request( ### CALL HOOKS ### - modify incoming data / reject request before calling the model _parsed_body = await proxy_logging_obj.pre_call_hook( user_api_key_dict=user_api_key_dict, - data=_parsed_body, + data=_parsed_body or {}, call_type="pass_through_endpoint", ) @@ -360,15 +363,24 @@ async def pass_through_request( # combine url with query params for logging - requested_query_params = query_params or request.query_params.__dict__ - requested_query_params_str = "&".join( - f"{k}={v}" for k, v in requested_query_params.items() + requested_query_params: Optional[dict] = ( + query_params or request.query_params.__dict__ ) + if requested_query_params == request.query_params.__dict__: + requested_query_params = None - if "?" in str(url): - logging_url = str(url) + "&" + requested_query_params_str - else: - logging_url = str(url) + "?" + requested_query_params_str + requested_query_params_str = None + if requested_query_params: + requested_query_params_str = "&".join( + f"{k}={v}" for k, v in requested_query_params.items() + ) + + logging_url = str(url) + if requested_query_params_str: + if "?" in str(url): + logging_url = str(url) + "&" + requested_query_params_str + else: + logging_url = str(url) + "?" + requested_query_params_str logging_obj.pre_call( input=[{"role": "user", "content": "no-message-pass-through-endpoint"}], @@ -409,6 +421,14 @@ async def pass_through_request( status_code=response.status_code, ) + verbose_proxy_logger.debug("request method: {}".format(request.method)) + verbose_proxy_logger.debug("request url: {}".format(url)) + verbose_proxy_logger.debug("request headers: {}".format(headers)) + verbose_proxy_logger.debug( + "requested_query_params={}".format(requested_query_params) + ) + verbose_proxy_logger.debug("request body: {}".format(_parsed_body)) + response = await async_client.request( method=request.method, url=url, diff --git a/litellm/proxy/proxy_config.yaml b/litellm/proxy/proxy_config.yaml index 3c61b30cc65..5d8e221442b 100644 --- a/litellm/proxy/proxy_config.yaml +++ b/litellm/proxy/proxy_config.yaml @@ -1,20 +1,13 @@ model_list: - - model_name: fake-openai-endpoint + - model_name: multimodalembedding@001 litellm_params: - model: openai/fake - api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ - - model_name: openai-embedding - litellm_params: - model: openai/text-embedding-3-small - api_key: os.environ/OPENAI_API_KEY -litellm_settings: - set_verbose: True - cache: True # set cache responses to True, litellm defaults to using a redis cache - cache_params: - type: qdrant-semantic - qdrant_semantic_cache_embedding_model: openai-embedding - qdrant_collection_name: test_collection - qdrant_quantization_config: binary - similarity_threshold: 0.8 # similarity threshold for semantic cache \ No newline at end of file + model: vertex_ai/multimodalembedding@001 + vertex_project: "adroit-crow-413218" + vertex_location: "us-central1" + vertex_credentials: adroit-crow-413218-a956eef1a2a8.json + +default_vertex_config: + vertex_project: "adroit-crow-413218" + vertex_location: "us-central1" + vertex_credentials: adroit-crow-413218-a956eef1a2a8.json diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 2cfc2186154..d07f63f53a6 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -3484,6 +3484,11 @@ async def embeddings( body = await request.body() data = orjson.loads(body) + verbose_proxy_logger.debug( + "Request received by LiteLLM:\n%s", + json.dumps(data, indent=4), + ) + # Include original request and headers in the data data = await add_litellm_data_to_request( data=data, diff --git a/litellm/proxy/tests/test_vtx_embedding.py b/litellm/proxy/tests/test_vtx_embedding.py new file mode 100644 index 00000000000..4c770ae2e9d --- /dev/null +++ b/litellm/proxy/tests/test_vtx_embedding.py @@ -0,0 +1,21 @@ +import openai + +client = openai.OpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000") + +# # request sent to model set on litellm proxy, `litellm --model` +response = client.embeddings.create( + model="multimodalembedding@001", + input=[], + extra_body={ + "instances": [ + { + "image": { + "gcsUri": "gs://cloud-samples-data/vertex-ai/llm/prompts/landmark1.png" + }, + "text": "this is a unicorn", + }, + ], + }, +) + +print(response) diff --git a/litellm/proxy/tests/test_vtx_sdk_embedding.py b/litellm/proxy/tests/test_vtx_sdk_embedding.py new file mode 100644 index 00000000000..a6468884f9a --- /dev/null +++ b/litellm/proxy/tests/test_vtx_sdk_embedding.py @@ -0,0 +1,59 @@ +import vertexai +from google.auth.credentials import Credentials +from vertexai.vision_models import ( + Image, + MultiModalEmbeddingModel, + Video, + VideoSegmentConfig, +) + +LITELLM_PROXY_API_KEY = "sk-1234" +LITELLM_PROXY_BASE = "http://0.0.0.0:4000/vertex-ai" + +import datetime + + +class CredentialsWrapper(Credentials): + def __init__(self, token=None): + super().__init__() + self.token = token + self.expiry = None # or set to a future date if needed + + def refresh(self, request): + pass + + def apply(self, headers, token=None): + headers["Authorization"] = f"Bearer {self.token}" + + @property + def expired(self): + return False # Always consider the token as non-expired + + @property + def valid(self): + return True # Always consider the credentials as valid + + +credentials = CredentialsWrapper(token=LITELLM_PROXY_API_KEY) + +vertexai.init( + project="adroit-crow-413218", + location="us-central1", + api_endpoint=LITELLM_PROXY_BASE, + credentials=credentials, + api_transport="rest", + request_metadata=[("Authorization", f"Bearer {LITELLM_PROXY_API_KEY}")], +) + +model = MultiModalEmbeddingModel.from_pretrained("multimodalembedding") +image = Image.load_from_file( + "gs://cloud-samples-data/vertex-ai/llm/prompts/landmark1.png" +) + +embeddings = model.get_embeddings( + image=image, + contextual_text="Colosseum", + dimension=1408, +) +print(f"Image Embedding: {embeddings.image_embedding}") +print(f"Text Embedding: {embeddings.text_embedding}") diff --git a/litellm/proxy/vertex_ai_endpoints/vertex_endpoints.py b/litellm/proxy/vertex_ai_endpoints/vertex_endpoints.py index 1bfb1c2a098..53edbbcfd3c 100644 --- a/litellm/proxy/vertex_ai_endpoints/vertex_endpoints.py +++ b/litellm/proxy/vertex_ai_endpoints/vertex_endpoints.py @@ -25,6 +25,9 @@ from litellm.batches.main import FileObject from litellm.fine_tuning.main import vertex_fine_tuning_apis_instance from litellm.proxy._types import * from litellm.proxy.auth.user_api_key_auth import user_api_key_auth +from litellm.proxy.pass_through_endpoints.pass_through_endpoints import ( + create_pass_through_route, +) router = APIRouter() default_vertex_config = None @@ -70,10 +73,17 @@ def exception_handler(e: Exception): ) -async def execute_post_vertex_ai_request( +@router.api_route( + "/vertex-ai/{endpoint:path}", methods=["GET", "POST", "PUT", "DELETE"] +) +async def vertex_proxy_route( + endpoint: str, request: Request, - route: str, + fastapi_response: Response, + user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), ): + encoded_endpoint = httpx.URL(endpoint).path + from litellm.fine_tuning.main import vertex_fine_tuning_apis_instance if default_vertex_config is None: @@ -83,250 +93,52 @@ async def execute_post_vertex_ai_request( vertex_project = default_vertex_config.get("vertex_project", None) vertex_location = default_vertex_config.get("vertex_location", None) vertex_credentials = default_vertex_config.get("vertex_credentials", None) + base_target_url = f"https://{vertex_location}-aiplatform.googleapis.com/" - request_data_json = {} - body = await request.body() - body_str = body.decode() - if len(body_str) > 0: - try: - request_data_json = ast.literal_eval(body_str) - except: - request_data_json = json.loads(body_str) - - verbose_proxy_logger.debug( - "Request received by LiteLLM:\n{}".format( - json.dumps(request_data_json, indent=4) - ), + auth_header, _ = vertex_fine_tuning_apis_instance._get_token_and_url( + model="", + gemini_api_key=None, + vertex_credentials=vertex_credentials, + vertex_project=vertex_project, + vertex_location=vertex_location, + stream=False, + custom_llm_provider="vertex_ai_beta", + api_base="", ) - response = ( - await vertex_fine_tuning_apis_instance.pass_through_vertex_ai_POST_request( - request_data=request_data_json, - vertex_project=vertex_project, - vertex_location=vertex_location, - vertex_credentials=vertex_credentials, - request_route=route, - ) + headers = { + "Authorization": f"Bearer {auth_header}", + } + + request_route = encoded_endpoint + verbose_proxy_logger.debug("request_route %s", request_route) + + # Ensure endpoint starts with '/' for proper URL construction + if not encoded_endpoint.startswith("/"): + encoded_endpoint = "/" + encoded_endpoint + + # Construct the full target URL using httpx + base_url = httpx.URL(base_target_url) + updated_url = base_url.copy_with(path=encoded_endpoint) + + verbose_proxy_logger.debug("updated url %s", updated_url) + + ## check for streaming + is_streaming_request = False + if "stream" in str(updated_url): + is_streaming_request = True + + ## CREATE PASS-THROUGH + endpoint_func = create_pass_through_route( + endpoint=endpoint, + target=str(updated_url), + custom_headers=headers, + ) # dynamically construct pass-through endpoint based on incoming path + received_value = await endpoint_func( + request, + fastapi_response, + user_api_key_dict, + stream=is_streaming_request, ) - return response - - -@router.post( - "/vertex-ai/publishers/google/models/{model_id:path}:generateContent", - dependencies=[Depends(user_api_key_auth)], - tags=["Vertex AI endpoints"], -) -async def vertex_generate_content( - request: Request, - fastapi_response: Response, - model_id: str, - user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), -): - """ - this is a pass through endpoint for the Vertex AI API. /generateContent endpoint - - Example Curl: - ``` - curl http://localhost:4000/vertex-ai/publishers/google/models/gemini-1.5-flash-001:generateContent \ - -H "Content-Type: application/json" \ - -H "Authorization: Bearer sk-1234" \ - -d '{"contents":[{"role": "user", "parts":[{"text": "hi"}]}]}' - ``` - - Vertex API Reference: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/inference#rest - it uses the vertex ai credentials on the proxy and forwards to vertex ai api - """ - try: - response = await execute_post_vertex_ai_request( - request=request, - route=f"/publishers/google/models/{model_id}:generateContent", - ) - return response - except Exception as e: - raise exception_handler(e) from e - - -@router.post( - "/vertex-ai/publishers/google/models/{model_id:path}:predict", - dependencies=[Depends(user_api_key_auth)], - tags=["Vertex AI endpoints"], -) -async def vertex_predict_endpoint( - request: Request, - fastapi_response: Response, - model_id: str, - user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), -): - """ - this is a pass through endpoint for the Vertex AI API. /predict endpoint - Use this for: - - Embeddings API - Text Embedding, Multi Modal Embedding - - Imagen API - - Code Completion API - - Example Curl: - ``` - curl http://localhost:4000/vertex-ai/publishers/google/models/textembedding-gecko@001:predict \ - -H "Content-Type: application/json" \ - -H "Authorization: Bearer sk-1234" \ - -d '{"instances":[{"content": "gm"}]}' - ``` - - Vertex API Reference: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/text-embeddings-api#generative-ai-get-text-embedding-drest - it uses the vertex ai credentials on the proxy and forwards to vertex ai api - """ - try: - response = await execute_post_vertex_ai_request( - request=request, - route=f"/publishers/google/models/{model_id}:predict", - ) - return response - except Exception as e: - raise exception_handler(e) from e - - -@router.post( - "/vertex-ai/publishers/google/models/{model_id:path}:countTokens", - dependencies=[Depends(user_api_key_auth)], - tags=["Vertex AI endpoints"], -) -async def vertex_countTokens_endpoint( - request: Request, - fastapi_response: Response, - model_id: str, - user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), -): - """ - this is a pass through endpoint for the Vertex AI API. /countTokens endpoint - https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/count-tokens#curl - - - Example Curl: - ``` - curl http://localhost:4000/vertex-ai/publishers/google/models/gemini-1.5-flash-001:countTokens \ - -H "Content-Type: application/json" \ - -H "Authorization: Bearer sk-1234" \ - -d '{"contents":[{"role": "user", "parts":[{"text": "hi"}]}]}' - ``` - - it uses the vertex ai credentials on the proxy and forwards to vertex ai api - """ - try: - response = await execute_post_vertex_ai_request( - request=request, - route=f"/publishers/google/models/{model_id}:countTokens", - ) - return response - except Exception as e: - raise exception_handler(e) from e - - -@router.post( - "/vertex-ai/batchPredictionJobs", - dependencies=[Depends(user_api_key_auth)], - tags=["Vertex AI endpoints"], -) -async def vertex_create_batch_prediction_job( - request: Request, - fastapi_response: Response, - user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), -): - """ - this is a pass through endpoint for the Vertex AI API. /batchPredictionJobs endpoint - - Vertex API Reference: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/batch-prediction-api#syntax - - it uses the vertex ai credentials on the proxy and forwards to vertex ai api - """ - try: - response = await execute_post_vertex_ai_request( - request=request, - route="/batchPredictionJobs", - ) - return response - except Exception as e: - raise exception_handler(e) from e - - -@router.post( - "/vertex-ai/tuningJobs", - dependencies=[Depends(user_api_key_auth)], - tags=["Vertex AI endpoints"], -) -async def vertex_create_fine_tuning_job( - request: Request, - fastapi_response: Response, - user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), -): - """ - this is a pass through endpoint for the Vertex AI API. /tuningJobs endpoint - - Vertex API Reference: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/tuning - - it uses the vertex ai credentials on the proxy and forwards to vertex ai api - """ - try: - response = await execute_post_vertex_ai_request( - request=request, - route="/tuningJobs", - ) - return response - except Exception as e: - raise exception_handler(e) from e - - -@router.post( - "/vertex-ai/tuningJobs/{job_id:path}:cancel", - dependencies=[Depends(user_api_key_auth)], - tags=["Vertex AI endpoints"], -) -async def vertex_cancel_fine_tuning_job( - request: Request, - job_id: str, - fastapi_response: Response, - user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), -): - """ - this is a pass through endpoint for the Vertex AI API. tuningJobs/{job_id:path}:cancel - - Vertex API Reference: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/tuning#cancel_a_tuning_job - - it uses the vertex ai credentials on the proxy and forwards to vertex ai api - """ - try: - - response = await execute_post_vertex_ai_request( - request=request, - route=f"/tuningJobs/{job_id}:cancel", - ) - return response - except Exception as e: - raise exception_handler(e) from e - - -@router.post( - "/vertex-ai/cachedContents", - dependencies=[Depends(user_api_key_auth)], - tags=["Vertex AI endpoints"], -) -async def vertex_create_add_cached_content( - request: Request, - fastapi_response: Response, - user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), -): - """ - this is a pass through endpoint for the Vertex AI API. /cachedContents endpoint - - Vertex API Reference: https://cloud.google.com/vertex-ai/generative-ai/docs/context-cache/context-cache-create#create-context-cache-sample-drest - - it uses the vertex ai credentials on the proxy and forwards to vertex ai api - """ - try: - response = await execute_post_vertex_ai_request( - request=request, - route="/cachedContents", - ) - return response - except Exception as e: - raise exception_handler(e) from e + return received_value