diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 27791f034bb..a38988b2e38 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -939,12 +939,20 @@ class BaseLLMHTTPHandler: sync_httpx_client = client try: - response = sync_httpx_client.post( - url=api_base, - headers=headers, - data=signed_body if signed_body is not None else json.dumps(data), - timeout=timeout, - ) + if signed_body is not None: + response = sync_httpx_client.post( + url=api_base, + headers=headers, + data=signed_body, + timeout=timeout, + ) + else: + response = sync_httpx_client.post( + url=api_base, + headers=headers, + json=data, + timeout=timeout, + ) except Exception as e: raise self._handle_error( e=e, diff --git a/litellm/llms/oci/chat/cohere.py b/litellm/llms/oci/chat/cohere.py index 0ffa685a9f2..d5d136d179d 100644 --- a/litellm/llms/oci/chat/cohere.py +++ b/litellm/llms/oci/chat/cohere.py @@ -216,8 +216,15 @@ def handle_cohere_response( finish_reason = "length" elif oci_finish_reason == "TOOL_CALL": finish_reason = "tool_calls" + elif oci_finish_reason is not None: + # OCI Cohere can emit error/cancel finish reasons (e.g. ``ERROR``, + # ``ERROR_TOXIC``, ``ERROR_LIMIT``, ``USER_CANCEL``) that aren't part + # of OpenAI's standard set. Normalize them to ``"stop"`` so downstream + # consumers switching on ``finish_reason`` keep working — matches the + # streaming handler below. + finish_reason = "stop" else: - finish_reason = oci_finish_reason + finish_reason = None tool_calls: Optional[List[Dict[str, Any]]] = None if cohere_response.chatResponse.toolCalls: diff --git a/litellm/llms/oci/chat/generic.py b/litellm/llms/oci/chat/generic.py index c860c3fefd5..e0700ca2ead 100644 --- a/litellm/llms/oci/chat/generic.py +++ b/litellm/llms/oci/chat/generic.py @@ -332,7 +332,11 @@ def handle_generic_response( elif oci_finish_reason == "TOOL_CALLS": model_response.choices[0].finish_reason = "tool_calls" # type: ignore[union-attr] elif oci_finish_reason is not None: - model_response.choices[0].finish_reason = oci_finish_reason # type: ignore[union-attr,assignment] + # OCI GENERIC can emit non-OpenAI finish reasons (e.g. ``ERROR``, + # ``CONTENT_FILTERED``, ``CANCELLED``). Normalize to ``"stop"`` so + # downstream consumers switching on ``finish_reason`` keep working — + # matches the streaming handler. + model_response.choices[0].finish_reason = "stop" # type: ignore[union-attr] oci_usage = completion_response.chatResponse.usage reasoning_tokens: Optional[int] = None @@ -398,8 +402,13 @@ def handle_generic_stream_chunk(dict_chunk: dict) -> ModelResponseStream: finish_reason = "length" elif oci_finish_reason == "TOOL_CALLS": finish_reason = "tool_calls" + elif oci_finish_reason is not None: + # OCI GENERIC can emit non-OpenAI finish reasons (e.g. ``ERROR``, + # ``CONTENT_FILTERED``, ``CANCELLED``). Normalize to ``"stop"`` so + # downstream consumers switching on ``finish_reason`` keep working. + finish_reason = "stop" else: - finish_reason = oci_finish_reason + finish_reason = None return ModelResponseStream( choices=[ diff --git a/litellm/llms/oci/chat/transformation.py b/litellm/llms/oci/chat/transformation.py index 1bf960e5efc..740292dbc74 100644 --- a/litellm/llms/oci/chat/transformation.py +++ b/litellm/llms/oci/chat/transformation.py @@ -252,6 +252,12 @@ class OCIChatConfig(BaseConfig): adapted_params[key] = value continue adapted_params[alias] = value + # Preserve the original OpenAI ``response_format`` key alongside the + # OCI-mapped ``responseFormat`` so downstream litellm framework code + # (e.g. ``json_mode`` detection, logging) that inspects + # ``optional_params["response_format"]`` continues to work. + if alias == "responseFormat": + adapted_params["response_format"] = value return adapted_params