mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-29 01:42:19 +00:00
feat(interactions): migrate to Google Interactions API steps schema (May 2026)
Default to Api-Revision: 2026-05-20 (new `steps` schema). Add `litellm.use_legacy_interactions_schema` global flag that sends Api-Revision: 2026-05-07 for operators who need the legacy `outputs` schema until June 8, 2026. - Inject Api-Revision header in GoogleAIStudioInteractionsConfig.validate_environment() - Auto-coalesce response_mime_type → response_format and image_config migration on new schema - Add steps field to InteractionsAPIResponse and InteractionsAPIStreamingResponse - Add StepStart/StepDelta/StepStop/InteractionCreated/etc. SSE event types - Update streaming completion detection to handle interaction.completed event - Bridge transformer populates both outputs and steps fields - Bridge streaming iterator emits new-schema events by default Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
e59e34bed3
commit
58a52da9d5
9 changed files with 536 additions and 170 deletions
|
|
@ -225,6 +225,10 @@ use_chat_completions_url_for_anthropic_messages: bool = bool(
|
|||
route_all_chat_openai_to_responses: bool = (
|
||||
os.getenv("LITELLM_ROUTE_ALL_CHAT_OPENAI_TO_RESPONSES", "false").lower() == "true"
|
||||
) # When True, routes all OpenAI /chat/completions requests through the Responses API bridge
|
||||
use_legacy_interactions_schema: bool = (
|
||||
os.getenv("LITELLM_USE_LEGACY_INTERACTIONS_SCHEMA", "false").lower() == "true"
|
||||
) # When True, sends Api-Revision: 2026-05-07 to Google so responses use the legacy `outputs`
|
||||
# schema instead of the new `steps` schema. Remove this flag after June 8, 2026.
|
||||
retry = True
|
||||
### AUTH ###
|
||||
api_key: Optional[str] = None
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
Streaming iterator for transforming Responses API stream to Interactions API stream.
|
||||
"""
|
||||
|
||||
from typing import Any, AsyncIterator, Dict, Iterator, List, Optional, cast
|
||||
from typing import Any, AsyncIterator, Dict, Iterator, Optional, cast
|
||||
|
||||
from litellm.responses.streaming_iterator import (
|
||||
BaseResponsesAPIStreamingIterator,
|
||||
|
|
@ -15,7 +15,6 @@ from litellm.types.interactions import (
|
|||
InteractionsAPIStreamingResponse,
|
||||
)
|
||||
from litellm.types.llms.openai import (
|
||||
ContentPartAddedEvent,
|
||||
OutputTextDeltaEvent,
|
||||
ResponseCompletedEvent,
|
||||
ResponseCreatedEvent,
|
||||
|
|
@ -30,7 +29,13 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
|
||||
This class handles both sync and async iteration, transforming Responses API
|
||||
streaming events (output.text.delta, response.completed, etc.) to Interactions
|
||||
API streaming events (content.delta, interaction.complete, etc.).
|
||||
API streaming events.
|
||||
|
||||
Schema selection:
|
||||
- New schema (default, use_legacy_interactions_schema=False):
|
||||
interaction.created → step.start → step.delta … → step.stop → interaction.completed
|
||||
- Legacy schema (use_legacy_interactions_schema=True, remove after June 8 2026):
|
||||
interaction.start → content.start → content.delta … → content.stop → interaction.complete
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
|
|
@ -52,7 +57,12 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
self.collected_text = ""
|
||||
self.sent_interaction_start = False
|
||||
self.sent_content_start = False
|
||||
self._pending_events: List[InteractionsAPIStreamingResponse] = []
|
||||
|
||||
@property
|
||||
def _use_legacy(self) -> bool:
|
||||
import litellm
|
||||
|
||||
return litellm.use_legacy_interactions_schema
|
||||
|
||||
def _transform_responses_chunk_to_interactions_chunk(
|
||||
self,
|
||||
|
|
@ -61,91 +71,78 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
"""
|
||||
Transform a Responses API streaming chunk to an Interactions API streaming chunk.
|
||||
|
||||
Responses API events:
|
||||
- output.text.delta -> content.delta
|
||||
- response.completed -> interaction.complete
|
||||
|
||||
Interactions API events:
|
||||
- interaction.start
|
||||
- content.start
|
||||
- content.delta
|
||||
- content.stop
|
||||
- interaction.complete
|
||||
Emits new-schema events by default; falls back to legacy events when
|
||||
``litellm.use_legacy_interactions_schema`` is True.
|
||||
Remove legacy branch after June 8, 2026.
|
||||
"""
|
||||
if not responses_chunk:
|
||||
return None
|
||||
|
||||
# Handle OutputTextDeltaEvent -> content.delta
|
||||
use_legacy = self._use_legacy
|
||||
|
||||
# Handle OutputTextDeltaEvent
|
||||
if isinstance(responses_chunk, OutputTextDeltaEvent):
|
||||
delta_text = (
|
||||
responses_chunk.delta if isinstance(responses_chunk.delta, str) else ""
|
||||
)
|
||||
self.collected_text += delta_text
|
||||
item_id = (
|
||||
getattr(responses_chunk, "item_id", None) or f"interaction_{id(self)}"
|
||||
)
|
||||
|
||||
# Fallback: emit interaction.start, and queue content.start carrying this
|
||||
# delta so the first token is preserved in the stream.
|
||||
# Send the "interaction started" event on the first delta
|
||||
if not self.sent_interaction_start:
|
||||
self.sent_interaction_start = True
|
||||
self.sent_content_start = True
|
||||
self._pending_events.append(
|
||||
InteractionsAPIStreamingResponse(
|
||||
event_type="content.start",
|
||||
id=getattr(responses_chunk, "item_id", None),
|
||||
object="content",
|
||||
delta={"type": "text", "text": delta_text},
|
||||
if use_legacy:
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.start",
|
||||
id=item_id,
|
||||
object="interaction",
|
||||
status="in_progress",
|
||||
model=self.model,
|
||||
)
|
||||
else:
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.created",
|
||||
id=item_id,
|
||||
object="interaction",
|
||||
status="in_progress",
|
||||
model=self.model,
|
||||
)
|
||||
)
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.start",
|
||||
id=getattr(responses_chunk, "item_id", None)
|
||||
or f"interaction_{id(self)}",
|
||||
object="interaction",
|
||||
status="in_progress",
|
||||
model=self.model,
|
||||
)
|
||||
|
||||
# Fallback: emit content.start if ContentPartAddedEvent never arrived
|
||||
# Send the "content/step started" event on the second delta
|
||||
if not self.sent_content_start:
|
||||
self.sent_content_start = True
|
||||
if use_legacy:
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="content.start",
|
||||
id=item_id,
|
||||
object="content",
|
||||
delta={"type": "text", "text": ""},
|
||||
)
|
||||
else:
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="step.start",
|
||||
index=0,
|
||||
step={"type": "model_output", "content": []},
|
||||
)
|
||||
|
||||
# Emit the delta itself
|
||||
if use_legacy:
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="content.start",
|
||||
id=getattr(responses_chunk, "item_id", None),
|
||||
event_type="content.delta",
|
||||
id=item_id,
|
||||
object="content",
|
||||
delta={"text": delta_text},
|
||||
)
|
||||
else:
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="step.delta",
|
||||
index=0,
|
||||
delta={"type": "text", "text": delta_text},
|
||||
)
|
||||
|
||||
# Normal path: emit content.delta with type field
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="content.delta",
|
||||
id=getattr(responses_chunk, "item_id", None),
|
||||
object="content",
|
||||
delta={"type": "text", "text": delta_text},
|
||||
)
|
||||
|
||||
# Handle ContentPartAddedEvent -> content.start (arrives before text deltas)
|
||||
if isinstance(responses_chunk, ContentPartAddedEvent):
|
||||
# Fallback: emit interaction.start if ResponseCreatedEvent never arrived
|
||||
if not self.sent_interaction_start:
|
||||
self.sent_interaction_start = True
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.start",
|
||||
id=getattr(responses_chunk, "item_id", None)
|
||||
or f"interaction_{id(self)}",
|
||||
object="interaction",
|
||||
status="in_progress",
|
||||
model=self.model,
|
||||
)
|
||||
if not self.sent_content_start:
|
||||
self.sent_content_start = True
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="content.start",
|
||||
id=getattr(responses_chunk, "item_id", None),
|
||||
object="content",
|
||||
delta={"type": "text", "text": ""},
|
||||
)
|
||||
return None
|
||||
|
||||
# Handle ResponseCreatedEvent or ResponseInProgressEvent -> interaction.start
|
||||
# Handle ResponseCreatedEvent or ResponseInProgressEvent
|
||||
if isinstance(responses_chunk, (ResponseCreatedEvent, ResponseInProgressEvent)):
|
||||
if not self.sent_interaction_start:
|
||||
self.sent_interaction_start = True
|
||||
|
|
@ -153,39 +150,47 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
getattr(responses_chunk.response, "id", None)
|
||||
if hasattr(responses_chunk, "response")
|
||||
else None
|
||||
) or f"interaction_{id(self)}"
|
||||
event_type = (
|
||||
"interaction.start" if use_legacy else "interaction.created"
|
||||
)
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.start",
|
||||
id=response_id or f"interaction_{id(self)}",
|
||||
event_type=event_type,
|
||||
id=response_id,
|
||||
object="interaction",
|
||||
status="in_progress",
|
||||
model=self.model,
|
||||
)
|
||||
|
||||
# Handle ResponseCompletedEvent -> interaction.complete
|
||||
# Handle ResponseCompletedEvent
|
||||
if isinstance(responses_chunk, ResponseCompletedEvent):
|
||||
self.finished = True
|
||||
response = responses_chunk.response
|
||||
response_id = getattr(response, "id", None) or f"interaction_{id(self)}"
|
||||
|
||||
# Send content.stop first if content was started
|
||||
if self.sent_content_start:
|
||||
# Note: We'll send this in the iterator, not here
|
||||
pass
|
||||
|
||||
# Send interaction.complete
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.complete",
|
||||
id=getattr(response, "id", None) or f"interaction_{id(self)}",
|
||||
object="interaction",
|
||||
status="completed",
|
||||
model=self.model,
|
||||
outputs=[
|
||||
{
|
||||
"type": "text",
|
||||
"text": self.collected_text,
|
||||
}
|
||||
],
|
||||
)
|
||||
if use_legacy:
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.complete",
|
||||
id=response_id,
|
||||
object="interaction",
|
||||
status="completed",
|
||||
model=self.model,
|
||||
outputs=[{"type": "text", "text": self.collected_text}],
|
||||
)
|
||||
else:
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="interaction.completed",
|
||||
id=response_id,
|
||||
object="interaction",
|
||||
status="completed",
|
||||
model=self.model,
|
||||
steps=[
|
||||
{
|
||||
"type": "model_output",
|
||||
"content": [{"type": "text", "text": self.collected_text}],
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
# For other event types, return None (skip)
|
||||
return None
|
||||
|
|
@ -207,10 +212,6 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
delattr(self, "_pending_interaction_complete")
|
||||
return pending
|
||||
|
||||
# Drain events queued from a prior chunk (e.g. content.start emitted alongside
|
||||
# the interaction.start fallback for the first OutputTextDeltaEvent).
|
||||
if self._pending_events:
|
||||
return self._pending_events.pop(0)
|
||||
# Use a loop instead of recursion to avoid stack overflow
|
||||
sync_iterator = cast(
|
||||
SyncResponsesAPIStreamingIterator, self.responses_stream_iterator
|
||||
|
|
@ -226,22 +227,29 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
)
|
||||
|
||||
if transformed:
|
||||
# If we finished and content was started, send content.stop before interaction.complete
|
||||
completion_event_type = (
|
||||
"interaction.complete"
|
||||
if self._use_legacy
|
||||
else "interaction.completed"
|
||||
)
|
||||
stop_event_type = (
|
||||
"content.stop" if self._use_legacy else "step.stop"
|
||||
)
|
||||
# If content was started, send the stop event before the completion event.
|
||||
if (
|
||||
self.finished
|
||||
and self.sent_content_start
|
||||
and transformed.event_type == "interaction.complete"
|
||||
and transformed.event_type == completion_event_type
|
||||
):
|
||||
# Send content.stop first
|
||||
content_stop = InteractionsAPIStreamingResponse(
|
||||
event_type="content.stop",
|
||||
stop_chunk = InteractionsAPIStreamingResponse(
|
||||
event_type=stop_event_type,
|
||||
index=0,
|
||||
id=transformed.id,
|
||||
object="content",
|
||||
delta={"type": "text", "text": self.collected_text},
|
||||
)
|
||||
# Store the interaction.complete to send next
|
||||
self._pending_interaction_complete = transformed
|
||||
return content_stop
|
||||
return stop_chunk
|
||||
return transformed
|
||||
|
||||
# If no transformation, continue to next chunk (loop continues)
|
||||
|
|
@ -249,10 +257,14 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
except StopIteration:
|
||||
self.finished = True
|
||||
|
||||
# Send final events if needed
|
||||
# Send final stop event if content was started
|
||||
if self.sent_content_start:
|
||||
stop_event_type = (
|
||||
"content.stop" if self._use_legacy else "step.stop"
|
||||
)
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="content.stop",
|
||||
event_type=stop_event_type,
|
||||
index=0,
|
||||
object="content",
|
||||
delta={"type": "text", "text": self.collected_text},
|
||||
)
|
||||
|
|
@ -276,10 +288,6 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
delattr(self, "_pending_interaction_complete")
|
||||
return pending
|
||||
|
||||
# Drain events queued from a prior chunk (e.g. content.start emitted alongside
|
||||
# the interaction.start fallback for the first OutputTextDeltaEvent).
|
||||
if self._pending_events:
|
||||
return self._pending_events.pop(0)
|
||||
# Use a loop instead of recursion to avoid stack overflow
|
||||
async_iterator = cast(
|
||||
ResponsesAPIStreamingIterator, self.responses_stream_iterator
|
||||
|
|
@ -295,22 +303,29 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
)
|
||||
|
||||
if transformed:
|
||||
# If we finished and content was started, send content.stop before interaction.complete
|
||||
completion_event_type = (
|
||||
"interaction.complete"
|
||||
if self._use_legacy
|
||||
else "interaction.completed"
|
||||
)
|
||||
stop_event_type = (
|
||||
"content.stop" if self._use_legacy else "step.stop"
|
||||
)
|
||||
# If content was started, send the stop event before the completion event.
|
||||
if (
|
||||
self.finished
|
||||
and self.sent_content_start
|
||||
and transformed.event_type == "interaction.complete"
|
||||
and transformed.event_type == completion_event_type
|
||||
):
|
||||
# Send content.stop first
|
||||
content_stop = InteractionsAPIStreamingResponse(
|
||||
event_type="content.stop",
|
||||
stop_chunk = InteractionsAPIStreamingResponse(
|
||||
event_type=stop_event_type,
|
||||
index=0,
|
||||
id=transformed.id,
|
||||
object="content",
|
||||
delta={"type": "text", "text": self.collected_text},
|
||||
)
|
||||
# Store the interaction.complete to send next
|
||||
self._pending_interaction_complete = transformed
|
||||
return content_stop
|
||||
return stop_chunk
|
||||
return transformed
|
||||
|
||||
# If no transformation, continue to next chunk (loop continues)
|
||||
|
|
@ -318,10 +333,14 @@ class LiteLLMResponsesInteractionsStreamingIterator:
|
|||
except StopAsyncIteration:
|
||||
self.finished = True
|
||||
|
||||
# Send final events if needed
|
||||
# Send final stop event if content was started
|
||||
if self.sent_content_start:
|
||||
stop_event_type = (
|
||||
"content.stop" if self._use_legacy else "step.stop"
|
||||
)
|
||||
return InteractionsAPIStreamingResponse(
|
||||
event_type="content.stop",
|
||||
event_type=stop_event_type,
|
||||
index=0,
|
||||
object="content",
|
||||
delta={"type": "text", "text": self.collected_text},
|
||||
)
|
||||
|
|
|
|||
|
|
@ -226,29 +226,36 @@ class LiteLLMResponsesInteractionsConfig:
|
|||
- Map status
|
||||
- Extract usage
|
||||
"""
|
||||
# Extract text from outputs
|
||||
outputs = []
|
||||
# Extract text from outputs and build both `outputs` (legacy) and `steps` (new schema).
|
||||
outputs: List[Dict[str, Any]] = []
|
||||
steps: List[Dict[str, Any]] = []
|
||||
if hasattr(responses_response, "output") and responses_response.output:
|
||||
for output_item in responses_response.output:
|
||||
# Use getattr with None default to safely access content
|
||||
content = getattr(output_item, "content", None)
|
||||
if content is not None:
|
||||
content_items = content if isinstance(content, list) else [content]
|
||||
model_output_contents: List[Dict[str, Any]] = []
|
||||
for content_item in content_items:
|
||||
# Check if content_item has text attribute
|
||||
text = getattr(content_item, "text", None)
|
||||
if text is not None:
|
||||
outputs.append(
|
||||
{
|
||||
"type": "text",
|
||||
"text": text,
|
||||
}
|
||||
)
|
||||
text_entry = {"type": "text", "text": text}
|
||||
outputs.append(text_entry)
|
||||
model_output_contents.append(text_entry)
|
||||
elif (
|
||||
isinstance(content_item, dict)
|
||||
and content_item.get("type") == "text"
|
||||
):
|
||||
outputs.append(content_item)
|
||||
model_output_contents.append(content_item)
|
||||
if model_output_contents:
|
||||
steps.append(
|
||||
{
|
||||
"type": "model_output",
|
||||
"content": model_output_contents,
|
||||
}
|
||||
)
|
||||
|
||||
# Convert created_at to ISO string
|
||||
created_at = getattr(responses_response, "created_at", None)
|
||||
|
|
@ -270,12 +277,14 @@ class LiteLLMResponsesInteractionsConfig:
|
|||
else:
|
||||
interactions_status = status
|
||||
|
||||
# Build interactions response
|
||||
# Build interactions response — populate both `outputs` (legacy schema) and
|
||||
# `steps` (new schema) so callers work regardless of which schema they expect.
|
||||
interactions_response_dict: Dict[str, Any] = {
|
||||
"id": getattr(responses_response, "id", ""),
|
||||
"object": "interaction",
|
||||
"status": interactions_status,
|
||||
"outputs": outputs,
|
||||
"steps": steps,
|
||||
"model": model or getattr(responses_response, "model", ""),
|
||||
"created": created,
|
||||
}
|
||||
|
|
|
|||
|
|
@ -101,10 +101,14 @@ class BaseInteractionsAPIStreamingIterator:
|
|||
)
|
||||
)
|
||||
|
||||
# Store the completed response (check for status=completed)
|
||||
if (
|
||||
streaming_response
|
||||
and getattr(streaming_response, "status", None) == "completed"
|
||||
# Store the completed response.
|
||||
# Legacy schema signals completion via status="completed".
|
||||
# New schema (Api-Revision: 2026-05-20) uses event_type="interaction.completed".
|
||||
# Remove the legacy check after June 8, 2026.
|
||||
if streaming_response and (
|
||||
getattr(streaming_response, "status", None) == "completed"
|
||||
or getattr(streaming_response, "event_type", None)
|
||||
== "interaction.completed"
|
||||
):
|
||||
self.completed_response = streaming_response
|
||||
self._handle_logging_completed_response()
|
||||
|
|
|
|||
|
|
@ -6,13 +6,18 @@ Per OpenAPI spec (https://ai.google.dev/static/api/interactions.openapi.json):
|
|||
- Get: GET https://generativelanguage.googleapis.com/{api_version}/interactions/{interaction_id}
|
||||
- Delete: DELETE https://generativelanguage.googleapis.com/{api_version}/interactions/{interaction_id}
|
||||
|
||||
This is a thin wrapper - no transformation needed since we follow the spec directly.
|
||||
Schema versioning:
|
||||
- Default (Api-Revision: 2026-05-20): new `steps` schema.
|
||||
- Legacy (Api-Revision: 2026-05-07): old `outputs` schema, controlled via
|
||||
litellm.use_legacy_interactions_schema = True. Remove flag after June 8, 2026.
|
||||
"""
|
||||
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple
|
||||
|
||||
import httpx
|
||||
|
||||
import litellm
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.litellm_core_utils.core_helpers import process_response_headers
|
||||
from litellm.litellm_core_utils.url_utils import encode_url_path_segment
|
||||
|
|
@ -84,6 +89,15 @@ class GoogleAIStudioInteractionsConfig(BaseInteractionsAPIConfig):
|
|||
api_key = GeminiModelInfo.get_api_key(litellm_params.get("api_key"))
|
||||
if api_key:
|
||||
headers["x-goog-api-key"] = api_key
|
||||
|
||||
# Inject the Api-Revision header to select the response schema.
|
||||
# Default to the new `steps` schema unless the operator has opted out.
|
||||
# Remove this conditional after June 8, 2026 and always use 2026-05-20.
|
||||
if litellm.use_legacy_interactions_schema:
|
||||
headers["Api-Revision"] = "2026-05-07"
|
||||
else:
|
||||
headers["Api-Revision"] = "2026-05-20"
|
||||
|
||||
return headers
|
||||
|
||||
def get_complete_url(
|
||||
|
|
@ -119,8 +133,19 @@ class GoogleAIStudioInteractionsConfig(BaseInteractionsAPIConfig):
|
|||
headers: dict,
|
||||
) -> Dict:
|
||||
"""
|
||||
Build request body per OpenAPI spec - minimal transformation.
|
||||
Build request body per OpenAPI spec.
|
||||
|
||||
When on the new schema (use_legacy_interactions_schema=False, the default):
|
||||
- ``response_mime_type`` is folded into ``response_format`` and stripped from
|
||||
the body (the field was removed in Api-Revision 2026-05-20).
|
||||
- ``generation_config.image_config`` is moved to a ``response_format`` entry
|
||||
with ``"type": "image"`` (also removed from generation_config in 2026-05-20).
|
||||
|
||||
When on the legacy schema (use_legacy_interactions_schema=True):
|
||||
- All fields are forwarded as-is.
|
||||
"""
|
||||
use_legacy: bool = litellm.use_legacy_interactions_schema
|
||||
|
||||
request_body: Dict[str, Any] = {}
|
||||
|
||||
# Model or Agent (one required)
|
||||
|
|
@ -135,24 +160,72 @@ class GoogleAIStudioInteractionsConfig(BaseInteractionsAPIConfig):
|
|||
if input is not None:
|
||||
request_body["input"] = input
|
||||
|
||||
# Pass through optional params directly (they match the spec)
|
||||
# Pass through optional params — legacy schema keeps all fields as-is.
|
||||
optional_keys = [
|
||||
"tools",
|
||||
"system_instruction",
|
||||
"generation_config",
|
||||
"stream",
|
||||
"store",
|
||||
"background",
|
||||
"environment",
|
||||
"response_modalities",
|
||||
"response_format",
|
||||
"response_mime_type",
|
||||
"previous_interaction_id",
|
||||
]
|
||||
for key in optional_keys:
|
||||
if optional_params.get(key) is not None:
|
||||
request_body[key] = optional_params[key]
|
||||
|
||||
if use_legacy:
|
||||
# Legacy schema: forward response_mime_type and response_format as-is.
|
||||
for key in ("response_format", "response_mime_type", "generation_config"):
|
||||
if optional_params.get(key) is not None:
|
||||
request_body[key] = optional_params[key]
|
||||
else:
|
||||
# New schema (Api-Revision: 2026-05-20):
|
||||
# response_mime_type is removed — fold it into response_format.
|
||||
response_format = optional_params.get("response_format")
|
||||
response_mime_type = optional_params.get("response_mime_type")
|
||||
|
||||
if response_mime_type and (
|
||||
not isinstance(response_format, dict)
|
||||
or "mime_type" not in response_format
|
||||
):
|
||||
# Wrap the legacy schema into the new polymorphic format.
|
||||
response_format = {
|
||||
"type": "text",
|
||||
"mime_type": response_mime_type,
|
||||
"schema": response_format,
|
||||
}
|
||||
|
||||
if response_format is not None:
|
||||
request_body["response_format"] = response_format
|
||||
|
||||
# image_config moves out of generation_config into response_format.
|
||||
generation_config: Optional[Dict[str, Any]] = optional_params.get(
|
||||
"generation_config"
|
||||
)
|
||||
if generation_config is not None:
|
||||
image_config = None
|
||||
if isinstance(generation_config, dict):
|
||||
image_config = generation_config.pop("image_config", None)
|
||||
if not generation_config:
|
||||
generation_config = None
|
||||
|
||||
if generation_config is not None:
|
||||
request_body["generation_config"] = generation_config
|
||||
|
||||
if image_config is not None:
|
||||
# Move image_config to response_format with type=image.
|
||||
image_rf: Dict[str, Any] = {"type": "image", **image_config}
|
||||
existing_rf = request_body.get("response_format")
|
||||
if existing_rf is None:
|
||||
request_body["response_format"] = image_rf
|
||||
elif isinstance(existing_rf, list):
|
||||
existing_rf.append(image_rf)
|
||||
else:
|
||||
# Convert single entry to array for multimodal output.
|
||||
request_body["response_format"] = [existing_rf, image_rf]
|
||||
|
||||
return request_body
|
||||
|
||||
def transform_response(
|
||||
|
|
|
|||
|
|
@ -4328,6 +4328,11 @@ class ProxyConfig:
|
|||
"health_check_concurrency", None
|
||||
)
|
||||
health_check_details = general_settings.get("health_check_details", True)
|
||||
### INTERACTIONS API SCHEMA ###
|
||||
if general_settings.get("use_legacy_interactions_schema") is not None:
|
||||
litellm.use_legacy_interactions_schema = bool(
|
||||
general_settings["use_legacy_interactions_schema"]
|
||||
)
|
||||
# Health-check-driven routing (opt-in, passes through to Router later)
|
||||
_enable_hc_routing = general_settings.get(
|
||||
"enable_health_check_routing", False
|
||||
|
|
|
|||
|
|
@ -36,9 +36,13 @@ from litellm.types.interactions.generated import (
|
|||
GoogleSearchResultContent,
|
||||
ImageContent,
|
||||
Interaction,
|
||||
InteractionCompleted,
|
||||
InteractionCreated,
|
||||
InteractionEvent,
|
||||
InteractionEnvironment,
|
||||
InteractionInProgress,
|
||||
InteractionInput,
|
||||
InteractionRequiresAction,
|
||||
InteractionsAPIOptionalRequestParams,
|
||||
InteractionsAPIResponse,
|
||||
InteractionsAPIStreamingResponse,
|
||||
|
|
@ -50,6 +54,9 @@ from litellm.types.interactions.generated import (
|
|||
McpServerToolResultContent,
|
||||
ModelOption,
|
||||
ResponseModality,
|
||||
StepDelta,
|
||||
StepStart,
|
||||
StepStop,
|
||||
)
|
||||
from litellm.types.interactions.generated import (
|
||||
Status3 as InteractionStatus, # Main request/response types; Content types; Turn for multi-turn conversations; Tool types; Config types; Usage; Status enum; Events for streaming; Agent configs; Model/Agent options; Response modality; Annotation; LiteLLM types; Backwards compat aliases
|
||||
|
|
@ -115,6 +122,14 @@ __all__ = [
|
|||
"AgentOption",
|
||||
"ResponseModality",
|
||||
"Annotation",
|
||||
# New schema SSE event types (Api-Revision: 2026-05-20)
|
||||
"StepStart",
|
||||
"StepDelta",
|
||||
"StepStop",
|
||||
"InteractionCreated",
|
||||
"InteractionInProgress",
|
||||
"InteractionCompleted",
|
||||
"InteractionRequiresAction",
|
||||
# LiteLLM types
|
||||
"InteractionEnvironment",
|
||||
"InteractionInput",
|
||||
|
|
|
|||
|
|
@ -1151,9 +1151,114 @@ class InteractionEvent(BaseModel):
|
|||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------
|
||||
# New schema SSE event types (Api-Revision: 2026-05-20)
|
||||
# These replace the legacy content.* / interaction.start|complete
|
||||
# events and will become the only events after June 8, 2026.
|
||||
# ---------------------------------------------------------------
|
||||
|
||||
|
||||
class StepStart(BaseModel):
|
||||
"""Emitted when a new step begins (replaces content.start)."""
|
||||
|
||||
event_type: Literal["step.start"] = "step.start"
|
||||
index: Optional[int] = None
|
||||
step: Optional[Dict[str, Any]] = Field(
|
||||
None,
|
||||
description="The initial step data (type, content, signature, etc.).",
|
||||
)
|
||||
event_id: Optional[str] = Field(
|
||||
None,
|
||||
description="The event_id token to be used to resume the interaction stream.",
|
||||
)
|
||||
|
||||
|
||||
class StepDelta(BaseModel):
|
||||
"""Emitted for incremental step content (replaces content.delta)."""
|
||||
|
||||
event_type: Literal["step.delta"] = "step.delta"
|
||||
index: Optional[int] = None
|
||||
delta: Optional[Dict[str, Any]] = Field(
|
||||
None,
|
||||
description="Incremental content delta (e.g. text, arguments_delta for function calls).",
|
||||
)
|
||||
event_id: Optional[str] = Field(
|
||||
None,
|
||||
description="The event_id token to be used to resume the interaction stream.",
|
||||
)
|
||||
|
||||
|
||||
class StepStop(BaseModel):
|
||||
"""Emitted when a step finishes (replaces content.stop)."""
|
||||
|
||||
event_type: Literal["step.stop"] = "step.stop"
|
||||
index: Optional[int] = None
|
||||
status: Optional[str] = Field(
|
||||
None,
|
||||
description="Step completion status (e.g. 'done').",
|
||||
)
|
||||
event_id: Optional[str] = Field(
|
||||
None,
|
||||
description="The event_id token to be used to resume the interaction stream.",
|
||||
)
|
||||
|
||||
|
||||
class InteractionCreated(BaseModel):
|
||||
"""Emitted when the interaction is first created (replaces interaction.start)."""
|
||||
|
||||
event_type: Literal["interaction.created"] = "interaction.created"
|
||||
interaction: Optional[Dict[str, Any]] = None
|
||||
event_id: Optional[str] = Field(
|
||||
None,
|
||||
description="The event_id token to be used to resume the interaction stream.",
|
||||
)
|
||||
|
||||
|
||||
class InteractionInProgress(BaseModel):
|
||||
"""Emitted while the interaction is running."""
|
||||
|
||||
event_type: Literal["interaction.in_progress"] = "interaction.in_progress"
|
||||
interaction_id: Optional[str] = None
|
||||
event_id: Optional[str] = Field(
|
||||
None,
|
||||
description="The event_id token to be used to resume the interaction stream.",
|
||||
)
|
||||
|
||||
|
||||
class InteractionCompleted(BaseModel):
|
||||
"""Emitted when the interaction finishes (replaces interaction.complete)."""
|
||||
|
||||
event_type: Literal["interaction.completed"] = "interaction.completed"
|
||||
interaction: Optional[Dict[str, Any]] = None
|
||||
event_id: Optional[str] = Field(
|
||||
None,
|
||||
description="The event_id token to be used to resume the interaction stream.",
|
||||
)
|
||||
|
||||
|
||||
class InteractionRequiresAction(BaseModel):
|
||||
"""Emitted when the interaction is paused waiting for a tool result."""
|
||||
|
||||
event_type: Literal["interaction.requires_action"] = "interaction.requires_action"
|
||||
interaction_id: Optional[str] = None
|
||||
event_id: Optional[str] = Field(
|
||||
None,
|
||||
description="The event_id token to be used to resume the interaction stream.",
|
||||
)
|
||||
|
||||
|
||||
class InteractionSseEvent(
|
||||
RootModel[
|
||||
Union[
|
||||
# New schema events (Api-Revision: 2026-05-20)
|
||||
StepStart,
|
||||
StepDelta,
|
||||
StepStop,
|
||||
InteractionCreated,
|
||||
InteractionInProgress,
|
||||
InteractionCompleted,
|
||||
InteractionRequiresAction,
|
||||
# Legacy schema events (Api-Revision: 2026-05-07, removed June 8 2026)
|
||||
InteractionEvent,
|
||||
InteractionStatusUpdate,
|
||||
ContentStart,
|
||||
|
|
@ -1164,6 +1269,15 @@ class InteractionSseEvent(
|
|||
]
|
||||
):
|
||||
root: Union[
|
||||
# New schema events (Api-Revision: 2026-05-20)
|
||||
StepStart,
|
||||
StepDelta,
|
||||
StepStop,
|
||||
InteractionCreated,
|
||||
InteractionInProgress,
|
||||
InteractionCompleted,
|
||||
InteractionRequiresAction,
|
||||
# Legacy schema events (Api-Revision: 2026-05-07, removed June 8 2026)
|
||||
InteractionEvent,
|
||||
InteractionStatusUpdate,
|
||||
ContentStart,
|
||||
|
|
@ -1193,6 +1307,11 @@ class InteractionsAPIResponse(BaseLiteLLMOpenAIResponseObject):
|
|||
Response from the Interactions API.
|
||||
|
||||
Wraps the API response with LiteLLM-specific hidden params.
|
||||
|
||||
Schema notes:
|
||||
- New schema (Api-Revision: 2026-05-20, default): response contains ``steps``.
|
||||
- Legacy schema (Api-Revision: 2026-05-07, removed June 8 2026): response contains ``outputs``.
|
||||
Both fields are kept here so callers work with either schema.
|
||||
"""
|
||||
|
||||
id: Optional[str] = None
|
||||
|
|
@ -1203,7 +1322,10 @@ class InteractionsAPIResponse(BaseLiteLLMOpenAIResponseObject):
|
|||
created: Optional[str] = None
|
||||
updated: Optional[str] = None
|
||||
role: Optional[str] = None
|
||||
# Legacy schema field (Api-Revision: 2026-05-07). Remove after June 8, 2026.
|
||||
outputs: Optional[List[Dict[str, Any]]] = None
|
||||
# New schema field (Api-Revision: 2026-05-20).
|
||||
steps: Optional[List[Dict[str, Any]]] = None
|
||||
usage: Optional[Dict[str, Any]] = None
|
||||
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
|
@ -1213,7 +1335,12 @@ class InteractionsAPIStreamingResponse(BaseLiteLLMOpenAIResponseObject):
|
|||
"""
|
||||
Streaming response chunk from the Interactions API.
|
||||
|
||||
Event types per OpenAPI spec:
|
||||
New schema event types (Api-Revision: 2026-05-20):
|
||||
- interaction.created, interaction.in_progress, interaction.completed,
|
||||
interaction.requires_action
|
||||
- step.start, step.delta, step.stop
|
||||
|
||||
Legacy event types (Api-Revision: 2026-05-07, removed June 8 2026):
|
||||
- interaction.start, interaction.status_update, interaction.complete
|
||||
- content.start, content.delta, content.stop
|
||||
- error
|
||||
|
|
@ -1228,9 +1355,17 @@ class InteractionsAPIStreamingResponse(BaseLiteLLMOpenAIResponseObject):
|
|||
created: Optional[str] = None
|
||||
updated: Optional[str] = None
|
||||
role: Optional[str] = None
|
||||
# Legacy schema field (Api-Revision: 2026-05-07). Remove after June 8, 2026.
|
||||
outputs: Optional[List[Dict[str, Any]]] = None
|
||||
# New schema field (Api-Revision: 2026-05-20).
|
||||
steps: Optional[List[Dict[str, Any]]] = None
|
||||
usage: Optional[Dict[str, Any]] = None
|
||||
delta: Optional[Dict[str, Any]] = None
|
||||
# New schema streaming fields
|
||||
index: Optional[int] = None
|
||||
step: Optional[Dict[str, Any]] = None
|
||||
interaction_id: Optional[str] = None
|
||||
interaction: Optional[Dict[str, Any]] = None
|
||||
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,10 +1,11 @@
|
|||
"""
|
||||
Tests for Gemini Interactions API transformation.
|
||||
|
||||
Covers credential leak prevention changes:
|
||||
- validate_environment sets x-goog-api-key header
|
||||
- get_complete_url excludes API key from URL
|
||||
- get/delete/cancel interaction request URLs exclude API key
|
||||
Covers:
|
||||
- validate_environment: x-goog-api-key header, Api-Revision schema selection
|
||||
- get_complete_url: API key excluded from URL
|
||||
- get/delete/cancel interaction request URLs
|
||||
- transform_request: response_mime_type coalescing, image_config migration
|
||||
"""
|
||||
|
||||
import os
|
||||
|
|
@ -15,6 +16,7 @@ import pytest
|
|||
|
||||
sys.path.insert(0, os.path.abspath("../../.."))
|
||||
|
||||
import litellm
|
||||
from litellm.interactions.litellm_responses_transformation.streaming_iterator import (
|
||||
LiteLLMResponsesInteractionsStreamingIterator,
|
||||
)
|
||||
|
|
@ -24,7 +26,6 @@ from litellm.llms.gemini.interactions.transformation import (
|
|||
from litellm.types.llms.openai import (
|
||||
ContentPartAddedEvent,
|
||||
OutputTextDeltaEvent,
|
||||
ResponseCompletedEvent,
|
||||
ResponseCreatedEvent,
|
||||
)
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
|
@ -85,6 +86,30 @@ class TestValidateEnvironment:
|
|||
assert headers["X-Custom"] == "value"
|
||||
assert headers["x-goog-api-key"] == "test-key"
|
||||
|
||||
def test_api_revision_new_schema_by_default(self, config):
|
||||
# Default: use_legacy_interactions_schema=False → new steps schema
|
||||
original = litellm.use_legacy_interactions_schema
|
||||
try:
|
||||
litellm.use_legacy_interactions_schema = False
|
||||
headers = config.validate_environment(
|
||||
headers={}, model="gemini-2.5-flash", litellm_params=None
|
||||
)
|
||||
assert headers["Api-Revision"] == "2026-05-20"
|
||||
finally:
|
||||
litellm.use_legacy_interactions_schema = original
|
||||
|
||||
def test_api_revision_legacy_schema_when_flag_set(self, config):
|
||||
# Flag on → legacy outputs schema until June 8, 2026
|
||||
original = litellm.use_legacy_interactions_schema
|
||||
try:
|
||||
litellm.use_legacy_interactions_schema = True
|
||||
headers = config.validate_environment(
|
||||
headers={}, model="gemini-2.5-flash", litellm_params=None
|
||||
)
|
||||
assert headers["Api-Revision"] == "2026-05-07"
|
||||
finally:
|
||||
litellm.use_legacy_interactions_schema = original
|
||||
|
||||
|
||||
class TestGetCompleteUrl:
|
||||
def test_url_excludes_api_key(self, config):
|
||||
|
|
@ -172,6 +197,35 @@ class TestTransformRequest:
|
|||
)
|
||||
|
||||
assert request_body["environment"] == env_id
|
||||
|
||||
def test_stream_param_included_in_request_body(self, config):
|
||||
"""When stream=True is in optional_params, the request body must include it
|
||||
so the proxy forwards the SSE streaming flag to Google's backend."""
|
||||
body = config.transform_request(
|
||||
model="gemini-2.5-flash",
|
||||
agent=None,
|
||||
input="Hello",
|
||||
optional_params={"stream": True},
|
||||
litellm_params=GenericLiteLLMParams(api_key="test-key"),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert body.get("stream") is True
|
||||
assert body.get("input") == "Hello"
|
||||
|
||||
def test_stream_false_not_included_when_absent(self, config):
|
||||
body = config.transform_request(
|
||||
model="gemini-2.5-flash",
|
||||
agent=None,
|
||||
input="Hello",
|
||||
optional_params={},
|
||||
litellm_params=GenericLiteLLMParams(api_key="test-key"),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert "stream" not in body
|
||||
|
||||
|
||||
class TestStreamingIterator:
|
||||
def _make_iterator(self) -> LiteLLMResponsesInteractionsStreamingIterator:
|
||||
return LiteLLMResponsesInteractionsStreamingIterator(
|
||||
|
|
@ -323,35 +377,6 @@ class TestStreamingIterator:
|
|||
assert second.delta == {"type": "text", "text": " World"}
|
||||
|
||||
|
||||
class TestTransformRequest:
|
||||
def test_stream_param_included_in_request_body(self, config):
|
||||
"""When stream=True is in optional_params, the request body must include it
|
||||
so the proxy forwards the SSE streaming flag to Google's backend."""
|
||||
body = config.transform_request(
|
||||
model="gemini-2.5-flash",
|
||||
agent=None,
|
||||
input="Hello",
|
||||
optional_params={"stream": True},
|
||||
litellm_params=GenericLiteLLMParams(api_key="test-key"),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert body.get("stream") is True
|
||||
assert body.get("input") == "Hello"
|
||||
|
||||
def test_stream_false_not_included_when_absent(self, config):
|
||||
body = config.transform_request(
|
||||
model="gemini-2.5-flash",
|
||||
agent=None,
|
||||
input="Hello",
|
||||
optional_params={},
|
||||
litellm_params=GenericLiteLLMParams(api_key="test-key"),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert "stream" not in body
|
||||
|
||||
|
||||
class TestInteractionOperationUrls:
|
||||
"""Test that get/delete/cancel interaction URLs exclude API key."""
|
||||
|
||||
|
|
@ -410,3 +435,80 @@ class TestInteractionOperationUrls:
|
|||
litellm_params=GenericLiteLLMParams(api_key=None),
|
||||
headers={},
|
||||
)
|
||||
|
||||
|
||||
class TestTransformRequestSchemaCoalescing:
|
||||
"""Test new-schema request coalescing (Api-Revision: 2026-05-20)."""
|
||||
|
||||
def test_response_mime_type_folded_into_response_format(self, config):
|
||||
original = litellm.use_legacy_interactions_schema
|
||||
try:
|
||||
litellm.use_legacy_interactions_schema = False
|
||||
body = config.transform_request(
|
||||
model="gemini/gemini-2.5-flash",
|
||||
agent=None,
|
||||
input="summarise",
|
||||
optional_params={
|
||||
"response_mime_type": "application/json",
|
||||
"response_format": {"type": "object", "properties": {}},
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
finally:
|
||||
litellm.use_legacy_interactions_schema = original
|
||||
|
||||
# response_mime_type must not appear as a top-level body key
|
||||
assert "response_mime_type" not in body
|
||||
rf = body["response_format"]
|
||||
assert rf["type"] == "text"
|
||||
assert rf["mime_type"] == "application/json"
|
||||
assert "schema" in rf
|
||||
|
||||
def test_image_config_moved_to_response_format(self, config):
|
||||
original = litellm.use_legacy_interactions_schema
|
||||
try:
|
||||
litellm.use_legacy_interactions_schema = False
|
||||
body = config.transform_request(
|
||||
model="gemini/gemini-2.5-flash",
|
||||
agent=None,
|
||||
input="draw a sunset",
|
||||
optional_params={
|
||||
"generation_config": {
|
||||
"temperature": 0.7,
|
||||
"image_config": {"aspect_ratio": "1:1", "image_size": "1K"},
|
||||
}
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
finally:
|
||||
litellm.use_legacy_interactions_schema = original
|
||||
|
||||
# image_config removed from generation_config
|
||||
assert "image_config" not in body.get("generation_config", {})
|
||||
# moved into response_format with type=image
|
||||
rf = body["response_format"]
|
||||
assert rf["type"] == "image"
|
||||
assert rf["aspect_ratio"] == "1:1"
|
||||
|
||||
def test_legacy_schema_passes_fields_unchanged(self, config):
|
||||
original = litellm.use_legacy_interactions_schema
|
||||
try:
|
||||
litellm.use_legacy_interactions_schema = True
|
||||
body = config.transform_request(
|
||||
model="gemini/gemini-2.5-flash",
|
||||
agent=None,
|
||||
input="hello",
|
||||
optional_params={
|
||||
"response_mime_type": "application/json",
|
||||
"generation_config": {"image_config": {"aspect_ratio": "16:9"}},
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
finally:
|
||||
litellm.use_legacy_interactions_schema = original
|
||||
|
||||
assert body["response_mime_type"] == "application/json"
|
||||
assert body["generation_config"]["image_config"]["aspect_ratio"] == "16:9"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue