mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(chatgpt): derive Responses session id from prompt_cache_key
ChatGPT partitions the Codex prompt cache by the Responses session-id header (the Codex CLI sends a stable per-conversation key derived from prompt_cache_key). The provider only consulted explicit session ids and otherwise stamped a per-request uuid4, so no turn ever reused the prompt cache -- 0% cache hits on repeat turns, an order-of-magnitude quota/cost multiplier on subscription accounts. Resolution order after this change: explicit session config (session_id / litellm_session_id / metadata.session_id) > prompt_cache_key (sanitized with _safe_header_value) > litellm-internal ids > uuid4. Two guards, mirroring get_fireworks_session_id: - proxy-generated session ids (missing_session_id: "generate") are per-request and are skipped, checked in both "metadata" and "litellm_metadata" (LITELLM_METADATA_ROUTES, responses included) - internal per-request ids (litellm_trace_id / litellm_call_id) lose to a stable conversation key Supersedes the body-key forwarding of #37630: a random session header overrides the body key on the provider side, so the header must carry the caller's key. Related: #29978, #29993. Verified against a real ChatGPT Codex account: stable session id -> 97-100% cached on repeat turns; random per-request -> 0%.
This commit is contained in:
parent
615ed7900f
commit
5e187903a4
2 changed files with 94 additions and 8 deletions
|
|
@ -9,6 +9,7 @@ from uuid import uuid4
|
|||
|
||||
import httpx
|
||||
|
||||
from litellm.constants import SESSION_ID_GENERATED_METADATA_KEY
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
# OAuth + API constants (derived from openai/codex)
|
||||
|
|
@ -270,15 +271,41 @@ def _normalize_litellm_params(litellm_params: Any | None) -> dict:
|
|||
|
||||
def get_chatgpt_session_id(litellm_params: object) -> str | None:
|
||||
params: Final = _normalize_litellm_params(litellm_params)
|
||||
for key in ("litellm_session_id", "session_id"):
|
||||
value = params.get(key)
|
||||
if value:
|
||||
return str(value)
|
||||
metadata: Final = params.get("metadata")
|
||||
if isinstance(metadata, dict):
|
||||
value = metadata.get("session_id")
|
||||
if value:
|
||||
return str(value)
|
||||
# A session id the proxy generated for a request that had none
|
||||
# (general_settings.missing_session_id: "generate") is per-request; using
|
||||
# it as identity pins every request to a different ChatGPT cache shard
|
||||
# and the prompt cache never hits (same guard as fireworks'
|
||||
# get_fireworks_session_id). The marker can sit in "metadata" or
|
||||
# "litellm_metadata" -- the LITELLM_METADATA_ROUTES (responses included)
|
||||
# carry internal metadata under the latter. Generated ids are skipped,
|
||||
# not returned, so a caller-supplied stable prompt_cache_key still wins.
|
||||
generated: Final = any(
|
||||
isinstance(params.get(name), dict)
|
||||
and params[name].get(SESSION_ID_GENERATED_METADATA_KEY)
|
||||
for name in ("metadata", "litellm_metadata")
|
||||
)
|
||||
if not generated:
|
||||
for key in ("litellm_session_id", "session_id"):
|
||||
value = params.get(key)
|
||||
if value:
|
||||
return str(value)
|
||||
if isinstance(metadata, dict):
|
||||
value = metadata.get("session_id")
|
||||
if value:
|
||||
return str(value)
|
||||
# ChatGPT derives prompt-cache affinity from the Responses session-id
|
||||
# header (the Codex CLI sends a stable per-conversation key derived from
|
||||
# prompt_cache_key). Callers of the responses API already send a stable
|
||||
# prompt_cache_key; retaining it as the session id lets repeat turns of a
|
||||
# conversation reuse the provider's prompt cache instead of landing on a
|
||||
# fresh shard every request. Explicit operator configuration above still
|
||||
# wins; litellm-internal per-request ids below still lose to it.
|
||||
prompt_cache_key: Final = params.get("prompt_cache_key")
|
||||
if prompt_cache_key:
|
||||
return _safe_header_value(str(prompt_cache_key)) or None
|
||||
if generated:
|
||||
return None
|
||||
for key in ("litellm_trace_id", "litellm_call_id"):
|
||||
value = params.get(key)
|
||||
if value:
|
||||
|
|
|
|||
59
tests/unit/llms/chatgpt/test_session_id.py
Normal file
59
tests/unit/llms/chatgpt/test_session_id.py
Normal file
|
|
@ -0,0 +1,59 @@
|
|||
"""Session-id resolution for the ChatGPT provider.
|
||||
|
||||
ChatGPT partitions the Codex prompt cache by the Responses session-id
|
||||
header. These tests pin the resolution order:
|
||||
explicit operator config > prompt_cache_key > litellm-internal ids,
|
||||
with proxy-generated session ids skipped entirely.
|
||||
"""
|
||||
|
||||
from litellm.llms.chatgpt.common_utils import (
|
||||
ensure_chatgpt_session_id,
|
||||
get_chatgpt_session_id,
|
||||
)
|
||||
|
||||
|
||||
def test_explicit_session_ids_win():
|
||||
assert get_chatgpt_session_id({"session_id": "s", "prompt_cache_key": "k"}) == "s"
|
||||
assert get_chatgpt_session_id({"litellm_session_id": "ls", "prompt_cache_key": "k"}) == "ls"
|
||||
assert (
|
||||
get_chatgpt_session_id({"metadata": {"session_id": "ms"}, "prompt_cache_key": "k"})
|
||||
== "ms"
|
||||
)
|
||||
|
||||
|
||||
def test_prompt_cache_key_becomes_session_id():
|
||||
assert get_chatgpt_session_id({"prompt_cache_key": "conv-abc"}) == "conv-abc"
|
||||
assert ensure_chatgpt_session_id({"prompt_cache_key": "conv-abc"}) == "conv-abc"
|
||||
|
||||
|
||||
def test_prompt_cache_key_beats_internal_request_ids():
|
||||
# litellm_trace_id / litellm_call_id are per-request; they must not
|
||||
# shadow a stable per-conversation key.
|
||||
assert get_chatgpt_session_id({"litellm_call_id": "c", "prompt_cache_key": "k"}) == "k"
|
||||
assert get_chatgpt_session_id({"litellm_trace_id": "t", "prompt_cache_key": "k"}) == "k"
|
||||
|
||||
|
||||
def test_prompt_cache_key_is_sanitized_for_header_use():
|
||||
# Caller-controlled value headed for an HTTP header: control chars are
|
||||
# replaced (h11 field-value validation would otherwise 500 the request),
|
||||
# keeping the key stable.
|
||||
assert get_chatgpt_session_id({"prompt_cache_key": "a\nb"}) == "a_b"
|
||||
|
||||
|
||||
def test_generated_session_ids_are_skipped():
|
||||
# missing_session_id: "generate" stamps a per-request id in every channel;
|
||||
# it must not be used as identity, and the cache key must still win.
|
||||
for metadata_key in ("metadata", "litellm_metadata"):
|
||||
generated = {
|
||||
"litellm_session_id": "gen",
|
||||
metadata_key: {"session_id": "gen", "litellm_session_id_generated": True},
|
||||
}
|
||||
assert get_chatgpt_session_id({**generated, "prompt_cache_key": "k"}) == "k"
|
||||
assert get_chatgpt_session_id(generated) is None
|
||||
a, b = ensure_chatgpt_session_id(generated), ensure_chatgpt_session_id(generated)
|
||||
assert a != "gen" and b != "gen" and a != b
|
||||
|
||||
|
||||
def test_uuid4_fallback_without_any_key():
|
||||
a, b = ensure_chatgpt_session_id({}), ensure_chatgpt_session_id({})
|
||||
assert a and b and a != b
|
||||
Loading…
Add table
Reference in a new issue