From 5e187903a42d8ea3f8f8e33af60f65f507338cca Mon Sep 17 00:00:00 2001 From: Onur Cakmak Date: Sat, 19 Sep 2026 15:55:33 -0400 Subject: [PATCH] fix(chatgpt): derive Responses session id from prompt_cache_key ChatGPT partitions the Codex prompt cache by the Responses session-id header (the Codex CLI sends a stable per-conversation key derived from prompt_cache_key). The provider only consulted explicit session ids and otherwise stamped a per-request uuid4, so no turn ever reused the prompt cache -- 0% cache hits on repeat turns, an order-of-magnitude quota/cost multiplier on subscription accounts. Resolution order after this change: explicit session config (session_id / litellm_session_id / metadata.session_id) > prompt_cache_key (sanitized with _safe_header_value) > litellm-internal ids > uuid4. Two guards, mirroring get_fireworks_session_id: - proxy-generated session ids (missing_session_id: "generate") are per-request and are skipped, checked in both "metadata" and "litellm_metadata" (LITELLM_METADATA_ROUTES, responses included) - internal per-request ids (litellm_trace_id / litellm_call_id) lose to a stable conversation key Supersedes the body-key forwarding of #37630: a random session header overrides the body key on the provider side, so the header must carry the caller's key. Related: #29978, #29993. Verified against a real ChatGPT Codex account: stable session id -> 97-100% cached on repeat turns; random per-request -> 0%. --- litellm/llms/chatgpt/common_utils.py | 43 +++++++++++++--- tests/unit/llms/chatgpt/test_session_id.py | 59 ++++++++++++++++++++++ 2 files changed, 94 insertions(+), 8 deletions(-) create mode 100644 tests/unit/llms/chatgpt/test_session_id.py diff --git a/litellm/llms/chatgpt/common_utils.py b/litellm/llms/chatgpt/common_utils.py index fe33219f110..7717ec629c9 100644 --- a/litellm/llms/chatgpt/common_utils.py +++ b/litellm/llms/chatgpt/common_utils.py @@ -9,6 +9,7 @@ from uuid import uuid4 import httpx +from litellm.constants import SESSION_ID_GENERATED_METADATA_KEY from litellm.llms.base_llm.chat.transformation import BaseLLMException # OAuth + API constants (derived from openai/codex) @@ -270,15 +271,41 @@ def _normalize_litellm_params(litellm_params: Any | None) -> dict: def get_chatgpt_session_id(litellm_params: object) -> str | None: params: Final = _normalize_litellm_params(litellm_params) - for key in ("litellm_session_id", "session_id"): - value = params.get(key) - if value: - return str(value) metadata: Final = params.get("metadata") - if isinstance(metadata, dict): - value = metadata.get("session_id") - if value: - return str(value) + # A session id the proxy generated for a request that had none + # (general_settings.missing_session_id: "generate") is per-request; using + # it as identity pins every request to a different ChatGPT cache shard + # and the prompt cache never hits (same guard as fireworks' + # get_fireworks_session_id). The marker can sit in "metadata" or + # "litellm_metadata" -- the LITELLM_METADATA_ROUTES (responses included) + # carry internal metadata under the latter. Generated ids are skipped, + # not returned, so a caller-supplied stable prompt_cache_key still wins. + generated: Final = any( + isinstance(params.get(name), dict) + and params[name].get(SESSION_ID_GENERATED_METADATA_KEY) + for name in ("metadata", "litellm_metadata") + ) + if not generated: + for key in ("litellm_session_id", "session_id"): + value = params.get(key) + if value: + return str(value) + if isinstance(metadata, dict): + value = metadata.get("session_id") + if value: + return str(value) + # ChatGPT derives prompt-cache affinity from the Responses session-id + # header (the Codex CLI sends a stable per-conversation key derived from + # prompt_cache_key). Callers of the responses API already send a stable + # prompt_cache_key; retaining it as the session id lets repeat turns of a + # conversation reuse the provider's prompt cache instead of landing on a + # fresh shard every request. Explicit operator configuration above still + # wins; litellm-internal per-request ids below still lose to it. + prompt_cache_key: Final = params.get("prompt_cache_key") + if prompt_cache_key: + return _safe_header_value(str(prompt_cache_key)) or None + if generated: + return None for key in ("litellm_trace_id", "litellm_call_id"): value = params.get(key) if value: diff --git a/tests/unit/llms/chatgpt/test_session_id.py b/tests/unit/llms/chatgpt/test_session_id.py new file mode 100644 index 00000000000..3a07f593dff --- /dev/null +++ b/tests/unit/llms/chatgpt/test_session_id.py @@ -0,0 +1,59 @@ +"""Session-id resolution for the ChatGPT provider. + +ChatGPT partitions the Codex prompt cache by the Responses session-id +header. These tests pin the resolution order: +explicit operator config > prompt_cache_key > litellm-internal ids, +with proxy-generated session ids skipped entirely. +""" + +from litellm.llms.chatgpt.common_utils import ( + ensure_chatgpt_session_id, + get_chatgpt_session_id, +) + + +def test_explicit_session_ids_win(): + assert get_chatgpt_session_id({"session_id": "s", "prompt_cache_key": "k"}) == "s" + assert get_chatgpt_session_id({"litellm_session_id": "ls", "prompt_cache_key": "k"}) == "ls" + assert ( + get_chatgpt_session_id({"metadata": {"session_id": "ms"}, "prompt_cache_key": "k"}) + == "ms" + ) + + +def test_prompt_cache_key_becomes_session_id(): + assert get_chatgpt_session_id({"prompt_cache_key": "conv-abc"}) == "conv-abc" + assert ensure_chatgpt_session_id({"prompt_cache_key": "conv-abc"}) == "conv-abc" + + +def test_prompt_cache_key_beats_internal_request_ids(): + # litellm_trace_id / litellm_call_id are per-request; they must not + # shadow a stable per-conversation key. + assert get_chatgpt_session_id({"litellm_call_id": "c", "prompt_cache_key": "k"}) == "k" + assert get_chatgpt_session_id({"litellm_trace_id": "t", "prompt_cache_key": "k"}) == "k" + + +def test_prompt_cache_key_is_sanitized_for_header_use(): + # Caller-controlled value headed for an HTTP header: control chars are + # replaced (h11 field-value validation would otherwise 500 the request), + # keeping the key stable. + assert get_chatgpt_session_id({"prompt_cache_key": "a\nb"}) == "a_b" + + +def test_generated_session_ids_are_skipped(): + # missing_session_id: "generate" stamps a per-request id in every channel; + # it must not be used as identity, and the cache key must still win. + for metadata_key in ("metadata", "litellm_metadata"): + generated = { + "litellm_session_id": "gen", + metadata_key: {"session_id": "gen", "litellm_session_id_generated": True}, + } + assert get_chatgpt_session_id({**generated, "prompt_cache_key": "k"}) == "k" + assert get_chatgpt_session_id(generated) is None + a, b = ensure_chatgpt_session_id(generated), ensure_chatgpt_session_id(generated) + assert a != "gen" and b != "gen" and a != b + + +def test_uuid4_fallback_without_any_key(): + a, b = ensure_chatgpt_session_id({}), ensure_chatgpt_session_id({}) + assert a and b and a != b