diff --git a/strix/config/models.py b/strix/config/models.py index 6f0162a00..ea5d574c5 100644 --- a/strix/config/models.py +++ b/strix/config/models.py @@ -807,7 +807,7 @@ def _install_openrouter_stream_cost_capture() -> None: # survives between turns. body = super().transform_request(*args, **kwargs) agent_id = request_log.current_call_context().agent_id - if agent_id: + if agent_id and load_settings().llm.openrouter_sticky_sessions: session_id = _OPENROUTER_SESSION_IDS.setdefault(agent_id, str(uuid.uuid4())) body.setdefault("session_id", session_id) return body diff --git a/strix/config/settings.py b/strix/config/settings.py index f6c85dd03..69e8575c0 100644 --- a/strix/config/settings.py +++ b/strix/config/settings.py @@ -62,6 +62,10 @@ class LlmSettings(BaseSettings): # can read back up to a block short. 128 covers the largest common size # (OpenAI; DeepSeek and GLM use 64, vLLM defaults to 16). cache_block_tokens: int = Field(default=128, ge=1, alias="STRIX_CACHE_BLOCK_TOKENS") + openrouter_sticky_sessions: bool = Field( + default=True, + alias="STRIX_OPENROUTER_STICKY_SESSIONS", + ) disable_streaming: bool = Field( default=False, alias="LLM_DISABLE_STREAMING", diff --git a/tests/test_cost_tracking.py b/tests/test_cost_tracking.py index 9d7ea6db4..5b108abde 100644 --- a/tests/test_cost_tracking.py +++ b/tests/test_cost_tracking.py @@ -384,5 +384,8 @@ def test_openrouter_request_carries_agent_session_id() -> None: session_id = body()["session_id"] assert str(uuid.UUID(session_id)) == session_id assert body()["session_id"] == session_id + with patch("strix.config.models.load_settings") as settings: + settings.return_value.llm.openrouter_sticky_sessions = False + assert "session_id" not in body() finally: request_log.reset_call_context(token)