fix(mcp): redact provider error from client-facing semantic filter message

Keep the full provider exception in server-side logs only; the client
receives a fixed actionable message. Also follow implicit exception
context when detecting context window overflows and pin the detection
variants plus the redaction in tests
This commit is contained in:
Tin Chi Lo 2026-07-09 20:35:57 -07:00
parent 1e8c2f7240
commit 898182b0e6
2 changed files with 48 additions and 2 deletions

View file

@ -29,7 +29,7 @@ class SemanticToolFilterContextWindowError(Exception):
f"exceeded its context window while embedding {stage}. "
f"The request was blocked instead of silently passing all tools through. "
f"Switch to an embedding model with a larger context window, or disable "
f"semantic tool filtering. Original error: {original_error}"
f"semantic tool filtering."
)
@ -41,7 +41,7 @@ def _is_context_window_error(error: Optional[BaseException], depth: int = 5) ->
return True
if ExceptionCheckers.is_error_str_context_window_exceeded(str(error)):
return True
return _is_context_window_error(error.__cause__, depth - 1)
return _is_context_window_error(error.__cause__ or error.__context__, depth - 1)
class SemanticMCPToolFilter:
@ -231,6 +231,10 @@ class SemanticMCPToolFilter:
except Exception as e:
if _is_context_window_error(e):
verbose_logger.error(
f"Semantic tool filter embedding exceeded its context window: {e}",
exc_info=True,
)
raise SemanticToolFilterContextWindowError(
embedding_model=self.embedding_model,
stage="the user query",

View file

@ -1510,6 +1510,7 @@ async def test_semantic_filter_hook_fails_closed_on_context_window_error():
assert "context window" in error_message
assert "text-embedding-3-small" in error_message
assert "larger context window" in error_message
assert "maximum input length" not in error_message
print("✅ Hook fails closed with actionable 400 on context window overflow")
@ -1575,6 +1576,7 @@ async def test_semantic_filter_hook_fails_closed_on_expanded_tools_context_windo
error_message = exc_info.value.detail["error"]
assert "context window" in error_message
assert "larger context window" in error_message
assert "maximum input length" not in error_message
print("✅ Expansion path fails closed with actionable 400 on context window overflow")
@ -1622,3 +1624,43 @@ async def test_semantic_filter_hook_ignores_build_error_for_native_only_tools():
assert result is not None
assert result["tools"] == native_tools
print("✅ Native-only requests pass through despite recorded build error")
def test_is_context_window_error_detection_variants():
"""
_is_context_window_error must detect the overflow in every shape it
reaches filter_tools in: the raw typed exception, the encoder's
explicitly chained ValueError wrapper, an implicitly chained wrapper,
and a bare error whose message carries a known overflow phrase; a
generic error must not match.
"""
import litellm
from litellm.proxy._experimental.mcp_server.semantic_tool_filter import (
_is_context_window_error,
)
cwe = litellm.ContextWindowExceededError(
message="Invalid 'input[0]': maximum input length is 8192 tokens.",
model="text-embedding-3-small",
llm_provider="openai",
)
assert _is_context_window_error(cwe)
try:
raise ValueError("Internal_litellm_router API call failed") from cwe
except ValueError as explicitly_chained:
assert _is_context_window_error(explicitly_chained)
try:
try:
raise litellm.ContextWindowExceededError(
message="overflow", model="m", llm_provider="openai"
)
except litellm.ContextWindowExceededError:
raise ValueError("wrapper without explicit chaining")
except ValueError as implicitly_chained:
assert _is_context_window_error(implicitly_chained)
assert _is_context_window_error(ValueError("Invalid 'input[0]': maximum input length is 8192 tokens."))
assert not _is_context_window_error(ValueError("A generic API error occurred."))
assert not _is_context_window_error(None)