diff --git a/litellm/litellm_core_utils/llm_judge.py b/litellm/litellm_core_utils/llm_judge.py index b632d3a9af9..3a4091ce345 100644 --- a/litellm/litellm_core_utils/llm_judge.py +++ b/litellm/litellm_core_utils/llm_judge.py @@ -15,7 +15,11 @@ if TYPE_CHECKING: from litellm.types.llms.openai import AllMessageValues from litellm.types.utils import ModelResponse -JSON_FENCE_RE: Final = re.compile(r"```(?:json)?\s*(.*?)\s*```", re.DOTALL | re.IGNORECASE) +# No `\s*` around the capture: under DOTALL a dot matches a space too, so each +# `\s*` overlaps the `.*?` beside it, and a reply that opens a fence without closing +# one can be split between them in cubic-many ways -- a failing match walks all of +# them. The only caller strips the capture, which is what those `\s*` were doing. +JSON_FENCE_RE: Final = re.compile(r"```(?:json)?(.*?)```", re.DOTALL | re.IGNORECASE) def default_router_provider() -> Router | None: diff --git a/tests/test_litellm/litellm_core_utils/test_llm_judge.py b/tests/test_litellm/litellm_core_utils/test_llm_judge.py index 3bcfde76450..d82bafd1c96 100644 --- a/tests/test_litellm/litellm_core_utils/test_llm_judge.py +++ b/tests/test_litellm/litellm_core_utils/test_llm_judge.py @@ -1,6 +1,8 @@ """Unit tests for the shared LLM-judge primitives: verdict parsing, router resolution, dispatch.""" import json +import time +from typing import Final from unittest.mock import AsyncMock, MagicMock import pytest @@ -27,6 +29,39 @@ def test_parse_json_verdict_tolerates_fences_and_prose(raw, expected): assert parse_json_verdict(raw)["preference"] == expected +@pytest.mark.parametrize( + "raw,expected", + [ + ('```json {"preference": "A"} ```', "A"), + ('```\n\n {"preference": "B"} \n\n```', "B"), + ('```json{"preference": "tie"}```', "tie"), + ], +) +def test_parse_json_verdict_still_trims_padding_inside_the_fence(raw, expected): + """Whitespace between the fence and the JSON is dropped, however it is spelled.""" + assert parse_json_verdict(raw)["preference"] == expected + + +def test_parse_json_verdict_is_not_quadratic_in_an_unclosed_fence(): + """A judge reply that opens a fence and never closes it must not burn the event loop. + + The judge is fed the end user's own text, so the reply it produces is steerable by + that user. The fence regex used to put a `\\s*` on each side of its capture; under + DOTALL those overlap the capture, and a failing match walks cubic-many ways to split + the whitespace run between them -- 4KB of padding took over a minute of CPU, on the + loop, inside an async guardrail hook. The budget below is ~500x the fixed cost and + ~1/14th the unfixed one, so it separates the two without depending on runner speed. + """ + reply: Final = "```" + " " * 4000 + "no closing fence" + + start: Final = time.perf_counter() + with pytest.raises((json.JSONDecodeError, ValueError)): + parse_json_verdict(reply) + elapsed: Final = time.perf_counter() - start + + assert elapsed < 5.0, f"parsing a 4KB unclosed fence took {elapsed:.1f}s" + + def test_parse_json_verdict_rejects_non_object(): with pytest.raises(ValueError, match='judge response is not a JSON object'): parse_json_verdict('["not", "an", "object"]')