From 4673754857e8fec1506caef7da6bce74ea0748bb Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Mon, 13 Jul 2026 18:16:15 -0700 Subject: [PATCH] test(e2e): run arize_phoenix suite on the otel_v2 pipeline The logging client registers the suite's Claude deployments at build time and tests call models by their plain provider-qualified names. Phoenix read-back uses the same env vars as the proxy's arize_phoenix integration and matches generation spans by span kind plus a unique prompt marker, since otel_v2 names spans "chat {model}" and leaves successful spans' status UNSET. The compose stack passes the phoenix env through and opts into span message content so prompts and completions land on traces; the shared conftest loads tests/e2e/.env so bare pytest runs work locally --- tests/e2e/conftest.py | 2 + tests/e2e/docker-compose.yml | 5 + tests/e2e/logging/logging_client.py | 217 ++++++------------ tests/e2e/logging/test_langfuse_e2e.py | 67 +----- .../e2e/logging/test_logging_arize_phoenix.py | 98 ++++---- tests/e2e/models.py | 4 +- 6 files changed, 129 insertions(+), 264 deletions(-) diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index 08df334d4b8..15c6d02e1b9 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -20,6 +20,7 @@ from typing import Iterator import pytest import requests +from dotenv import load_dotenv from e2e_config import CONTROL_PLANE_BASE_URL, PROXY_BASE_URL from lifecycle import GatewayProvider, ResourceManager @@ -29,6 +30,7 @@ _E2E_TEST_RAN = pytest.StashKey[bool]() def pytest_configure(config: pytest.Config) -> None: + load_dotenv(Path(__file__).parent / ".env") config.addinivalue_line( "markers", "e2e: live test that requires a running proxy and real provider keys", diff --git a/tests/e2e/docker-compose.yml b/tests/e2e/docker-compose.yml index 0cfbb5b0b66..08770c8a67f 100644 --- a/tests/e2e/docker-compose.yml +++ b/tests/e2e/docker-compose.yml @@ -10,6 +10,7 @@ configs: proxy_budget_rescheduler_max_time: 10 litellm_settings: + callbacks: ["prometheus", "datadog", "arize_phoenix"] drop_params: true num_retries: 3 request_timeout: 600 @@ -87,6 +88,10 @@ services: MISTRAL_API_KEY: ${MISTRAL_API_KEY:-} AZURE_API_BASE: ${AZURE_API_BASE:-} AZURE_API_KEY: ${AZURE_API_KEY:-} + PHOENIX_COLLECTOR_ENDPOINT: ${PHOENIX_COLLECTOR_ENDPOINT:-} + PHOENIX_API_KEY: ${PHOENIX_API_KEY:-} + PHOENIX_PROJECT_NAME: ${PHOENIX_PROJECT_NAME:-} + OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT: span_only ports: - "4000:4000" configs: diff --git a/tests/e2e/logging/logging_client.py b/tests/e2e/logging/logging_client.py index c032931e961..7ea956b1d29 100644 --- a/tests/e2e/logging/logging_client.py +++ b/tests/e2e/logging/logging_client.py @@ -1,14 +1,10 @@ -"""Client for the logging e2e suite: team/key/org-scoped Langfuse OTEL callbacks, -chat (including tools), Prometheus scrape, and Langfuse observation read-back. +"""Client for the logging e2e suite. -Holds the shared Gateway so the ``resources`` fixture cleans up keys, teams, -users, orgs, and models it creates. External Langfuse reads go through -``e2e_http`` (the only module allowed to call ``requests.*``). +The suite drives the otel_v2 logging pipeline; providers covered are langfuse, +arize_phoenix, datadog, and prometheus. -Uses the ``langfuse_otel`` callback (OTLP to ``{host}/api/public/otel``), not -the classic ``langfuse`` SDK callback. OTEL generations land as name -``litellm_request``; correlate by unique prompt marker and ``user_api_key_alias`` -in metadata. Spend is on ``calculatedTotalCost`` (StandardLogging response_cost). +Holds the shared Gateway so the ResourceManager cleans up keys, teams, users, +orgs, and models it creates. """ from __future__ import annotations @@ -21,7 +17,7 @@ from dataclasses import dataclass from typing import Literal import pytest -from pydantic import BaseModel, ConfigDict, Field, JsonValue, TypeAdapter, ValidationError +from pydantic import BaseModel, ConfigDict, Field, JsonValue, RootModel, TypeAdapter, ValidationError from e2e_config import POLL_INTERVAL, POLL_TIMEOUT from e2e_gateway import Gateway, build_gateway @@ -77,8 +73,6 @@ WEATHER_TOOL = ChatTool( ), ) -PHOENIX_GENERATION_SPAN = "litellm_request" -PHOENIX_PROXY_PARENT_SPAN = "litellm_proxy_request" _PHOENIX_MAX_PAGES = 20 CLAUDE_CODE_TOOLS = [ @@ -111,35 +105,9 @@ class TeamCallbackBody(BaseModel): class TeamCallbackResponse(BaseModel): model_config = ConfigDict(extra="ignore") - status: str -class GuardrailLitellmParams(BaseModel): - guardrail: str - mode: str - default_on: bool = False - rules: list[dict[str, object]] | None = None - default_action: str | None = None - on_disallowed_action: str | None = None - - -class GuardrailSpec(BaseModel): - guardrail_name: str - litellm_params: GuardrailLitellmParams - - -class CreateGuardrailBody(BaseModel): - guardrail: GuardrailSpec - - -class CreateGuardrailResponse(BaseModel): - model_config = ConfigDict(extra="ignore") - - guardrail_id: str | None = None - guardrail_name: str | None = None - - class LangfuseObservation(BaseModel): model_config = ConfigDict(extra="ignore", populate_by_name=True) @@ -206,36 +174,6 @@ class LangfuseCreds: ) -class PhoenixSpanContext(BaseModel): - model_config = ConfigDict(extra="ignore") - - trace_id: str - span_id: str - - -class PhoenixSpan(BaseModel): - model_config = ConfigDict(extra="ignore") - - name: str - context: PhoenixSpanContext - parent_id: str | None = None - status_code: str - attributes: dict[str, JsonValue] = {} - - -class PhoenixSpansPage(BaseModel): - model_config = ConfigDict(extra="ignore") - - data: list[PhoenixSpan] = [] - next_cursor: str | None = None - - -class PhoenixSpansParams(BaseModel): - limit: int = 100 - cursor: str | None = None - start_time: str | None = None - - @dataclass(frozen=True, slots=True) class PhoenixCreds: base_url: str @@ -250,15 +188,24 @@ class PhoenixCreds: def load_phoenix_creds() -> PhoenixCreds: - """Local compose Phoenix by default (docker-compose.yml `phoenix` service); - env overrides point the read-back at a deployed Phoenix instead.""" - base_url = (os.getenv("PHOENIX_BASE_URL") or "http://localhost:6006").rstrip("/") - project = os.getenv("PHOENIX_PROJECT_NAME") or "litellm-e2e" - return PhoenixCreds(base_url=base_url, project=project, api_key=os.getenv("PHOENIX_API_KEY")) + """Reads the exact env vars the proxy's arize_phoenix integration reads, so + the read-back always targets the same Phoenix host and project the proxy + ships traces to (locally and on the EKS cluster).""" + endpoint = os.getenv("PHOENIX_COLLECTOR_ENDPOINT") + if not endpoint: + pytest.fail( + "Arize Phoenix e2e requires PHOENIX_COLLECTOR_ENDPOINT; " + "missing credentials is a hard failure, not a skip" + ) + return PhoenixCreds( + base_url=endpoint.rstrip("/").removesuffix("/v1/traces"), + project=os.getenv("PHOENIX_PROJECT_NAME") or "default", + api_key=os.getenv("PHOENIX_API_KEY"), + ) -def phoenix_span_blob(span: PhoenixSpan) -> str: - return json.dumps(span.attributes, default=str) +def phoenix_span_blob(span: dict[str, JsonValue]) -> str: + return json.dumps(span.get("attributes"), default=str) def load_langfuse_creds() -> LangfuseCreds: @@ -323,15 +270,6 @@ def observation_mentions_tool(obs: LangfuseObservation, tool_name: str) -> bool: return tool_name in blob -def observation_has_guardrail(obs: LangfuseObservation, *, guardrail_name: str) -> bool: - blob = json.dumps(obs.metadata, default=str) if obs.metadata is not None else "" - if guardrail_name in blob or "guardrail" in blob.lower(): - return True - if obs.name is not None and "guardrail" in obs.name.lower(): - return True - return False - - @dataclass(frozen=True, slots=True) class LoggingClient: gateway: Gateway @@ -451,46 +389,6 @@ class LoggingClient: f"POST /team/{team_id}/callback must return status=success; got {response.status!r}" ) - def create_tool_permission_guardrail(self, name: str, *, allowed_tool: str) -> str: - """Register a tool_permission guardrail that allows one tool and denies the rest.""" - response = unwrap( - self.gateway.transport.post( - "/guardrails", - headers=self.gateway.transport.master, - json=CreateGuardrailBody( - guardrail=GuardrailSpec( - guardrail_name=name, - litellm_params=GuardrailLitellmParams( - guardrail="tool_permission", - mode="post_call", - default_on=False, - default_action="deny", - on_disallowed_action="block", - rules=[ - { - "id": "allow-named-tool", - "tool_name": allowed_tool, - "decision": "allow", - } - ], - ), - ) - ), - response_type=CreateGuardrailResponse, - ) - ) - guardrail_id = response.guardrail_id - assert guardrail_id, f"create guardrail returned no id: {response!r}" - return guardrail_id - - def delete_guardrail(self, guardrail_id: str) -> None: - _ = self.gateway.transport.delete( - f"/guardrails/{guardrail_id}", - headers=self.gateway.transport.master, - json=NoBody(), - response_type=NoBody, - ) - def create_model(self, model_name: str, litellm_params: LiteLLMParamsBody) -> str: return self.gateway.create_model(model_name, litellm_params) @@ -518,7 +416,6 @@ class LoggingClient: stream: bool = False, tools: list[ChatTool] | None = None, tool_choice: str | None = None, - guardrails: list[str] | None = None, max_tokens: int = 64, ) -> StreamingResponse: body = ChatBody( @@ -528,7 +425,6 @@ class LoggingClient: stream=stream, tools=tools, tool_choice=tool_choice, - guardrails=guardrails, ) if stream: return self.gateway.chat_stream(key, body) @@ -602,41 +498,48 @@ class LoggingClient: creds: PhoenixCreds, *, since: str | None = None, - ) -> list[PhoenixSpan]: + ) -> list[dict[str, JsonValue]]: """Every span Phoenix holds for the project (paged read of the /v1/projects/{project}/spans REST route), newest window bounded by ``since``.""" - pages: list[list[PhoenixSpan]] = [] + spans: list[dict[str, JsonValue]] = [] cursor: str | None = None for _ in range(_PHOENIX_MAX_PAGES): + query = {"limit": "100"} | ({"cursor": cursor} if cursor else {}) | ({"start_time": since} if since else {}) result = get( URL(f"{creds.base_url}/v1/projects/{creds.project}/spans"), headers=creds.auth_headers, - params=PhoenixSpansParams(limit=100, cursor=cursor, start_time=since), - response_type=PhoenixSpansPage, + params=RootModel[dict[str, str]].model_validate(query), + response_type=RootModel[JsonValue], timeout=30.0, ) - match result: - case Success(data=page): - pages.append(page.data) - cursor = page.next_cursor - if cursor is None: - break - case _: - break - return [span for page in pages for span in page] + if not isinstance(result, Success): + break + page = result.data.root + if not isinstance(page, dict): + break + data = page.get("data") + if isinstance(data, list): + spans.extend(span for span in data if isinstance(span, dict)) + next_cursor = page.get("next_cursor") + if not isinstance(next_cursor, str): + break + cursor = next_cursor + return spans def find_phoenix_spans( self, creds: PhoenixCreds, *, marker: str, - span_name: str = PHOENIX_GENERATION_SPAN, since: str | None = None, - ) -> list[PhoenixSpan]: + ) -> list[dict[str, JsonValue]]: + """LLM generation spans carrying ``marker``. Only generation spans hold + message content, so the marker match excludes the request's child spans + (db writes, cache reads) that share the LLM span kind.""" return [ span for span in self.list_phoenix_spans(creds, since=since) - if span.name == span_name and marker in phoenix_span_blob(span) + if span.get("span_kind") == "LLM" and marker in phoenix_span_blob(span) ] def poll_phoenix_spans( @@ -644,18 +547,15 @@ class LoggingClient: creds: PhoenixCreds, *, marker: str, - span_name: str = PHOENIX_GENERATION_SPAN, min_count: int = 1, since: str | None = None, - ) -> list[PhoenixSpan]: - """Poll until at least ``min_count`` spans named ``span_name`` carry - ``marker`` in their attributes, or the poll budget runs out.""" + ) -> list[dict[str, JsonValue]]: + """Poll until at least ``min_count`` generation spans carry ``marker`` + in their attributes, or the poll budget runs out.""" deadline = time.monotonic() + POLL_TIMEOUT - matched: list[PhoenixSpan] = [] + matched: list[dict[str, JsonValue]] = [] while time.monotonic() < deadline: - matched = self.find_phoenix_spans( - creds, marker=marker, span_name=span_name, since=since - ) + matched = self.find_phoenix_spans(creds, marker=marker, since=since) if len(matched) >= min_count: return matched time.sleep(POLL_INTERVAL) @@ -733,7 +633,7 @@ class LoggingClient: key_alias: str, prompt_marker: str, ) -> list[LangfuseObservation]: - """Generation plus any sibling/child observations (guardrail spans, etc.).""" + """Generation plus any sibling/child observations.""" gen = self.poll_langfuse_observation( creds, key_alias=key_alias, prompt_marker=prompt_marker ) @@ -743,4 +643,17 @@ class LoggingClient: def build_logging_client() -> LoggingClient: - return LoggingClient(gateway=build_gateway()) + client = LoggingClient(gateway=build_gateway()) + client.create_model( + "bedrock/us.anthropic.claude-sonnet-5", + LiteLLMParamsBody(model="bedrock/us.anthropic.claude-sonnet-5"), + ) + client.create_model( + "bedrock/us.anthropic.claude-opus-4-8", + LiteLLMParamsBody(model="bedrock/us.anthropic.claude-opus-4-8"), + ) + client.create_model( + "anthropic/claude-sonnet-5", + LiteLLMParamsBody(model="anthropic/claude-sonnet-5", api_key="os.environ/ANTHROPIC_API_KEY"), + ) + return client diff --git a/tests/e2e/logging/test_langfuse_e2e.py b/tests/e2e/logging/test_langfuse_e2e.py index d014b5d8291..002b7849d6c 100644 --- a/tests/e2e/logging/test_langfuse_e2e.py +++ b/tests/e2e/logging/test_langfuse_e2e.py @@ -15,7 +15,7 @@ Dynamic credentials by product surface: - user/key: key metadata.logging with callback_name=langfuse_otel - org: organization + team under it + team callback (no org-level callback API) -Extra success paths assert tool calls and applied guardrails land on the trace. +Extra success paths assert tool calls land on the trace. """ from __future__ import annotations @@ -34,7 +34,6 @@ from logging_client import ( LoggingClient, completion_response_id, costs_agree, - observation_has_guardrail, observation_mentions_tool, observation_spend, ) @@ -285,70 +284,6 @@ class TestLangfuseTeamLogging: scope="team-tools", ) - @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["chat_completions"]) - def test_tool_permission_guardrail_logged( - self, - client: LoggingClient, - resources: ResourceManager, - langfuse_creds: LangfuseCreds, - ) -> None: - """tool_permission post_call guardrail must appear on the Langfuse trace - (StandardLogging guardrail_information -> Langfuse guardrail span).""" - marker = unique_marker() - guardrail_name = f"e2e-lf-tool-perm-{marker}" - guardrail_id = client.create_tool_permission_guardrail( - guardrail_name, allowed_tool="get_weather" - ) - resources.defer(lambda: client.delete_guardrail(guardrail_id)) - - _, key, key_alias = self._team_key( - client, resources, langfuse_creds, models=[DRIVER_MODEL] - ) - prompt_marker = unique_marker() - outcome = client.chat_raw( - key, - DRIVER_MODEL, - f"Use get_weather for Berlin. marker={prompt_marker}", - tools=[WEATHER_TOOL], - tool_choice="required", - guardrails=[guardrail_name], - max_tokens=128, - ) - require_successful_call(outcome) - - observations = client.poll_langfuse_trace_observations( - langfuse_creds, key_alias=key_alias, prompt_marker=prompt_marker - ) - assert observations, ( - f"team+guardrail: no Langfuse observations for key_alias={key_alias!r}" - ) - gen = next( - ( - o - for o in observations - if prompt_marker in _json_blob(o.input) - or key_alias in _json_blob(o.metadata) - or o.name in (f"litellm:{key_alias}", "litellm_request") - ), - observations[0], - ) - _assert_logs_spend( - client, - key=key, - outcome=outcome, - obs_cost=observation_spend(gen), - scope="team-guardrail", - ) - assert any( - observation_has_guardrail(o, guardrail_name=guardrail_name) - or (o.name is not None and "guardrail" in o.name.lower()) - for o in observations - ), ( - f"Langfuse trace must include applied guardrail {guardrail_name!r}; " - f"observation names={[o.name for o in observations]}" - ) - - class TestLangfuseUserKeyLogging: """User-owned key with metadata.logging (key-level dynamic Langfuse credentials). diff --git a/tests/e2e/logging/test_logging_arize_phoenix.py b/tests/e2e/logging/test_logging_arize_phoenix.py index 4aa7af38006..c3fa0b17af3 100644 --- a/tests/e2e/logging/test_logging_arize_phoenix.py +++ b/tests/e2e/logging/test_logging_arize_phoenix.py @@ -1,29 +1,9 @@ """Live e2e: Arize Phoenix trace delivery for the global ``arize_phoenix`` callback. -Registry cells: -- logging.arize_phoenix.success.logs_spend (messages) -- logging.arize_phoenix.stream.logs_spend (messages) - -The proxy ships OTLP spans to the compose ``phoenix`` service -(PHOENIX_COLLECTOR_HTTP_ENDPOINT); read-back is Phoenix's REST route -``/v1/projects/{project}/spans``. Every request must yield exactly one -``litellm_request`` generation span in Phoenix. ArizePhoenixLogger builds its -own TracerProvider precisely so it can coexist with other OTEL-based callbacks -(datadog, otel, arize) without either dropping or double-shipping traces, so -the burst test counts spans per request instead of just probing for presence. - -The burst class simulates Claude Code traffic: consecutive Anthropic-format -``/v1/messages`` calls, mixing streaming and tool use, against a Claude model. - -Known-red (registry fail_before_fix: proven): non-streaming /v1/messages -currently ships two complete traces per request. The @client async wrapper -enqueues async_success_handler AND executor-submits the sync success_handler; -the sync CustomLogger gate keys off _is_sync_litellm_request, whose flag list -(acompletion/aresponses/aembedding/...) has no marker for the async-only -anthropic_messages entrypoint, so log_success_event fires a second time. -Streaming escapes because only the async handler assembles the final stream. -test_burst_ships_exactly_one_trace_per_request stays strict so the fix has a -regression gate. +Every test sends real /v1/messages traffic through the proxy and reads spans +back from Phoenix over its REST API, matched by a unique prompt marker. The +burst tests stay strict (exactly one generation span and one spend row per +request) so duplicate-logging regressions fail loudly. """ from __future__ import annotations @@ -38,17 +18,20 @@ from e2e_http import StreamingResponse, require_successful_call from lifecycle import ResourceManager from logging_client import ( CLAUDE_CODE_TOOLS, - PHOENIX_GENERATION_SPAN, LoggingClient, PhoenixCreds, costs_agree, phoenix_span_blob, ) -from models import AnthropicMessagesResponse, AnthropicToolChoice, SpendLogRow, SpendLogsParams +from models import ( + AnthropicMessagesResponse, + AnthropicToolChoice, + SpendLogRow, + SpendLogsParams, +) pytestmark = pytest.mark.e2e -DRIVER_MODEL = "claude-haiku-4-5" CLAUDE_CODE_BURST = 10 @@ -58,7 +41,12 @@ def _utc_now_iso() -> str: def _fresh_key(client: LoggingClient, resources: ResourceManager) -> str: key = client.key_with_alias( - f"e2e-phoenix-key-{unique_marker()}", models=[DRIVER_MODEL] + f"e2e-phoenix-key-{unique_marker()}", + models=[ + "bedrock/us.anthropic.claude-sonnet-5", + "bedrock/us.anthropic.claude-opus-4-8", + "anthropic/claude-sonnet-5", + ], ) resources.defer(lambda: client.delete_key(key)) return key @@ -74,41 +62,55 @@ class TestArizePhoenixLogging: """The arize_phoenix callback on Anthropic-format /v1/messages traffic.""" @pytest.mark.covers("logging.arize_phoenix.success.logs_spend", exercised_on=["messages"]) + @pytest.mark.parametrize( + "model", + [ + "bedrock/us.anthropic.claude-sonnet-5", + "bedrock/us.anthropic.claude-opus-4-8", + "anthropic/claude-sonnet-5", + ], + ) def test_messages_success_delivers_trace( self, client: LoggingClient, resources: ResourceManager, phoenix_creds: PhoenixCreds, + model: str, ) -> None: key = _fresh_key(client, resources) since = _utc_now_iso() marker = unique_marker() - outcome = client.messages_raw( - key, DRIVER_MODEL, f"reply with one word only {marker}" - ) + outcome = client.messages_raw(key, model, f"What's the capital of France? (run {marker})") require_successful_call(outcome) spans = client.poll_phoenix_spans(phoenix_creds, marker=marker, since=since) assert spans, ( - f"Phoenix never received a {PHOENIX_GENERATION_SPAN} span for " + f"Phoenix never received a generation span for " f"marker={marker!r} in project {phoenix_creds.project!r}" ) @pytest.mark.covers("logging.arize_phoenix.success.logs_spend", exercised_on=["messages"]) + @pytest.mark.parametrize( + "model", + [ + "bedrock/us.anthropic.claude-sonnet-5", + "bedrock/us.anthropic.claude-opus-4-8", + "anthropic/claude-sonnet-5", + ], + ) def test_messages_spend_log_flushed( self, client: LoggingClient, resources: ResourceManager, phoenix_creds: PhoenixCreds, + model: str, ) -> None: """The Phoenix trace and the proxy's own spend row must both land for one request, and their costs must agree with x-litellm-response-cost.""" key = _fresh_key(client, resources) since = _utc_now_iso() marker = unique_marker() - outcome = client.messages_raw( - key, DRIVER_MODEL, f"reply with one word only {marker}" - ) + outcome = client.messages_raw(key, model, f"Who is Lebron James? (run {marker})") require_successful_call(outcome) spans = client.poll_phoenix_spans(phoenix_creds, marker=marker, since=since) @@ -129,18 +131,27 @@ class TestArizePhoenixLogging: ) @pytest.mark.covers("logging.arize_phoenix.success.logs_spend", exercised_on=["messages"]) + @pytest.mark.parametrize( + "model", + [ + "bedrock/us.anthropic.claude-sonnet-5", + "bedrock/us.anthropic.claude-opus-4-8", + "anthropic/claude-sonnet-5", + ], + ) def test_messages_tool_use_on_trace( self, client: LoggingClient, resources: ResourceManager, phoenix_creds: PhoenixCreds, + model: str, ) -> None: key = _fresh_key(client, resources) since = _utc_now_iso() marker = unique_marker() outcome = client.messages_raw( key, - DRIVER_MODEL, + model, f"Read the file /workspace/config-{marker}.yaml", tools=CLAUDE_CODE_TOOLS, tool_choice=AnthropicToolChoice(type="any"), @@ -163,9 +174,8 @@ class TestArizePhoenixLogging: ) -class TestClaudeCodeSimulation: - """A Claude Code-style burst over /v1/messages: many consecutive calls with - streaming and tool use mixed in, then exact per-request trace accounting.""" +class TestAreThereDuplicateTraces: + "Using both anthropic /v1/messages and bedrock (uses passthrough) logger to check if it logs duplicate traces." @pytest.mark.covers("logging.arize_phoenix.stream.logs_spend", exercised_on=["messages"]) def test_burst_ships_exactly_one_trace_per_request( @@ -183,7 +193,7 @@ class TestClaudeCodeSimulation: for i, marker in enumerate(markers): outcome = client.messages_raw( key, - DRIVER_MODEL, + "bedrock/us.anthropic.claude-sonnet-5" if i % 2 == 0 else "anthropic/claude-sonnet-5", f"Claude Code session step: reply with one word only {marker}", stream=i % 2 == 1, tools=CLAUDE_CODE_TOOLS if i % 3 == 0 else None, @@ -202,7 +212,7 @@ class TestClaudeCodeSimulation: spans = [ span for span in client.find_phoenix_spans(phoenix_creds, marker=run, since=since) - if span.status_code == "OK" + if span.get("status_code") != "ERROR" ] counts = { marker: sum(1 for span in spans if marker in phoenix_span_blob(span)) @@ -210,7 +220,7 @@ class TestClaudeCodeSimulation: } wrong = {marker: n for marker, n in counts.items() if n != 1} assert not wrong, ( - f"every request must ship exactly one {PHOENIX_GENERATION_SPAN} span to " + f"every request must ship exactly one generation span to " f"Phoenix; off-by-count markers (duplicates > 1, dropped == 0): {wrong}" ) assert len(spans) == CLAUDE_CODE_BURST, ( @@ -233,7 +243,7 @@ class TestClaudeCodeSimulation: require_successful_call( client.messages_raw( key, - DRIVER_MODEL, + "bedrock/us.anthropic.claude-sonnet-5" if i % 2 == 0 else "anthropic/claude-sonnet-5", f"Claude Code session step: reply with one word only {marker}", stream=i % 2 == 1, ) @@ -258,6 +268,6 @@ class TestClaudeCodeSimulation: spans = client.poll_phoenix_spans( phoenix_creds, marker=run, min_count=CLAUDE_CODE_BURST, since=since ) - assert len([s for s in spans if s.status_code == "OK"]) >= CLAUDE_CODE_BURST, ( + assert len([s for s in spans if s.get("status_code") != "ERROR"]) >= CLAUDE_CODE_BURST, ( f"Phoenix must hold a trace for each of the {CLAUDE_CODE_BURST} spend rows" ) diff --git a/tests/e2e/models.py b/tests/e2e/models.py index f7717a86ea5..f0583cd603c 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -407,14 +407,14 @@ class FileListResponse(BaseModel): data: list[FileEntry] + +# creating fine tuning jobs (deprecated as of July 2026), we list finetuning jobs so tests wont fail on our suite. class FineTuningJobsParams(BaseModel): custom_llm_provider: Literal["openai", "azure"] - class FineTuningJobEntry(BaseModel): id: str - class FineTuningJobsResponse(BaseModel): """GET /fine_tuning/jobs answer; `data` required for the same reason as FileListResponse."""